{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"\n<img src=\"https://markdown.data-ensta.fr/uploads/upload_edb81d24cfac372bb5e8e0d2186b086e.png\" alt=\"drawing\" width=\"1400\"/>\n<center><h1 style=\"color:Blue;font-size:40px;\">Mayo Clinic - STRIP AI</h1> </center>\n<center><h1 style=\"color:Blue;font-size:27px;\">Full VGG11 Train and Submission Pipeline</h1> </center>","metadata":{"_uuid":"5e3e84b04b0843f2d577775ff4495206b10acdd7"}},{"cell_type":"markdown","source":"## Gist & acknowledgements\n\n+ This notebook proposes a simple Keras pipeline to make a quick submission by training a flavor of VGG11.\n+ The use of generators allows for large images to be treated without RAM failure. It also exploits data augmentation.\n+ Thanks for @jirkaborovec's heavy lifting, this challenge's first [downscaled dataset](https://www.kaggle.com/datasets/jirkaborovec/stroke-blood-clot-origin-1k-scale-bg-crop) can be used early in the competition.","metadata":{"_uuid":"efd4a6f6a239307697039c872e3299303ca600ec"}},{"cell_type":"markdown","source":"Imports","metadata":{}},{"cell_type":"code","source":"from numpy.random import seed\n\nimport pandas as pd\nimport numpy as np\n\n\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D\nfrom tensorflow.keras.layers import Dense, Dropout, Flatten, Activation\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau, ModelCheckpoint\nfrom tensorflow.keras.optimizers import Adam\n\nimport os\nimport cv2\n\nfrom sklearn.utils import shuffle\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.model_selection import train_test_split\nimport itertools\nimport shutil\nimport matplotlib.pyplot as plt\n\n%matplotlib inline","metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-12T20:20:58.924311Z","iopub.execute_input":"2022-07-12T20:20:58.924979Z","iopub.status.idle":"2022-07-12T20:21:06.01639Z","shell.execute_reply.started":"2022-07-12T20:20:58.924889Z","shell.execute_reply":"2022-07-12T20:21:06.015016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMAGE_SIZE = 96\nIMAGE_CHANNELS = 3","metadata":{"_uuid":"b7bbdd52c81188b8e9c528b88d9fd0da176bf4bc","execution":{"iopub.status.busy":"2022-07-12T20:21:06.019162Z","iopub.execute_input":"2022-07-12T20:21:06.019869Z","iopub.status.idle":"2022-07-12T20:21:06.031917Z","shell.execute_reply.started":"2022-07-12T20:21:06.01983Z","shell.execute_reply":"2022-07-12T20:21:06.030998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Imported datasets","metadata":{"_uuid":"25802a26e4afba8f6dc42754f7f61da4c8cfea5c"}},{"cell_type":"code","source":"os.listdir(\"../input\")","metadata":{"_uuid":"699bb899bde433ba20fb0d086fa0f33a0f61a250","execution":{"iopub.status.busy":"2022-07-12T20:21:06.033423Z","iopub.execute_input":"2022-07-12T20:21:06.034213Z","iopub.status.idle":"2022-07-12T20:21:06.05183Z","shell.execute_reply.started":"2022-07-12T20:21:06.034177Z","shell.execute_reply":"2022-07-12T20:21:06.050706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### How many images are in each folder?","metadata":{"_uuid":"5a285343c286be191aa827ff9b836b67821176a5"}},{"cell_type":"code","source":"IMAGE_PATH = \"../input/stroke-blood-clot-origin-1k-scale-bg-crop/train_images/\"\nTEST_IMAGE_PATH = \"../input/test-images-downscaled/test_images\"\n\nprint(\n    f\"We train on a scaled down dataset containing {len(os.listdir(IMAGE_PATH))} images\"\n)\nprint(\n    f\"We will test on a downscaled test set containing {len(os.listdir(TEST_IMAGE_PATH))} images\"\n)\n# print(len(os.listdir('../input/test')))","metadata":{"_uuid":"54461212efed65ac377369a468c80e7d708010f4","execution":{"iopub.status.busy":"2022-07-12T20:21:06.055228Z","iopub.execute_input":"2022-07-12T20:21:06.055775Z","iopub.status.idle":"2022-07-12T20:21:06.32364Z","shell.execute_reply.started":"2022-07-12T20:21:06.055681Z","shell.execute_reply":"2022-07-12T20:21:06.322673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Create a Dataframe containing all images","metadata":{"_uuid":"b90854e07d495d9f945a0e40189fd32a0c32bff5"}},{"cell_type":"code","source":"df_data = pd.read_csv(\"../input/mayo-clinic-strip-ai/train.csv\")\n\n# remove pictures that did not make it to the downscaled dataset:\ndf_data = df_data.drop(df_data.query(f\"image_id=='b894f4_0'\").index)\ndf_data = df_data.drop(df_data.query(f\"image_id=='6baf51_0'\").index)\n\nprint(df_data.shape)","metadata":{"_uuid":"e9c9f40ffab35044641b0dc7d9b18609af1aa25e","execution":{"iopub.status.busy":"2022-07-12T20:21:06.325138Z","iopub.execute_input":"2022-07-12T20:21:06.325484Z","iopub.status.idle":"2022-07-12T20:21:06.363383Z","shell.execute_reply.started":"2022-07-12T20:21:06.325448Z","shell.execute_reply":"2022-07-12T20:21:06.362441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:21:06.366665Z","iopub.execute_input":"2022-07-12T20:21:06.367512Z","iopub.status.idle":"2022-07-12T20:21:06.383947Z","shell.execute_reply.started":"2022-07-12T20:21:06.367479Z","shell.execute_reply":"2022-07-12T20:21:06.383038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Check the class distribution","metadata":{"_uuid":"cfbd53b7f8ea1929952ffed6221b380012618e32"}},{"cell_type":"code","source":"CE_count, LAA_count = df_data[\"label\"].value_counts()\nprint(f\"There are {CE_count} CE samples and {LAA_count} LAA samples\")","metadata":{"_uuid":"e18560bf69d3dfc0c4772e7c79bb119fd2eb634b","execution":{"iopub.status.busy":"2022-07-12T20:21:06.385297Z","iopub.execute_input":"2022-07-12T20:21:06.386183Z","iopub.status.idle":"2022-07-12T20:21:06.393113Z","shell.execute_reply.started":"2022-07-12T20:21:06.386147Z","shell.execute_reply":"2022-07-12T20:21:06.391971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Display a random sample of train images  by class","metadata":{"_uuid":"6efe2e5de99c4bf92079b1a7d0b892d30fc9d518"}},{"cell_type":"code","source":"# source: https://www.kaggle.com/gpreda/honey-bee-subspecies-classification\n\n\ndef draw_category_images(col_name, figure_cols, df, IMAGE_PATH):\n    categories = (df.groupby([col_name])[col_name].nunique()).index\n    f, ax = plt.subplots(\n        nrows=len(categories),\n        ncols=figure_cols,\n        figsize=(4 * figure_cols, 4 * len(categories)),\n    )  # adjust size here\n    # draw a number of images for each location\n    for i, cat in enumerate(categories):\n        sample = df[df[col_name] == cat].sample(\n            figure_cols\n        )  # figure_cols is also the sample size\n        for j in range(0, figure_cols):\n            file = IMAGE_PATH + sample.iloc[j][\"image_id\"] + \".png\"\n            im = cv2.imread(file)\n            ax[i, j].imshow(im, resample=True, cmap=\"gray\")\n            ax[i, j].set_title(cat, fontsize=16)\n    plt.tight_layout()\n    plt.show()","metadata":{"_kg_hide-input":true,"_uuid":"1c5143f227da4262eafce8cf0210a02c8072fb8e","execution":{"iopub.status.busy":"2022-07-12T20:21:06.394692Z","iopub.execute_input":"2022-07-12T20:21:06.39547Z","iopub.status.idle":"2022-07-12T20:21:06.406152Z","shell.execute_reply.started":"2022-07-12T20:21:06.39543Z","shell.execute_reply":"2022-07-12T20:21:06.405116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"draw_category_images(\"label\", 4, df_data, IMAGE_PATH)","metadata":{"_uuid":"bd38bcfb5839975e4fee9e70b93d42c29c1b5d2e","execution":{"iopub.status.busy":"2022-07-12T20:21:06.408467Z","iopub.execute_input":"2022-07-12T20:21:06.409484Z","iopub.status.idle":"2022-07-12T20:21:09.126624Z","shell.execute_reply.started":"2022-07-12T20:21:06.409451Z","shell.execute_reply":"2022-07-12T20:21:09.124616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Create the Train and Val Sets","metadata":{"_uuid":"1da4226777aefe65b1bb3430208ea91ea7ca7d9a"}},{"cell_type":"markdown","source":"#### Balance the target distribution\nWe will reduce the number of samples in class CE.","metadata":{"_uuid":"c1150500d2772b7f36cdaa5aa5fd7f0fb4a72628"}},{"cell_type":"code","source":"# take a random sample of class 0 with size equal to num samples in class 1\ndf_0 = df_data[df_data[\"label\"] == \"CE\"].sample(LAA_count, random_state=101)\n# filter out class 1\ndf_1 = df_data[df_data[\"label\"] == \"LAA\"].sample(LAA_count, random_state=101)\n\n# concat the dataframes\ndf_data = pd.concat([df_0, df_1], axis=0).reset_index(drop=True)\n# shuffle\ndf_data = shuffle(df_data)\n\ndf_data[\"label\"].value_counts()","metadata":{"_uuid":"270fc18640b552ecc3cb0e1dd3036441db7a4a2b","execution":{"iopub.status.busy":"2022-07-12T20:21:09.130781Z","iopub.execute_input":"2022-07-12T20:21:09.131526Z","iopub.status.idle":"2022-07-12T20:21:09.152487Z","shell.execute_reply.started":"2022-07-12T20:21:09.13149Z","shell.execute_reply":"2022-07-12T20:21:09.151476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_data.head()","metadata":{"_uuid":"a166dec3ef84c66ad9cd815b63fc1a753df2eb76","execution":{"iopub.status.busy":"2022-07-12T20:21:09.154125Z","iopub.execute_input":"2022-07-12T20:21:09.154473Z","iopub.status.idle":"2022-07-12T20:21:09.166444Z","shell.execute_reply.started":"2022-07-12T20:21:09.154438Z","shell.execute_reply":"2022-07-12T20:21:09.165403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_test_split\n\n# stratify=y creates a balanced validation set.\ny = df_data[\"label\"]\n\ndf_train, df_val = train_test_split(\n    df_data, test_size=0.10, random_state=101, stratify=y\n)\n\nprint(df_train.shape)\nprint(df_val.shape)","metadata":{"_uuid":"15ba9792e6a370b7560330af15b3cfe21185c1cb","execution":{"iopub.status.busy":"2022-07-12T20:21:09.168164Z","iopub.execute_input":"2022-07-12T20:21:09.168906Z","iopub.status.idle":"2022-07-12T20:21:09.178235Z","shell.execute_reply.started":"2022-07-12T20:21:09.168872Z","shell.execute_reply":"2022-07-12T20:21:09.177046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train[\"label\"].value_counts()","metadata":{"_uuid":"7de70d915a5f1d2599725e00bdb3b9103d947883","execution":{"iopub.status.busy":"2022-07-12T20:21:09.180135Z","iopub.execute_input":"2022-07-12T20:21:09.180498Z","iopub.status.idle":"2022-07-12T20:21:09.193232Z","shell.execute_reply.started":"2022-07-12T20:21:09.180462Z","shell.execute_reply":"2022-07-12T20:21:09.191957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_val[\"label\"].value_counts()","metadata":{"_uuid":"392c0eea00be8e43a6e55438d1458650e842030b","execution":{"iopub.status.busy":"2022-07-12T20:21:09.195545Z","iopub.execute_input":"2022-07-12T20:21:09.195905Z","iopub.status.idle":"2022-07-12T20:21:09.205289Z","shell.execute_reply.started":"2022-07-12T20:21:09.19587Z","shell.execute_reply":"2022-07-12T20:21:09.204221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Create a Directory Structure","metadata":{"_uuid":"ba17dd34b75367fc61df6634d51dac94c3ab4951"}},{"cell_type":"code","source":"# Create a new directory\nbase_dir = \"base_dir\"\nos.makedirs(base_dir, exist_ok=True)\n\n\n# [CREATE FOLDERS INSIDE THE BASE DIRECTORY]\n\n# now we create 2 folders inside 'base_dir':\n\n# train_dir\n# a_CE\n# b_LAA\n\n# val_dir\n# a_CE\n# b_LAA\n\n# create a path to 'base_dir' to which we will join the names of the new folders\n# train_dir\ntrain_dir = os.path.join(base_dir, \"train_dir\")\nos.makedirs(train_dir, exist_ok=True)\n\n# val_dir\nval_dir = os.path.join(base_dir, \"val_dir\")\nos.makedirs(val_dir, exist_ok=True)\n\n\n# [CREATE FOLDERS INSIDE THE TRAIN AND VALIDATION FOLDERS]\n# Inside each folder we create seperate folders for each class\n\n# create new folders inside train_dir\nCE = os.path.join(train_dir, \"a_CE\")\nos.makedirs(CE, exist_ok=True)\nLAA = os.path.join(train_dir, \"b_LAA\")\nos.makedirs(LAA, exist_ok=True)\n\n\n# create new folders inside val_dir\nCE = os.path.join(val_dir, \"a_CE\")\nos.makedirs(CE, exist_ok=True)\nLAA = os.path.join(val_dir, \"b_LAA\")\nos.makedirs(LAA, exist_ok=True)","metadata":{"_uuid":"ff8acc2e92a1b1b5002d6e1bf9a1180c3256f19d","execution":{"iopub.status.busy":"2022-07-12T20:21:09.207113Z","iopub.execute_input":"2022-07-12T20:21:09.207761Z","iopub.status.idle":"2022-07-12T20:21:09.217854Z","shell.execute_reply.started":"2022-07-12T20:21:09.207727Z","shell.execute_reply":"2022-07-12T20:21:09.216992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check that the folders have been created\nos.listdir(\"base_dir/train_dir\")","metadata":{"_uuid":"03ca5d4b8b027c2712d7096314d3a79ef829b23c","execution":{"iopub.status.busy":"2022-07-12T20:21:09.219185Z","iopub.execute_input":"2022-07-12T20:21:09.220277Z","iopub.status.idle":"2022-07-12T20:21:09.233032Z","shell.execute_reply.started":"2022-07-12T20:21:09.22018Z","shell.execute_reply":"2022-07-12T20:21:09.23204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Transfer the images into the folders","metadata":{"_uuid":"6a2e56340ba18f3b63c1b129fd995fecfadaa21d"}},{"cell_type":"code","source":"# Set the id as the index in df_data\ndf_data.set_index(\"image_id\", inplace=True)","metadata":{"_uuid":"e84c8a9642b030094b1888af3299063f883112a6","execution":{"iopub.status.busy":"2022-07-12T20:21:09.235804Z","iopub.execute_input":"2022-07-12T20:21:09.236371Z","iopub.status.idle":"2022-07-12T20:21:09.243013Z","shell.execute_reply.started":"2022-07-12T20:21:09.236346Z","shell.execute_reply":"2022-07-12T20:21:09.242131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get a list of train and val images\ntrain_list = list(df_train[\"image_id\"])\nval_list = list(df_val[\"image_id\"])\n\n# Transfer the train images\nfor image in train_list:\n\n    # the id in the csv file does not have the .png extension therefore we add it here\n    fname = image + \".png\"\n    # get the label for a certain image\n    target = df_data.loc[image, \"label\"]\n\n    # these must match the folder names\n    if target == \"CE\":\n        label = \"a_CE\"\n    if target == \"LAA\":\n        label = \"b_LAA\"\n\n    # source path to image\n    src = os.path.join(IMAGE_PATH, fname)\n    # destination path to image\n    dst = os.path.join(train_dir, label, fname)\n    # copy the image from the source to the destination\n    shutil.copyfile(src, dst)\n\n\n# Transfer the val images\n\nfor image in val_list:\n\n    # the id in the csv file does not have the .png extension therefore we add it here\n    fname = image + \".png\"\n    # get the label for a certain image\n    target = df_data.loc[image, \"label\"]\n\n    # these must match the folder names\n    if target == \"CE\":\n        label = \"a_CE\"\n    if target == \"LAA\":\n        label = \"b_LAA\"\n\n    # source path to image\n    src = os.path.join(IMAGE_PATH, fname)\n    # destination path to image\n    dst = os.path.join(val_dir, label, fname)\n    # copy the image from the source to the destination\n    shutil.copyfile(src, dst)","metadata":{"_uuid":"afb8969a9ee75c13bddc808a4bcc326611baaaaf","execution":{"iopub.status.busy":"2022-07-12T20:21:09.244574Z","iopub.execute_input":"2022-07-12T20:21:09.245457Z","iopub.status.idle":"2022-07-12T20:21:16.982045Z","shell.execute_reply.started":"2022-07-12T20:21:09.245419Z","shell.execute_reply":"2022-07-12T20:21:16.981041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check how many train images we have in each folder\n\nprint(len(os.listdir(\"base_dir/train_dir/a_CE\")))\nprint(len(os.listdir(\"base_dir/train_dir/b_LAA\")))","metadata":{"_uuid":"71532bfc32608289b1f773ffdbc8a7cea1bfb94c","execution":{"iopub.status.busy":"2022-07-12T20:21:16.984194Z","iopub.execute_input":"2022-07-12T20:21:16.984875Z","iopub.status.idle":"2022-07-12T20:21:16.991751Z","shell.execute_reply.started":"2022-07-12T20:21:16.984834Z","shell.execute_reply":"2022-07-12T20:21:16.990632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check how many val images we have in each folder\n\nprint(len(os.listdir(\"base_dir/val_dir/a_CE\")))\nprint(len(os.listdir(\"base_dir/val_dir/b_LAA\")))","metadata":{"_uuid":"897e9df543bb65b47bb00019dc681125ca08ee5d","execution":{"iopub.status.busy":"2022-07-12T20:21:16.993555Z","iopub.execute_input":"2022-07-12T20:21:16.993902Z","iopub.status.idle":"2022-07-12T20:21:17.002161Z","shell.execute_reply.started":"2022-07-12T20:21:16.993868Z","shell.execute_reply":"2022-07-12T20:21:17.001046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Set Up the Generators","metadata":{"_uuid":"f8dce940ee8a7a42aacb062e4c6b5a4a54dba58f"}},{"cell_type":"code","source":"train_path = \"base_dir/train_dir\"\nvalid_path = \"base_dir/val_dir\"\ntest_path = \"../input/test\"\n\nnum_train_samples = len(df_train)\nnum_val_samples = len(df_val)\ntrain_batch_size = 5\nval_batch_size = 5\n\n\ntrain_steps = np.ceil(num_train_samples / train_batch_size)\nval_steps = np.ceil(num_val_samples / val_batch_size)","metadata":{"_uuid":"ef4fe7be09f11ff4badfd22d5fd5e03f8521ed58","execution":{"iopub.status.busy":"2022-07-12T20:21:17.003605Z","iopub.execute_input":"2022-07-12T20:21:17.004161Z","iopub.status.idle":"2022-07-12T20:21:17.011012Z","shell.execute_reply.started":"2022-07-12T20:21:17.004124Z","shell.execute_reply":"2022-07-12T20:21:17.009821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"datagen = ImageDataGenerator(\n    rescale=1.0 / 255,\n    rotation_range=10,  # rotation\n    width_shift_range=0.2,  # horizontal shift\n    height_shift_range=0.2,  # vertical shift\n    zoom_range=0.2,  # zoom\n    horizontal_flip=True,  # horizontal flip\n    brightness_range=[0.2, 1.2],\n)  # brightness)\n\ntrain_gen = datagen.flow_from_directory(\n    train_path,\n    target_size=(IMAGE_SIZE, IMAGE_SIZE),\n    batch_size=train_batch_size,\n    color_mode=\"rgb\",  # for coloured images\n    class_mode=\"categorical\",\n)\n\nval_gen = datagen.flow_from_directory(\n    valid_path,\n    target_size=(IMAGE_SIZE, IMAGE_SIZE),\n    batch_size=val_batch_size,\n    color_mode=\"rgb\",  # for coloured images\n    class_mode=\"categorical\",\n)\n\n# Note: shuffle=False causes the test dataset to not be shuffled\ntest_gen = datagen.flow_from_directory(\n    valid_path,\n    target_size=(IMAGE_SIZE, IMAGE_SIZE),\n    batch_size=1,\n    class_mode=\"categorical\",\n    color_mode=\"rgb\",  # for coloured images\n    shuffle=False,\n)","metadata":{"_uuid":"68fbd9d5fbb80859a82f94a12e335ce05a93bd51","execution":{"iopub.status.busy":"2022-07-12T20:21:17.012747Z","iopub.execute_input":"2022-07-12T20:21:17.013129Z","iopub.status.idle":"2022-07-12T20:21:17.331611Z","shell.execute_reply.started":"2022-07-12T20:21:17.013096Z","shell.execute_reply":"2022-07-12T20:21:17.330641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Create the Model Architecture¶","metadata":{"_uuid":"79da4d0a66a90cffe40580a596dd4d0e2bc45a9b"}},{"cell_type":"markdown","source":"I've used the CNN architecture presented by @fmarazzi in this kernel:<br>\nhttps://www.kaggle.com/fmarazzi/baseline-keras-cnn-roc-fast-5min-0-8253-lb","metadata":{"_uuid":"5494ee67d51e8fbd80585ffae5f2f3d86694dab2"}},{"cell_type":"code","source":"kernel_size = (3, 3)\npool_size = (2, 2)\nfirst_filters = 32\nsecond_filters = 64\nthird_filters = 128\n\ndropout_conv = 0.3\ndropout_dense = 0.3\n\n\nmodel = Sequential()\nmodel.add(\n    Conv2D(first_filters, kernel_size, activation=\"relu\", input_shape=(96, 96, 3))\n)\nmodel.add(Conv2D(first_filters, kernel_size, activation=\"relu\"))\nmodel.add(Conv2D(first_filters, kernel_size, activation=\"relu\"))\nmodel.add(MaxPooling2D(pool_size=pool_size))\nmodel.add(Dropout(dropout_conv))\n\nmodel.add(Conv2D(second_filters, kernel_size, activation=\"relu\"))\nmodel.add(Conv2D(second_filters, kernel_size, activation=\"relu\"))\nmodel.add(Conv2D(second_filters, kernel_size, activation=\"relu\"))\nmodel.add(MaxPooling2D(pool_size=pool_size))\nmodel.add(Dropout(dropout_conv))\n\nmodel.add(Conv2D(third_filters, kernel_size, activation=\"relu\"))\nmodel.add(Conv2D(third_filters, kernel_size, activation=\"relu\"))\nmodel.add(Conv2D(third_filters, kernel_size, activation=\"relu\"))\nmodel.add(MaxPooling2D(pool_size=pool_size))\nmodel.add(Dropout(dropout_conv))\n\nmodel.add(Flatten())\nmodel.add(Dense(256, activation=\"relu\"))\nmodel.add(Dropout(dropout_dense))\nmodel.add(Dense(2, activation=\"softmax\"))\n\nmodel.summary()","metadata":{"_uuid":"b9835ea0fd0bca54138904895c39d38227a70c22","execution":{"iopub.status.busy":"2022-07-12T20:21:17.333936Z","iopub.execute_input":"2022-07-12T20:21:17.334811Z","iopub.status.idle":"2022-07-12T20:21:20.281115Z","shell.execute_reply.started":"2022-07-12T20:21:17.334771Z","shell.execute_reply":"2022-07-12T20:21:20.280138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Train the Model","metadata":{"_uuid":"75cfc4fcb8dd3408d1c4fcf8cd85e0e2f5b611d7"}},{"cell_type":"code","source":"model.compile(Adam(lr=0.0001), loss=\"binary_crossentropy\", metrics=[\"accuracy\"])","metadata":{"_uuid":"9de9715f49a63b55775b10abd2f461b395e23b5d","execution":{"iopub.status.busy":"2022-07-12T20:21:20.282695Z","iopub.execute_input":"2022-07-12T20:21:20.283361Z","iopub.status.idle":"2022-07-12T20:21:20.298304Z","shell.execute_reply.started":"2022-07-12T20:21:20.283324Z","shell.execute_reply":"2022-07-12T20:21:20.297296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get the labels that are associated with each index\nprint(val_gen.class_indices)","metadata":{"_uuid":"227d84a4f44c0b7256855c06ba04dabd58d89d84","execution":{"iopub.status.busy":"2022-07-12T20:21:20.300008Z","iopub.execute_input":"2022-07-12T20:21:20.300348Z","iopub.status.idle":"2022-07-12T20:21:20.307402Z","shell.execute_reply.started":"2022-07-12T20:21:20.300313Z","shell.execute_reply":"2022-07-12T20:21:20.306044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filepath = \"base_dir/model.h5\"\ncheckpoint = ModelCheckpoint(\n    filepath, monitor=\"val_acc\", verbose=1, save_best_only=True, mode=\"max\"\n)\n\nreduce_lr = ReduceLROnPlateau(\n    monitor=\"val_acc\", factor=0.5, patience=2, verbose=1, mode=\"max\", min_lr=0.00001\n)\n\ncallbacks_list = [checkpoint, reduce_lr]\n\nhistory = model.fit_generator(\n    train_gen,\n    steps_per_epoch=train_steps,\n    validation_data=val_gen,\n    validation_steps=val_steps,\n    epochs=1,\n    verbose=1,\n    callbacks=checkpoint,\n)","metadata":{"scrolled":true,"_uuid":"a746769db61563f226288eba9aa8a6584b9e8e0b","execution":{"iopub.status.busy":"2022-07-12T20:21:20.309181Z","iopub.execute_input":"2022-07-12T20:21:20.309609Z","iopub.status.idle":"2022-07-12T20:22:02.325619Z","shell.execute_reply.started":"2022-07-12T20:21:20.309573Z","shell.execute_reply":"2022-07-12T20:22:02.324618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Evaluate the model using the val set","metadata":{"_uuid":"fa15a8afda3593973726e9087cbd98073041c908"}},{"cell_type":"code","source":"# get the metric names so we can use evaulate_generator\nmodel.metrics_names","metadata":{"_uuid":"70104420ec7f400cd06203f875dbeba30f4d8a96","execution":{"iopub.status.busy":"2022-07-12T20:22:02.327117Z","iopub.execute_input":"2022-07-12T20:22:02.327577Z","iopub.status.idle":"2022-07-12T20:22:02.33541Z","shell.execute_reply.started":"2022-07-12T20:22:02.32754Z","shell.execute_reply":"2022-07-12T20:22:02.334504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Here the best epoch will be used.\n\n# model.load_weights('base_dir/model.h5')\n\nval_loss, val_acc = model.evaluate(test_gen, steps=len(df_val))\n\nprint(\"val_loss:\", val_loss)\nprint(\"val_acc:\", val_acc)","metadata":{"_uuid":"428bdf5b24ff8cef35012205c3f2eb37006fc9e9","execution":{"iopub.status.busy":"2022-07-12T20:22:02.342345Z","iopub.execute_input":"2022-07-12T20:22:02.343283Z","iopub.status.idle":"2022-07-12T20:22:04.338907Z","shell.execute_reply.started":"2022-07-12T20:22:02.343238Z","shell.execute_reply":"2022-07-12T20:22:04.3379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.listdir(\"base_dir\")","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:22:04.340064Z","iopub.execute_input":"2022-07-12T20:22:04.340331Z","iopub.status.idle":"2022-07-12T20:22:04.348099Z","shell.execute_reply.started":"2022-07-12T20:22:04.340306Z","shell.execute_reply":"2022-07-12T20:22:04.347048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Plot the Training Curves","metadata":{"_uuid":"93556c9e4b6a188cf9cb67a6519c9bc365c60caf"}},{"cell_type":"code","source":"# display the loss and accuracy curves\n\nimport matplotlib.pyplot as plt\n\nacc = history.history[\"accuracy\"]\nval_acc = history.history[\"val_accuracy\"]\nloss = history.history[\"loss\"]\nval_loss = history.history[\"val_loss\"]\n\nepochs = range(1, len(acc) + 1)\n\nplt.plot(epochs, loss, \"bo\", label=\"Training loss\")\nplt.plot(epochs, val_loss, \"b\", label=\"Validation loss\")\nplt.title(\"Training and validation loss\")\nplt.legend()\nplt.figure()\n\nplt.plot(epochs, acc, \"bo\", label=\"Training acc\")\nplt.plot(epochs, val_acc, \"b\", label=\"Validation acc\")\nplt.title(\"Training and validation accuracy\")\nplt.legend()\nplt.figure()","metadata":{"_uuid":"385da8ba94a1079d17909790716b295fc2737584","_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-12T20:22:04.350104Z","iopub.execute_input":"2022-07-12T20:22:04.350887Z","iopub.status.idle":"2022-07-12T20:22:04.720506Z","shell.execute_reply.started":"2022-07-12T20:22:04.350849Z","shell.execute_reply":"2022-07-12T20:22:04.719612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Make a prediction on the val set\nWe need these predictions to calculate the AUC score, print the Confusion Matrix and calculate the F1 score.","metadata":{"_uuid":"5636e76f23202dd1f2a27ace25e15e09619a5e4e"}},{"cell_type":"code","source":"# make a prediction\npredictions = model.predict(test_gen, steps=len(df_val), verbose=1)","metadata":{"_uuid":"652d9d6aa51dc1818d1c5171212d10e141ad7de9","execution":{"iopub.status.busy":"2022-07-12T20:22:04.721893Z","iopub.execute_input":"2022-07-12T20:22:04.722358Z","iopub.status.idle":"2022-07-12T20:22:06.757279Z","shell.execute_reply.started":"2022-07-12T20:22:04.722319Z","shell.execute_reply":"2022-07-12T20:22:06.756372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions.shape","metadata":{"_uuid":"edf1866df4638ded26de2e0e3d2ba0f5e00e1ace","execution":{"iopub.status.busy":"2022-07-12T20:22:06.758723Z","iopub.execute_input":"2022-07-12T20:22:06.759179Z","iopub.status.idle":"2022-07-12T20:22:06.76554Z","shell.execute_reply.started":"2022-07-12T20:22:06.759143Z","shell.execute_reply":"2022-07-12T20:22:06.764538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### A note on Keras class index values\n\nKeras assigns it's own index value (here 0 and 1) to the classes.\nIt infers the classes based on the folder structure.<br>\nImportant: These index values may not match the index values we were given in the **train_labels.csv** file.\n\nI've used 'a' and 'b' folder name pre-fixes to get keras to assign index values to match what\nwas in the train_labels.csv file - I guessed that keras is assigning the index value based on\nfolder name alphabetical order.","metadata":{"_uuid":"3aabee812241fd8f1ca46a527bf50567e2a58241"}},{"cell_type":"code","source":"# This is how to check what index keras has internally assigned to each class.\ntest_gen.class_indices","metadata":{"_uuid":"dc71f69944e7db83329417c5265a5bc31f9c4fc3","execution":{"iopub.status.busy":"2022-07-12T20:22:06.766956Z","iopub.execute_input":"2022-07-12T20:22:06.767972Z","iopub.status.idle":"2022-07-12T20:22:06.777709Z","shell.execute_reply.started":"2022-07-12T20:22:06.767935Z","shell.execute_reply":"2022-07-12T20:22:06.776634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Put the predictions into a dataframe.\n# The columns need to be oredered to match the output of the previous cell\n\ndf_preds = pd.DataFrame(predictions, columns=[\"CE\", \"LAA\"])\n\ndf_preds.head()","metadata":{"_uuid":"4a6709d73969f7fd597128223b110be077f84edb","execution":{"iopub.status.busy":"2022-07-12T20:22:06.779171Z","iopub.execute_input":"2022-07-12T20:22:06.779533Z","iopub.status.idle":"2022-07-12T20:22:06.794312Z","shell.execute_reply.started":"2022-07-12T20:22:06.779497Z","shell.execute_reply":"2022-07-12T20:22:06.793288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### MAKE A TEST SET PREDICTION","metadata":{"_uuid":"9a10b51355f1aa8ae1f7148703791dc71f5f7e3e"}},{"cell_type":"code","source":"# Delete base_dir and it's sub folders to free up disk space.\n\nshutil.rmtree(\"base_dir\")","metadata":{"_uuid":"62104c1c149e12e2c6b525283466a0c8e835a020","execution":{"iopub.status.busy":"2022-07-12T20:22:06.795878Z","iopub.execute_input":"2022-07-12T20:22:06.796382Z","iopub.status.idle":"2022-07-12T20:22:06.882645Z","shell.execute_reply.started":"2022-07-12T20:22:06.796342Z","shell.execute_reply":"2022-07-12T20:22:06.881686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# [CREATE A TEST FOLDER DIRECTORY STRUCTURE]\n\n# We will be feeding test images from a folder into predict_generator().\n# Keras requires that the path should point to a folder containing images and not\n# to the images themselves. That is why we are creating a folder (test_images)\n# inside another folder (test_dir).\n\n# test_dir\n# test_images\n\n# create test_dir\ntest_dir = \"test_dir\"\nos.makedirs(test_dir, exist_ok=True)\n\n# create test_images inside test_dir\ntest_images = os.path.join(test_dir, \"test_images\")\nos.makedirs(test_images, exist_ok=True)","metadata":{"_uuid":"c91e56e9c45caa655f7e45ec98ae0c492ce5358e","execution":{"iopub.status.busy":"2022-07-12T20:22:06.885007Z","iopub.execute_input":"2022-07-12T20:22:06.885571Z","iopub.status.idle":"2022-07-12T20:22:06.891528Z","shell.execute_reply.started":"2022-07-12T20:22:06.885531Z","shell.execute_reply":"2022-07-12T20:22:06.890433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check that the directory we created exists\nos.listdir(\"test_dir\")","metadata":{"_uuid":"9a72c5c03b0d821762a4c61f2cdd7767b7b123fc","execution":{"iopub.status.busy":"2022-07-12T20:22:06.892955Z","iopub.execute_input":"2022-07-12T20:22:06.893357Z","iopub.status.idle":"2022-07-12T20:22:06.905918Z","shell.execute_reply.started":"2022-07-12T20:22:06.893323Z","shell.execute_reply":"2022-07-12T20:22:06.904568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Transfer the test images into image_dir\n\ntest_list = os.listdir(\"../input/test-images-downscaled/test_images\")\n\nfor image in test_list:\n\n    fname = image\n\n    # source path to image\n    src = os.path.join(\"../input/test-images-downscaled/test_images\", fname)\n    # destination path to image\n    dst = os.path.join(test_images, fname)\n    # copy the image from the source to the destination\n    shutil.copyfile(src, dst)","metadata":{"_uuid":"f98c94ea8355a6d56d3f4a2791934b3365ed7865","execution":{"iopub.status.busy":"2022-07-12T20:22:06.90729Z","iopub.execute_input":"2022-07-12T20:22:06.907674Z","iopub.status.idle":"2022-07-12T20:22:07.23689Z","shell.execute_reply.started":"2022-07-12T20:22:06.907603Z","shell.execute_reply":"2022-07-12T20:22:07.235896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check that the images are now in the test_images\n\nlen(os.listdir(\"test_dir/test_images\"))","metadata":{"_uuid":"22a46762dbfe32843d5e37c2204778f04fd55061","execution":{"iopub.status.busy":"2022-07-12T20:22:07.240021Z","iopub.execute_input":"2022-07-12T20:22:07.240323Z","iopub.status.idle":"2022-07-12T20:22:07.246933Z","shell.execute_reply.started":"2022-07-12T20:22:07.240297Z","shell.execute_reply":"2022-07-12T20:22:07.245868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Set up the generator","metadata":{"_uuid":"aa456427dabaaa4c01b0698e3af672c195a79de6"}},{"cell_type":"code","source":"test_path = \"test_dir\"\n\n# Here we change the path to point to the test_images folder.\ntest_gen = datagen.flow_from_directory(\n    test_path,\n    target_size=(IMAGE_SIZE, IMAGE_SIZE),\n    batch_size=1,\n    class_mode=\"categorical\",\n    shuffle=False,\n)","metadata":{"_uuid":"6529facf2e4b60f4962fb0d08528e9f1ecd895a7","execution":{"iopub.status.busy":"2022-07-12T20:22:07.24872Z","iopub.execute_input":"2022-07-12T20:22:07.249421Z","iopub.status.idle":"2022-07-12T20:22:07.359125Z","shell.execute_reply.started":"2022-07-12T20:22:07.249385Z","shell.execute_reply":"2022-07-12T20:22:07.358194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Make a prediction on the test images","metadata":{"_uuid":"736628d49204bd0b868903a00068e0c09261ab05"}},{"cell_type":"code","source":"num_test_images = 5\n\npredictions = model.predict(test_gen, steps=num_test_images, verbose=1)","metadata":{"_uuid":"150f61e5b959bcb2589d330652e3b1989caa35c4","execution":{"iopub.status.busy":"2022-07-12T20:22:07.360925Z","iopub.execute_input":"2022-07-12T20:22:07.361576Z","iopub.status.idle":"2022-07-12T20:22:07.716366Z","shell.execute_reply.started":"2022-07-12T20:22:07.361538Z","shell.execute_reply":"2022-07-12T20:22:07.715456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Are the number of predictions correct?\n\nlen(predictions)","metadata":{"_uuid":"d4e4e9c9a88bf01d151a1d706e314cb243e8f6b5","execution":{"iopub.status.busy":"2022-07-12T20:22:07.717825Z","iopub.execute_input":"2022-07-12T20:22:07.718891Z","iopub.status.idle":"2022-07-12T20:22:07.725826Z","shell.execute_reply.started":"2022-07-12T20:22:07.718852Z","shell.execute_reply":"2022-07-12T20:22:07.724527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Put the predictions into a dataframe\n\ndf_preds = pd.DataFrame(predictions, columns=[\"CE\", \"LAA\"])\n\ndf_preds.head()","metadata":{"_uuid":"1346b4e4f22f053a5a70bef1f5edd1c2aee48404","execution":{"iopub.status.busy":"2022-07-12T20:22:07.727245Z","iopub.execute_input":"2022-07-12T20:22:07.728409Z","iopub.status.idle":"2022-07-12T20:22:07.741563Z","shell.execute_reply.started":"2022-07-12T20:22:07.728369Z","shell.execute_reply":"2022-07-12T20:22:07.740632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This outputs the file names in the sequence in which\n# the generator processed the test images.\ntest_filenames = test_gen.filenames\n\n# add the filenames to the dataframe\ndf_preds[\"file_names\"] = test_filenames\n\ndf_preds.head()","metadata":{"_uuid":"ff5fb65c49b505b405c726573dd9285664641bc1","execution":{"iopub.status.busy":"2022-07-12T20:22:07.743291Z","iopub.execute_input":"2022-07-12T20:22:07.743781Z","iopub.status.idle":"2022-07-12T20:22:07.75971Z","shell.execute_reply.started":"2022-07-12T20:22:07.743731Z","shell.execute_reply":"2022-07-12T20:22:07.758393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create ID column\ndef extract_id(x):\n\n    # split into a list\n    a = x.split(\"/\")\n    # split into a list\n    b = a[1].split(\".\")\n    extracted_id = b[0]\n\n    return extracted_id\n\n\ndf_preds[\"image_id\"] = df_preds[\"file_names\"].apply(extract_id)\n\ndf_preds.head()","metadata":{"_uuid":"d04252577cd31825505fcbce69e7f3b6c052e35f","execution":{"iopub.status.busy":"2022-07-12T20:22:07.761473Z","iopub.execute_input":"2022-07-12T20:22:07.762425Z","iopub.status.idle":"2022-07-12T20:22:07.77695Z","shell.execute_reply.started":"2022-07-12T20:22:07.762381Z","shell.execute_reply":"2022-07-12T20:22:07.775614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get the predicted labels.\n\ny_pred = df_preds[[\"CE\", \"LAA\"]]\n\n# get the id column\npatient_id = [ID[:-2] for ID in list(df_preds[\"image_id\"])]","metadata":{"_uuid":"6b29f14d5a29049a97d319eee2842d8c5ab3753e","execution":{"iopub.status.busy":"2022-07-12T20:22:07.778683Z","iopub.execute_input":"2022-07-12T20:22:07.779134Z","iopub.status.idle":"2022-07-12T20:22:07.788855Z","shell.execute_reply.started":"2022-07-12T20:22:07.7791Z","shell.execute_reply":"2022-07-12T20:22:07.788009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Create a submission file","metadata":{"_uuid":"012b489902bf57b6e3097d9dc38a2513aa260b88"}},{"cell_type":"code","source":"submission = pd.DataFrame(\n    {\n        \"patient_id\": [patient_id[0], patient_id[2], patient_id[1], patient_id[3]],\n        \"CE\": [y_pred[\"CE\"][0], y_pred[\"CE\"][2], y_pred[\"CE\"][1], y_pred[\"CE\"][3]],\n        \"LAA\": [y_pred[\"LAA\"][0], y_pred[\"LAA\"][2],  y_pred[\"LAA\"][1],  y_pred[\"LAA\"][3]],\n    }\n)","metadata":{"_uuid":"429ab64aa5fec3e7652cd30e4c402f835264c8eb","execution":{"iopub.status.busy":"2022-07-12T20:22:07.790581Z","iopub.execute_input":"2022-07-12T20:22:07.791006Z","iopub.status.idle":"2022-07-12T20:22:07.800587Z","shell.execute_reply.started":"2022-07-12T20:22:07.790891Z","shell.execute_reply":"2022-07-12T20:22:07.799756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission[[\"patient_id\", \"CE\", \"LAA\"]].to_csv(\"submission.csv\", index=False)\n\n!head submission.csv","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:22:07.802129Z","iopub.execute_input":"2022-07-12T20:22:07.802859Z","iopub.status.idle":"2022-07-12T20:22:08.516452Z","shell.execute_reply.started":"2022-07-12T20:22:07.802822Z","shell.execute_reply":"2022-07-12T20:22:08.515323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}