{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## ⚡TensorFlow Data Input Pipeline + AlexNet CNN ⚡\nThis notebook servers has a simple example of loading the dataset into TensorFlow for better processing and optimization.\nWe will process and visualize the dataset and later build a classification model on it. The dataset contains additional data such as segmentations and bounding boxes which are useful in bulding more robust models but we are not going to utilize that for this notebook.","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-10T07:02:52.973057Z","iopub.execute_input":"2022-08-10T07:02:52.973444Z","iopub.status.idle":"2022-08-10T07:03:35.606369Z","shell.execute_reply.started":"2022-08-10T07:02:52.973396Z","shell.execute_reply":"2022-08-10T07:03:35.605362Z"}}},{"cell_type":"markdown","source":"We get this note from the data description that some of the DICOM files are JPEG compressed. You may require additional resources to read the pixel array of these files, such as GDCM and pylibjpeg. WE will install this dependencies","metadata":{}},{"cell_type":"code","source":"!pip install -q ../input/for-pydicom/pylibjpeg-1.4.0-py3-none-any.whl\n!pip install -q ../input/for-pydicom/python_gdcm-3.0.14-cp37-cp37m-manylinux_2_17_x86_64.manylinux2014_x86_64.whl\n!pip install -q ../input/for-pydicom/pylibjpeg_libjpeg-1.3.1-cp37-cp37m-manylinux_2_17_x86_64.manylinux2014_x86_64.whl","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:39:24.204914Z","iopub.execute_input":"2022-10-13T14:39:24.205324Z","iopub.status.idle":"2022-10-13T14:40:56.460649Z","shell.execute_reply.started":"2022-10-13T14:39:24.205246Z","shell.execute_reply":"2022-10-13T14:40:56.459417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Handle imports","metadata":{}},{"cell_type":"code","source":"import os \nimport pathlib\nimport glob \nfrom tqdm import tqdm \n\nimport pandas as pd\nimport numpy as np\nimport tensorflow as tf\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport pydicom","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:41:51.024142Z","iopub.execute_input":"2022-10-13T14:41:51.024756Z","iopub.status.idle":"2022-10-13T14:41:51.030499Z","shell.execute_reply.started":"2022-10-13T14:41:51.024722Z","shell.execute_reply":"2022-10-13T14:41:51.029307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set()","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:41:54.069129Z","iopub.execute_input":"2022-10-13T14:41:54.070251Z","iopub.status.idle":"2022-10-13T14:41:54.076065Z","shell.execute_reply.started":"2022-10-13T14:41:54.070207Z","shell.execute_reply":"2022-10-13T14:41:54.07483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Parameters\nEPOCHS = 10\nBATCH_SIZE = 16\nIMAGE_SIZE = (512, 512)\nSEED = 42","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:41:55.885552Z","iopub.execute_input":"2022-10-13T14:41:55.88593Z","iopub.status.idle":"2022-10-13T14:41:55.891454Z","shell.execute_reply.started":"2022-10-13T14:41:55.885898Z","shell.execute_reply":"2022-10-13T14:41:55.89048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# set seed\nnp.random.seed(SEED)\ntf.random.set_seed(SEED)","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:42:19.621684Z","iopub.execute_input":"2022-10-13T14:42:19.622106Z","iopub.status.idle":"2022-10-13T14:42:19.627263Z","shell.execute_reply.started":"2022-10-13T14:42:19.622071Z","shell.execute_reply":"2022-10-13T14:42:19.626056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data  EDA and Processing","metadata":{}},{"cell_type":"code","source":" # the input root folder \nDATA_DIR = \"../input/rsna-2022-cervical-spine-fracture-detection/\"","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:42:22.889415Z","iopub.execute_input":"2022-10-13T14:42:22.889787Z","iopub.status.idle":"2022-10-13T14:42:22.895137Z","shell.execute_reply.started":"2022-10-13T14:42:22.889757Z","shell.execute_reply":"2022-10-13T14:42:22.893463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets list the contents inside the root folder\nos.listdir(DATA_DIR)","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:42:25.708431Z","iopub.execute_input":"2022-10-13T14:42:25.709137Z","iopub.status.idle":"2022-10-13T14:42:25.71993Z","shell.execute_reply.started":"2022-10-13T14:42:25.709099Z","shell.execute_reply":"2022-10-13T14:42:25.718747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# look at what is in the train.csv\ntrain_df = pd.read_csv(DATA_DIR + \"train.csv\")\ntrain_df.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:42:28.847871Z","iopub.execute_input":"2022-10-13T14:42:28.848596Z","iopub.status.idle":"2022-10-13T14:42:28.880484Z","shell.execute_reply.started":"2022-10-13T14:42:28.84856Z","shell.execute_reply":"2022-10-13T14:42:28.879416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.size","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:42:32.46794Z","iopub.execute_input":"2022-10-13T14:42:32.468445Z","iopub.status.idle":"2022-10-13T14:42:32.476388Z","shell.execute_reply.started":"2022-10-13T14:42:32.468404Z","shell.execute_reply":"2022-10-13T14:42:32.475355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# how many unique study instances do we have\ntrain_df.StudyInstanceUID.nunique()","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:42:32.809897Z","iopub.execute_input":"2022-10-13T14:42:32.810709Z","iopub.status.idle":"2022-10-13T14:42:32.828097Z","shell.execute_reply.started":"2022-10-13T14:42:32.810667Z","shell.execute_reply":"2022-10-13T14:42:32.82718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The train dataframe has metadata for the each study instance, c1..c7 are the cervical vertebrae planes and the values in the rows states whether it is fractured or not.\n\nWe will use this for our classification model.","metadata":{}},{"cell_type":"code","source":"# a little deeper inside the train images\nos.listdir(DATA_DIR + \"train_images\")[:5]","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:42:36.201969Z","iopub.execute_input":"2022-10-13T14:42:36.202445Z","iopub.status.idle":"2022-10-13T14:42:36.28197Z","shell.execute_reply.started":"2022-10-13T14:42:36.202406Z","shell.execute_reply":"2022-10-13T14:42:36.28103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"we can see that the train images folder has other subfolders with the study id has its name. Each study can contain several instances with several frames and for this case slices in dicom format.","metadata":{}},{"cell_type":"code","source":"study_instance = \"1.2.826.0.1.3680043.17625\"\n# list the first 5 frames in a select study instance\nos.listdir(DATA_DIR + f\"train_images/{study_instance}\")[:5]","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:42:47.893275Z","iopub.execute_input":"2022-10-13T14:42:47.89364Z","iopub.status.idle":"2022-10-13T14:42:47.937852Z","shell.execute_reply.started":"2022-10-13T14:42:47.893609Z","shell.execute_reply":"2022-10-13T14:42:47.93623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# select all the dicom files in the study instance\nimg_list = glob.glob(DATA_DIR + f\"/train_images/{study_instance}/*.dcm\")\nlen(img_list)","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:42:51.742785Z","iopub.execute_input":"2022-10-13T14:42:51.743171Z","iopub.status.idle":"2022-10-13T14:42:51.751993Z","shell.execute_reply.started":"2022-10-13T14:42:51.743138Z","shell.execute_reply":"2022-10-13T14:42:51.750953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Data Loading Functionalities ","metadata":{"execution":{"iopub.status.busy":"2022-08-10T07:50:38.108624Z","iopub.execute_input":"2022-08-10T07:50:38.109684Z","iopub.status.idle":"2022-08-10T07:50:38.114374Z","shell.execute_reply.started":"2022-08-10T07:50:38.109633Z","shell.execute_reply":"2022-08-10T07:50:38.113244Z"}}},{"cell_type":"code","source":"def load_dicom(path):\n    \"\"\"\n    reads a dicom file and loads the image array inside it\n    inputs:\n        path: the path of the required dicom file\n    returns:\n        data: image pixel arrays\n    \"\"\"\n    img=pydicom.dcmread(path)\n    data=img.pixel_array\n    data=data-np.min(data)\n    if np.max(data) != 0:\n        data=data/np.max(data)\n    data=(data*255).astype(np.uint8)\n    return data","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:42:55.590781Z","iopub.execute_input":"2022-10-13T14:42:55.591173Z","iopub.status.idle":"2022-10-13T14:42:55.597471Z","shell.execute_reply.started":"2022-10-13T14:42:55.591139Z","shell.execute_reply":"2022-10-13T14:42:55.59643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def data_generator():\n    \"\"\"\n    a function that will load the dataset from a list of image paths\n    \"\"\"\n    for path in img_list:\n        data = load_dicom(path)\n        yield data  # return the data has generator","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:43:01.032426Z","iopub.execute_input":"2022-10-13T14:43:01.032798Z","iopub.status.idle":"2022-10-13T14:43:01.038004Z","shell.execute_reply.started":"2022-10-13T14:43:01.032767Z","shell.execute_reply":"2022-10-13T14:43:01.037006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets define a tensorflow dataset variable that will use the generator to get the image data\n# this is efficient beacuse it will only load the data into memory when needed\ntrain_dataset = tf.data.Dataset.from_generator(data_generator, (tf.uint8))","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:43:03.357801Z","iopub.execute_input":"2022-10-13T14:43:03.358191Z","iopub.status.idle":"2022-10-13T14:43:06.059794Z","shell.execute_reply.started":"2022-10-13T14:43:03.35816Z","shell.execute_reply":"2022-10-13T14:43:06.058829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# a quick look of the dataset contents\nfor i in train_dataset.take(1):\n    print(i.shape)\n    print(type(i))","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:43:10.713382Z","iopub.execute_input":"2022-10-13T14:43:10.713756Z","iopub.status.idle":"2022-10-13T14:43:10.812561Z","shell.execute_reply.started":"2022-10-13T14:43:10.713725Z","shell.execute_reply":"2022-10-13T14:43:10.811643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Data Visualization","metadata":{"execution":{"iopub.status.busy":"2022-08-10T07:55:55.633855Z","iopub.execute_input":"2022-08-10T07:55:55.634222Z","iopub.status.idle":"2022-08-10T07:55:55.639198Z","shell.execute_reply.started":"2022-08-10T07:55:55.63419Z","shell.execute_reply":"2022-08-10T07:55:55.637968Z"}}},{"cell_type":"code","source":"def show_single(img, cmap=\"gray\"):\n    \"\"\"\n    plots a single image\n    \"\"\"\n    plt.imshow(img, cmap=cmap)\n    plt.axis(\"off\")","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:43:49.431838Z","iopub.execute_input":"2022-10-13T14:43:49.432515Z","iopub.status.idle":"2022-10-13T14:43:49.437413Z","shell.execute_reply.started":"2022-10-13T14:43:49.432477Z","shell.execute_reply":"2022-10-13T14:43:49.436413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_single(i)","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:43:51.117169Z","iopub.execute_input":"2022-10-13T14:43:51.11807Z","iopub.status.idle":"2022-10-13T14:43:51.328193Z","shell.execute_reply.started":"2022-10-13T14:43:51.118027Z","shell.execute_reply":"2022-10-13T14:43:51.327216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_batch(cmap=\"gray\"):\n    \"\"\"\n    visualizes a batch of images\n    \"\"\"\n    plt.figure(figsize=(16, 12))\n    for i, img in enumerate(train_dataset.take(20)):  # iterate through the dataset\n        plt.subplot(4, 5, i+1)\n        show_single(img, cmap=cmap)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:43:51.866003Z","iopub.execute_input":"2022-10-13T14:43:51.866377Z","iopub.status.idle":"2022-10-13T14:43:51.872516Z","shell.execute_reply.started":"2022-10-13T14:43:51.866347Z","shell.execute_reply":"2022-10-13T14:43:51.871484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### A look of the images using different color maps","metadata":{}},{"cell_type":"code","source":"show_batch(cmap=\"gray\")","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:43:52.761084Z","iopub.execute_input":"2022-10-13T14:43:52.761426Z","iopub.status.idle":"2022-10-13T14:43:54.983568Z","shell.execute_reply.started":"2022-10-13T14:43:52.761397Z","shell.execute_reply":"2022-10-13T14:43:54.982681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_batch(cmap=\"bone\")","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:43:54.985281Z","iopub.execute_input":"2022-10-13T14:43:54.986392Z","iopub.status.idle":"2022-10-13T14:43:56.830835Z","shell.execute_reply.started":"2022-10-13T14:43:54.986342Z","shell.execute_reply":"2022-10-13T14:43:56.830044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_batch(cmap=\"inferno\")","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:43:56.832047Z","iopub.execute_input":"2022-10-13T14:43:56.832785Z","iopub.status.idle":"2022-10-13T14:43:58.846145Z","shell.execute_reply.started":"2022-10-13T14:43:56.832744Z","shell.execute_reply":"2022-10-13T14:43:58.845153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We have so far loaded the images, but it is not ready for training. We need to map the labels from the train_df and as seen earlier, we have 2019 unique study instances. Hence, we have to create a nested loop to iterate through all the study instances and the dicom files inside each study instance.","metadata":{}},{"cell_type":"code","source":"# lets modify the data generator, use 10 study instances\ndef data_generator():\n    for i, study_instance in enumerate(train_df.StudyInstanceUID[:5]):\n        for dcm in os.listdir(DATA_DIR + f\"train_images/{study_instance}\"):\n            train_labels = []\n            path = DATA_DIR + f\"train_images/{study_instance}/{dcm}\"\n            \n            img = load_dicom(path)\n            \n            # resize each image into a shape of (512, 512)\n            img = np.resize(img, (512, 512))\n            #  normalize image\n            img = img / 255.0\n            # convert from gray scale to rgb, this will \nbe helpful incase we want to use pretrained models\n            img = tf.expand_dims(img, axis=-1)\n            img = tf.image.grayscale_to_rgb(img)\n        \n            \n            train_labels.extend([\n                train_df.loc[i, \"C1\"],\n                train_df.loc[i, \"C2\"],\n                train_df.loc[i, \"C3\"],\n                train_df.loc[i, \"C4\"],\n                train_df.loc[i, \"C5\"],\n                train_df.loc[i, \"C6\"],\n                train_df.loc[i, \"C7\"],\n                train_df.loc[i, \"patient_overall\"] # end with patient overall\n            ])\n            yield img, train_labels","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:43:58.848629Z","iopub.execute_input":"2022-10-13T14:43:58.84915Z","iopub.status.idle":"2022-10-13T14:43:58.859679Z","shell.execute_reply.started":"2022-10-13T14:43:58.849098Z","shell.execute_reply":"2022-10-13T14:43:58.858593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = tf.data.Dataset.from_generator(data_generator, (tf.float32, tf.int8))","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:43:58.861059Z","iopub.execute_input":"2022-10-13T14:43:58.861624Z","iopub.status.idle":"2022-10-13T14:43:58.890095Z","shell.execute_reply.started":"2022-10-13T14:43:58.861592Z","shell.execute_reply":"2022-10-13T14:43:58.889281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for img, label in train_data.take(1):\n    print(img.shape)\n    print(label.shape)\n    print(label)","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:43:58.891619Z","iopub.execute_input":"2022-10-13T14:43:58.892172Z","iopub.status.idle":"2022-10-13T14:43:58.990597Z","shell.execute_reply.started":"2022-10-13T14:43:58.892137Z","shell.execute_reply":"2022-10-13T14:43:58.989043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# visualize the image once again\nshow_single(img, cmap=\"gray\")","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:43:58.991748Z","iopub.execute_input":"2022-10-13T14:43:58.992046Z","iopub.status.idle":"2022-10-13T14:43:59.189916Z","shell.execute_reply.started":"2022-10-13T14:43:58.992002Z","shell.execute_reply":"2022-10-13T14:43:59.188939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now we have our one-hot encoded labels, the last thing is to prepare the dataset for training by batching, catching and shuffling","metadata":{"execution":{"iopub.status.busy":"2022-08-10T09:40:08.418329Z","iopub.execute_input":"2022-08-10T09:40:08.420004Z","iopub.status.idle":"2022-08-10T09:40:08.432983Z","shell.execute_reply.started":"2022-08-10T09:40:08.419921Z","shell.execute_reply":"2022-08-10T09:40:08.430658Z"}}},{"cell_type":"markdown","source":"### Split into train and validation","metadata":{}},{"cell_type":"code","source":"# we first need to know the number of data points we are dealing with\nimg_count = 0\nfor _, _ in enumerate(train_df.StudyInstanceUID[:5]):\n    for _ in os.listdir(DATA_DIR + f\"train_images/{study_instance}\"):\n        img_count += 1\nprint(img_count)","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:43:59.191995Z","iopub.execute_input":"2022-10-13T14:43:59.192634Z","iopub.status.idle":"2022-10-13T14:43:59.201968Z","shell.execute_reply.started":"2022-10-13T14:43:59.192597Z","shell.execute_reply":"2022-10-13T14:43:59.200822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_size = int(img_count * 0.2)\ntrain_data = train_data.skip(val_size)\nval_data = train_data.take(val_size)","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:43:59.203939Z","iopub.execute_input":"2022-10-13T14:43:59.204409Z","iopub.status.idle":"2022-10-13T14:43:59.210734Z","shell.execute_reply.started":"2022-10-13T14:43:59.204362Z","shell.execute_reply":"2022-10-13T14:43:59.209772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def configure_for_performance(data):\n    data = data.cache()\n#     data = data.shuffle(buffer_size=300)\n    data = data.batch(16)\n    data = data.prefetch(buffer_size=tf.data.AUTOTUNE)\n    return data","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:43:59.215132Z","iopub.execute_input":"2022-10-13T14:43:59.216029Z","iopub.status.idle":"2022-10-13T14:43:59.221484Z","shell.execute_reply.started":"2022-10-13T14:43:59.215976Z","shell.execute_reply":"2022-10-13T14:43:59.220527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = configure_for_performance(train_data)\nval_data = configure_for_performance(val_data)","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:43:59.222662Z","iopub.execute_input":"2022-10-13T14:43:59.2232Z","iopub.status.idle":"2022-10-13T14:43:59.238058Z","shell.execute_reply.started":"2022-10-13T14:43:59.223165Z","shell.execute_reply":"2022-10-13T14:43:59.237011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modelling","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, BatchNormalization, Dense, Dropout, Flatten","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:44:00.010265Z","iopub.execute_input":"2022-10-13T14:44:00.011089Z","iopub.status.idle":"2022-10-13T14:44:00.033886Z","shell.execute_reply.started":"2022-10-13T14:44:00.011037Z","shell.execute_reply":"2022-10-13T14:44:00.032913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define Alex Net model\ndef alex_net():\n    model = Sequential()\n\n    # 1st Convolutional Layer\n    model.add(Conv2D(filters=96, input_shape=(512,512,3), kernel_size=(11,11),\\\n     strides=(4,4), padding='valid', activation=\"relu\"))\n    # Pooling \n    model.add(MaxPooling2D(pool_size=(2,2), strides=(2,2), padding='valid'))\n    # Batch Normalisation before passing it to the next layer\n    model.add(BatchNormalization())\n\n    # 2nd Convolutional Layer\n    model.add(Conv2D(filters=256, kernel_size=(11,11), strides=(1,1), padding='valid', activation=\"relu\"))\n    # Pooling\n    model.add(MaxPooling2D(pool_size=(2,2), strides=(2,2), padding='valid'))\n    # Batch Normalisation\n    model.add(BatchNormalization())\n\n    # 3rd Convolutional Layer\n    model.add(Conv2D(filters=384, kernel_size=(3,3), strides=(1,1), padding='valid', activation=\"relu\"))\n    # Batch Normalisation\n    model.add(BatchNormalization())\n\n    # 4th Convolutional Layer\n    model.add(Conv2D(filters=384, kernel_size=(3,3), strides=(1,1), padding='valid', activation=\"relu\"))\n    # Batch Normalisation\n    model.add(BatchNormalization())\n\n    # 5th Convolutional Layer\n    model.add(Conv2D(filters=256, kernel_size=(3,3), strides=(1,1), padding='valid', activation=\"relu\"))\n    # Pooling\n    model.add(MaxPooling2D(pool_size=(2,2), strides=(2,2), padding='valid'))\n    # Batch Normalisation\n    model.add(BatchNormalization())\n\n    # Passing it to a dense layer\n    model.add(Flatten())\n    # 1st Dense Layer\n    model.add(Dense(4096, input_shape=(512*512*3,), activation=\"relu\"))\n    # Add Dropout to prevent overfitting\n    model.add(Dropout(0.4))\n    # Batch Normalisation\n    model.add(BatchNormalization())\n\n    # 2nd Dense Layer\n    model.add(Dense(4096, activation=\"relu\"))\n    # Add Dropout\n    model.add(Dropout(0.4))\n    # Batch Normalisation\n    model.add(BatchNormalization())\n\n    # 3rd Dense Layer\n    model.add(Dense(1000, activation=\"relu\"))\n    # Add Dropout\n    model.add(Dropout(0.4))\n    # Batch Normalisation\n    model.add(BatchNormalization())\n\n    # Output Layer with 8 probability classes\n    model.add(Dense(8, activation=\"softmax\"))\n    return model","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:44:00.035339Z","iopub.execute_input":"2022-10-13T14:44:00.035731Z","iopub.status.idle":"2022-10-13T14:44:00.050092Z","shell.execute_reply.started":"2022-10-13T14:44:00.035696Z","shell.execute_reply":"2022-10-13T14:44:00.04926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.applications.mobilenet.MobileNet(\n    input_shape=None,\n    alpha=1.0,\n    depth_multiplier=1,\n    dropout=0.001,\n    include_top=True,\n    weights='imagenet',\n    input_tensor=None,\n    pooling=None,\n    classes=1000,\n    classifier_activation='softmax',\n    **kwargs\n)\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model1 = alex_net()","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:44:00.052768Z","iopub.execute_input":"2022-10-13T14:44:00.053633Z","iopub.status.idle":"2022-10-13T14:44:00.261122Z","shell.execute_reply.started":"2022-10-13T14:44:00.053598Z","shell.execute_reply":"2022-10-13T14:44:00.260197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model1.summary()","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:44:00.262561Z","iopub.execute_input":"2022-10-13T14:44:00.262901Z","iopub.status.idle":"2022-10-13T14:44:00.273239Z","shell.execute_reply.started":"2022-10-13T14:44:00.262866Z","shell.execute_reply":"2022-10-13T14:44:00.272313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model1.compile(optimizer=tf.keras.optimizers.Adam(), \n              loss=tf.keras.losses.CategoricalCrossentropy(),\n              metrics=[tf.keras.metrics.CategoricalAccuracy()]\n             )","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:44:00.275518Z","iopub.execute_input":"2022-10-13T14:44:00.275872Z","iopub.status.idle":"2022-10-13T14:44:00.292405Z","shell.execute_reply.started":"2022-10-13T14:44:00.275837Z","shell.execute_reply":"2022-10-13T14:44:00.291537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# training\nhistory1 = model1.fit(train_data, validation_data=val_data,\n                   epochs=EPOCHS)","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:44:00.29387Z","iopub.execute_input":"2022-10-13T14:44:00.294227Z","iopub.status.idle":"2022-10-13T14:46:32.316721Z","shell.execute_reply.started":"2022-10-13T14:44:00.294194Z","shell.execute_reply":"2022-10-13T14:46:32.315777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# visualize training \ndef viz_loss(history):\n    train_loss = history[\"loss\"]\n    val_loss = history[\"val_loss\"]\n    iters = [i for i in range(EPOCHS)]\n    \n    plt.plot(iters, train_loss, label=\"Training Loss\")\n    plt.plot(iters, val_loss, label=\"Validation Loss\")\n    plt.title(\"A plot of Loss against number of iterations\")\n    plt.legend()\n    plt.show()\n    \ndef viz_acc(history):\n    train_loss = history[\"categorical_accuracy\"]\n    val_loss = history[\"val_categorical_accuracy\"]\n    iters = [i for i in range(EPOCHS)]\n    \n    plt.plot(iters, train_loss, label=\"Training Accuracy\")\n    plt.plot(iters, val_loss, label=\"Validation Accuracy\")\n    plt.title(\"A plot of Accuracy against number of iterations\")\n    plt.legend()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:46:32.318373Z","iopub.execute_input":"2022-10-13T14:46:32.319043Z","iopub.status.idle":"2022-10-13T14:46:32.327978Z","shell.execute_reply.started":"2022-10-13T14:46:32.319Z","shell.execute_reply":"2022-10-13T14:46:32.327034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"viz_loss(history1.history)\nviz_acc(history1.history)","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:46:32.32936Z","iopub.execute_input":"2022-10-13T14:46:32.330362Z","iopub.status.idle":"2022-10-13T14:46:32.984122Z","shell.execute_reply.started":"2022-10-13T14:46:32.330325Z","shell.execute_reply":"2022-10-13T14:46:32.982938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{"execution":{"iopub.status.busy":"2022-08-10T12:59:07.520703Z","iopub.execute_input":"2022-08-10T12:59:07.521353Z","iopub.status.idle":"2022-08-10T12:59:07.528859Z","shell.execute_reply.started":"2022-08-10T12:59:07.521319Z","shell.execute_reply":"2022-08-10T12:59:07.527249Z"}}},{"cell_type":"code","source":"# prep test data for submission\ntest_df = pd.read_csv(DATA_DIR + \"test.csv\")\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:46:32.987503Z","iopub.execute_input":"2022-10-13T14:46:32.987856Z","iopub.status.idle":"2022-10-13T14:46:33.005501Z","shell.execute_reply.started":"2022-10-13T14:46:32.987825Z","shell.execute_reply":"2022-10-13T14:46:33.003909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"global test_ids\ntest_ids = []\ndef test_data_generator():\n    for study_instance in os.listdir(DATA_DIR + f\"test_images\"):\n        for dcm in os.listdir(DATA_DIR + f\"test_images/{study_instance}\"):\n            path = DATA_DIR + f\"test_images/{study_instance}/{dcm}\"\n            img = load_dicom(path)\n            \n            # resize each image into a shape of (512, 512)\n            img = np.resize(img, (512, 512))\n            #  normalize image\n            img = tf.cast(img, tf.float32) / 255.0\n            # convert from gray scale to rgb, this will be helpful incase we want to use pretrained models\n            img = tf.expand_dims(img, axis=-1)\n            img = tf.image.grayscale_to_rgb(img)\n            test_ids.append(study_instance)\n            yield img","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:46:33.006665Z","iopub.execute_input":"2022-10-13T14:46:33.006935Z","iopub.status.idle":"2022-10-13T14:46:33.014541Z","shell.execute_reply.started":"2022-10-13T14:46:33.006909Z","shell.execute_reply":"2022-10-13T14:46:33.013483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = tf.data.Dataset.from_generator(test_data_generator, tf.float32).batch(1)","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:46:33.016281Z","iopub.execute_input":"2022-10-13T14:46:33.016753Z","iopub.status.idle":"2022-10-13T14:46:33.044123Z","shell.execute_reply.started":"2022-10-13T14:46:33.016718Z","shell.execute_reply":"2022-10-13T14:46:33.043143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# make predictions\npreds = []\nfor img in tqdm(test_data):\n    preds.append(model.predict(img)[0])\npreds = np.array(preds)","metadata":{"execution":{"iopub.status.busy":"2022-10-13T15:49:54.267101Z","iopub.execute_input":"2022-10-13T15:49:54.267963Z","iopub.status.idle":"2022-10-13T15:49:54.342492Z","shell.execute_reply.started":"2022-10-13T15:49:54.267869Z","shell.execute_reply":"2022-10-13T15:49:54.340916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"assert len(test_ids) == len(preds)","metadata":{"execution":{"iopub.status.busy":"2022-10-13T15:49:54.343424Z","iopub.status.idle":"2022-10-13T15:49:54.343762Z","shell.execute_reply.started":"2022-10-13T15:49:54.343602Z","shell.execute_reply":"2022-10-13T15:49:54.343618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result = pd.DataFrame(columns = [\"StudyInstanceUID\", 'C1', 'C2', 'C3', 'C4', 'C5', 'C6', 'C7', 'patient_overall'])","metadata":{"execution":{"iopub.status.busy":"2022-10-13T15:49:54.345137Z","iopub.status.idle":"2022-10-13T15:49:54.346285Z","shell.execute_reply.started":"2022-10-13T15:49:54.34604Z","shell.execute_reply":"2022-10-13T15:49:54.346067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in tqdm(range(len(test_ids))):\n    result.loc[i, 'StudyInstanceUID'] = test_ids[i]\n    rows = preds[i].round(3)\n    result.loc[i, 'C1'] = rows[0]\n    result.loc[i, 'C2'] = rows[1]\n    result.loc[i, 'C3'] = rows[2]\n    result.loc[i, 'C4'] = rows[3]\n    result.loc[i, 'C5'] = rows[4]\n    result.loc[i, 'C6'] = rows[5]\n    result.loc[i, 'C7'] = rows[6]\n    result.loc[i, 'patient_overall'] = rows[7]","metadata":{"execution":{"iopub.status.busy":"2022-10-13T15:49:54.347632Z","iopub.status.idle":"2022-10-13T15:49:54.348518Z","shell.execute_reply.started":"2022-10-13T15:49:54.348256Z","shell.execute_reply":"2022-10-13T15:49:54.348281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-13T15:49:54.357879Z","iopub.execute_input":"2022-10-13T15:49:54.358205Z","iopub.status.idle":"2022-10-13T15:49:54.372527Z","shell.execute_reply.started":"2022-10-13T15:49:54.358179Z","shell.execute_reply":"2022-10-13T15:49:54.37002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# means = result[['patient_overall', 'C1', 'C2', 'C3', 'C4', 'C5', 'C6', 'C7']].mean().to_dict()\n# sample_submission = pd.read_csv(DATA_DIR + \"sample_submission.csv\")\n# sample_submission.head()\n# sample_submission.to_csv(\"submission.csv\", index=\"false\")","metadata":{"execution":{"iopub.status.busy":"2022-10-13T15:49:54.380768Z","iopub.execute_input":"2022-10-13T15:49:54.38115Z","iopub.status.idle":"2022-10-13T15:49:54.384909Z","shell.execute_reply.started":"2022-10-13T15:49:54.381124Z","shell.execute_reply":"2022-10-13T15:49:54.38404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#make submission corresponding to the submission format\n# idxs = ['C1', 'C2', 'C3', 'C4', 'C5', 'C6', 'C7', 'patient_overall']\n# sub = []\n# for i in tqdm(range(len(test_ids))):\n#     for j, el in enumerate(idxs):\n#         sub.append([result.loc[i].StudyInstanceUID + f\"_{el}\", result.loc[0][j+1]])","metadata":{"execution":{"iopub.status.busy":"2022-10-13T15:49:54.39394Z","iopub.execute_input":"2022-10-13T15:49:54.394264Z","iopub.status.idle":"2022-10-13T15:49:54.400207Z","shell.execute_reply.started":"2022-10-13T15:49:54.394236Z","shell.execute_reply":"2022-10-13T15:49:54.399221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"means = result[['patient_overall', 'C1', 'C2', 'C3', 'C4', 'C5', 'C6', 'C7']].mean().to_dict()\nprint(means)","metadata":{"execution":{"iopub.status.busy":"2022-10-13T15:49:54.41193Z","iopub.execute_input":"2022-10-13T15:49:54.41222Z","iopub.status.idle":"2022-10-13T15:49:54.425173Z","shell.execute_reply.started":"2022-10-13T15:49:54.412196Z","shell.execute_reply":"2022-10-13T15:49:54.423644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['fractured'] = test_df['prediction_type'].map(means)\ntest_df[['row_id','fractured']].to_csv('submission.csv', index=False, float_format='%.1g')","metadata":{"execution":{"iopub.status.busy":"2022-10-13T15:49:54.429317Z","iopub.execute_input":"2022-10-13T15:49:54.429911Z","iopub.status.idle":"2022-10-13T15:49:54.442922Z","shell.execute_reply.started":"2022-10-13T15:49:54.429877Z","shell.execute_reply":"2022-10-13T15:49:54.441645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!cat submission.csv","metadata":{"execution":{"iopub.status.busy":"2022-10-13T15:49:54.444091Z","iopub.status.idle":"2022-10-13T15:49:54.444773Z","shell.execute_reply.started":"2022-10-13T15:49:54.444527Z","shell.execute_reply":"2022-10-13T15:49:54.444551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Conclussion \nWe have seen how to process the dicom files and transform the dataset into tensorflow format. However, we didn't use all the dataset and the cpu memory was filling up easily. We could do some optimizations to train with all the images.\n\n## What to do next\n1. Handle JPEG dicom format correctly (had some few errors)\n2. Train will all the images\n3. Use segmentations and bounding boxes\n4. Perform cross validation","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}