{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# As the data is intimedataing for new people to images analysis (like me !) I created this notebook which aims to understand data and make a first submission","metadata":{}},{"cell_type":"markdown","source":"### 1. Install this packages for correctly reading images (credit : [this notebook of @samuelcortinhas](https://www.kaggle.com/code/samuelcortinhas/rsna-fracture-detection-in-depth-eda))\n## BUT BEFORE YOU HAVE TO ADD pydicom utility : Add Data menu --> search for Pydicom Utility --> add the data","metadata":{"execution":{"iopub.status.busy":"2022-09-08T10:30:53.357779Z","iopub.execute_input":"2022-09-08T10:30:53.35868Z","iopub.status.idle":"2022-09-08T10:30:53.381486Z","shell.execute_reply.started":"2022-09-08T10:30:53.358571Z","shell.execute_reply":"2022-09-08T10:30:53.380449Z"}}},{"cell_type":"code","source":"!pip install -qU '../input/for-pydicom/python_gdcm-3.0.14-cp37-cp37m-manylinux_2_17_x86_64.manylinux2014_x86_64.whl' --find-links frozen_packages --no-index\n","metadata":{"execution":{"iopub.status.busy":"2022-09-14T13:04:36.814741Z","iopub.execute_input":"2022-09-14T13:04:36.815595Z","iopub.status.idle":"2022-09-14T13:04:50.002179Z","shell.execute_reply.started":"2022-09-14T13:04:36.815532Z","shell.execute_reply":"2022-09-14T13:04:50.000976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -qU '../input/for-pydicom/pylibjpeg-1.4.0-py3-none-any.whl' --find-links frozen_packages --no-index","metadata":{"execution":{"iopub.status.busy":"2022-09-14T13:04:50.005635Z","iopub.execute_input":"2022-09-14T13:04:50.006056Z","iopub.status.idle":"2022-09-14T13:05:01.023694Z","shell.execute_reply.started":"2022-09-14T13:04:50.006013Z","shell.execute_reply":"2022-09-14T13:05:01.022637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 2. Import the necessary packages","metadata":{"execution":{"iopub.status.busy":"2022-09-08T10:37:36.90759Z","iopub.execute_input":"2022-09-08T10:37:36.908029Z","iopub.status.idle":"2022-09-08T10:37:36.913197Z","shell.execute_reply.started":"2022-09-08T10:37:36.907986Z","shell.execute_reply":"2022-09-08T10:37:36.912168Z"}}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as patches\nimport seaborn as sns\nsns.set(style='darkgrid', font_scale=1.6)\nimport cv2\nfrom os import listdir\nimport re\nimport gc\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nfrom tqdm import tqdm\nimport nibabel as nib\nimport random","metadata":{"execution":{"iopub.status.busy":"2022-09-14T13:05:01.049182Z","iopub.execute_input":"2022-09-14T13:05:01.049504Z","iopub.status.idle":"2022-09-14T13:05:02.180769Z","shell.execute_reply.started":"2022-09-14T13:05:01.049474Z","shell.execute_reply":"2022-09-14T13:05:02.179826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 3. Explore the csv files : train.csv and train_bounding_boxes","metadata":{}},{"cell_type":"markdown","source":"#### Read train.csv","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('../input/rsna-2022-cervical-spine-fracture-detection/train.csv')\nprint(train.info())\ntrain.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-09-14T13:05:02.181994Z","iopub.execute_input":"2022-09-14T13:05:02.182551Z","iopub.status.idle":"2022-09-14T13:05:02.239273Z","shell.execute_reply.started":"2022-09-14T13:05:02.182518Z","shell.execute_reply":"2022-09-14T13:05:02.238417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Let's create a function that returns a list of the fractured bones of a given id. We may need it later","metadata":{}},{"cell_type":"code","source":"def get_fractured_bones(patient_id):\n    fractured_bones = []\n    temp = train.loc[train.StudyInstanceUID == patient_id,['C1','C2', 'C3', 'C4', 'C5', 'C6', 'C7']]\n    temp = list(temp.values[0]) # there is one row per id\n    for i in range(len(temp)):\n        if temp[i] == 1:\n            fractured_bones.append('C' + str(i+1))\n    return fractured_bones","metadata":{"execution":{"iopub.status.busy":"2022-09-14T13:05:02.240491Z","iopub.execute_input":"2022-09-14T13:05:02.241417Z","iopub.status.idle":"2022-09-14T13:05:02.247403Z","shell.execute_reply.started":"2022-09-14T13:05:02.241381Z","shell.execute_reply":"2022-09-14T13:05:02.246521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Test the function\nget_fractured_bones('1.2.826.0.1.3680043.6200')","metadata":{"execution":{"iopub.status.busy":"2022-09-14T13:05:02.248454Z","iopub.execute_input":"2022-09-14T13:05:02.249321Z","iopub.status.idle":"2022-09-14T13:05:02.267917Z","shell.execute_reply.started":"2022-09-14T13:05:02.249287Z","shell.execute_reply":"2022-09-14T13:05:02.266939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Read train_bounding_boxes","metadata":{}},{"cell_type":"code","source":"train_bb = pd.read_csv('../input/rsna-2022-cervical-spine-fracture-detection/train_bounding_boxes.csv')\nprint(train_bb.info())\ntrain_bb.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-09-14T13:05:02.269819Z","iopub.execute_input":"2022-09-14T13:05:02.270608Z","iopub.status.idle":"2022-09-14T13:05:02.321035Z","shell.execute_reply.started":"2022-09-14T13:05:02.270552Z","shell.execute_reply":"2022-09-14T13:05:02.319805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### What's the link with train.csv ?<br>\nFirst, the StudyInstanceUID column is present in both files (later on we will call it id)<br> \nBut, not all ids present in train are also present in train_bb, because train_bb contains only fractured bones.<br>\nLet's check that :","metadata":{}},{"cell_type":"code","source":"# get fractured and unfractured ids from train\nfractured_ids = list(train.loc[train['patient_overall']==1,:]['StudyInstanceUID'])\nunfractured_ids = list(train.loc[train['patient_overall']==0,:]['StudyInstanceUID'])\ntrain_ids = fractured_ids + unfractured_ids\nprint('Number of ids from train : ', len(train_ids))\nprint('Number of unfractured ids from train : ', len(unfractured_ids))\nprint('Number of fractured ids from train : ', len(fractured_ids))\n\n# get ids from train_bb\nids_bb = train_bb['StudyInstanceUID'].unique()\nprint('Number of ids in train_bb : ', len(ids_bb))\n\n# print ids not in train_bb\nids_not_in_train_bb = [i for i in train_ids  if not i in ids_bb]\nunfr_ids_not_in_train_bb = [i for i in unfractured_ids  if not i in ids_bb]\nfr_ids_not_in_train_bb = [i for i in fractured_ids  if not i in ids_bb]\nprint('Number of ids not in train_bb : ', len(ids_not_in_train_bb))\nprint('Number of unfractered ids not in train_bb : ', len(unfr_ids_not_in_train_bb))\nprint('Number of fractered ids not in train_bb : ', len(fr_ids_not_in_train_bb))","metadata":{"execution":{"iopub.status.busy":"2022-09-14T13:05:02.326008Z","iopub.execute_input":"2022-09-14T13:05:02.326751Z","iopub.status.idle":"2022-09-14T13:05:02.381079Z","shell.execute_reply.started":"2022-09-14T13:05:02.326701Z","shell.execute_reply":"2022-09-14T13:05:02.379749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### From line 2 and line 6, we can conclude that train_bb concerns only fractured bones.\n#### But from a total of 961 fractured bones (line 3), train_bb contains only 235 (line 4)","metadata":{}},{"cell_type":"markdown","source":"Second, slice number refers to a dcm file present in the folder with the same name as the coressponding id. For example for the first line of train_bb with id = '1.2.826.0.1.3680043.10051' and slice_number = 133, we can check that there is a 133.dcm file in **'../input/rsna-2022-cervical-spine-fracture-detection/train_images/1.2.826.0.1.3680043.10051'**<br>\nThis file is an image that we can read like this : ","metadata":{}},{"cell_type":"code","source":"# path of train images\npath_ti = '/kaggle/input/rsna-2022-cervical-spine-fracture-detection/train_images'\n# some id\nid1 = '1.2.826.0.1.3680043.10051'\n# some slice of that id\nslice_number = 136\n# Read the file of that id and that slice number\ndicom_file = pydicom.dcmread(path_ti + '/' + id1 + '/' + str(slice_number) + '.dcm')\n# print dicome_file\nprint(dicom_file)","metadata":{"execution":{"iopub.status.busy":"2022-09-14T13:05:02.382946Z","iopub.execute_input":"2022-09-14T13:05:02.383396Z","iopub.status.idle":"2022-09-14T13:05:02.407153Z","shell.execute_reply.started":"2022-09-14T13:05:02.383351Z","shell.execute_reply":"2022-09-14T13:05:02.405946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### We can plot the image from the pixel_array field like this :","metadata":{}},{"cell_type":"code","source":"# get the image as an array\ndicom_file_arr = dicom_file.pixel_array\nprint('sahpe of array image : ', dicom_file_arr.shape)","metadata":{"execution":{"iopub.status.busy":"2022-09-14T13:05:02.408791Z","iopub.execute_input":"2022-09-14T13:05:02.412142Z","iopub.status.idle":"2022-09-14T13:05:02.417748Z","shell.execute_reply.started":"2022-09-14T13:05:02.4121Z","shell.execute_reply":"2022-09-14T13:05:02.416923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### Plot the image and the bounding boxe :","metadata":{}},{"cell_type":"code","source":"# get bounding box for id1 and slice_number as a list\nbounding_boxe = train_bb.loc[(train_bb.StudyInstanceUID == id1) & (train_bb.slice_number == slice_number), :]\nbounding_boxe = list(bounding_boxe.values[0])\nx = bounding_boxe[1] # x-coordinate of the top left of the boxe\ny = bounding_boxe[2] # y-coordinate of the top left of the boxe\nw = bounding_boxe[3] # width of the boxe\nh = bounding_boxe[4] # height of the boxe\n# Create a Rectangle patch to add later to the figure\nrect = patches.Rectangle((x, y), w, h, linewidth=2, edgecolor='r', facecolor=\"none\") # (x, y), width, height\n\n#Define figure and axes\nfig, ax = plt.subplots(1,1, figsize=(10,10))\n# add the image\nax.imshow(dicom_file_arr, cmap=plt.cm.bone)\n# Add the boxe\nax.add_patch(rect)\n# diplay\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-09-14T13:05:02.419364Z","iopub.execute_input":"2022-09-14T13:05:02.420029Z","iopub.status.idle":"2022-09-14T13:05:02.915894Z","shell.execute_reply.started":"2022-09-14T13:05:02.419994Z","shell.execute_reply":"2022-09-14T13:05:02.914876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### From train.csv we can get fractured bones for that id :","metadata":{}},{"cell_type":"code","source":"str(get_fractured_bones(id1))","metadata":{"execution":{"iopub.status.busy":"2022-09-14T13:05:02.917114Z","iopub.execute_input":"2022-09-14T13:05:02.917962Z","iopub.status.idle":"2022-09-14T13:05:02.926237Z","shell.execute_reply.started":"2022-09-14T13:05:02.917925Z","shell.execute_reply":"2022-09-14T13:05:02.925262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### We have only one fractured bone (C4) but we have 16 bounding boxes !! :","metadata":{}},{"cell_type":"code","source":"bounding_boxes_id1_df = train_bb.loc[(train_bb.StudyInstanceUID == id1) , :]\nbounding_boxes_id1_df","metadata":{"execution":{"iopub.status.busy":"2022-09-14T13:05:02.927439Z","iopub.execute_input":"2022-09-14T13:05:02.928381Z","iopub.status.idle":"2022-09-14T13:05:02.951607Z","shell.execute_reply.started":"2022-09-14T13:05:02.928337Z","shell.execute_reply":"2022-09-14T13:05:02.950249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### So let's plot all of these, but by defining a function that we can use later :","metadata":{}},{"cell_type":"code","source":"def plot_16_first_slices(patient_id, MAX_PLOTS = 16, plot_size = (20, 20)):\n    \n    # get bounding boxes for that id (it may be empty !)\n    bounding_boxes_df = train_bb.loc[(train_bb.StudyInstanceUID == patient_id) , :].head(MAX_PLOTS).reset_index() # reset_index to get indexes starting from 0\n    \n    # number of slices\n    n = len(bounding_boxes_df)\n    \n    if n == 0:\n        return\n    \n    # number of rows and columns for the fig\n    n_rows = int(np.sqrt(n))\n    n_cols = n // n_rows\n    \n    if n_rows * n_cols < n:\n        n_cols = n_cols + 1\n        \n    # define the subplots\n    fig, ax = plt.subplots(n_rows, n_cols, figsize=plot_size)\n    fig.suptitle('ID : '+ patient_id + ' - Fractured bones : ' + str(get_fractured_bones(patient_id)), fontsize=16)\n    \n    \n\n    for row_index,row in bounding_boxes_df.iterrows():\n        # Create a Rectangle patch to add later to the figure\n        rect = patches.Rectangle((row.x, row.y), row.width, row.height, linewidth=2, edgecolor='r', facecolor=\"none\") # (x, y), width, height\n        # read the image\n        dicom_file = pydicom.dcmread(path_ti + '/' + row.StudyInstanceUID + '/' + str(row.slice_number) + '.dcm')\n        dicom_file_arr = dicom_file.pixel_array\n        # add the image\n        i = row_index // n_cols\n        j = row_index % n_cols\n        ax[i, j].imshow(dicom_file_arr, cmap=plt.cm.bone)\n        # Add the boxe\n        ax[i, j].add_patch(rect)\n        #title\n        ax[i, j].set_title('Slice : {}'.format(str(row.slice_number)), fontsize = 14)\n        \n    # diplay\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-09-14T13:05:02.953418Z","iopub.execute_input":"2022-09-14T13:05:02.953921Z","iopub.status.idle":"2022-09-14T13:05:02.967257Z","shell.execute_reply.started":"2022-09-14T13:05:02.953852Z","shell.execute_reply":"2022-09-14T13:05:02.966101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot for id1\nplot_16_first_slices(id1)","metadata":{"execution":{"iopub.status.busy":"2022-09-14T13:05:02.96918Z","iopub.execute_input":"2022-09-14T13:05:02.969919Z","iopub.status.idle":"2022-09-14T13:05:06.453593Z","shell.execute_reply.started":"2022-09-14T13:05:02.969849Z","shell.execute_reply":"2022-09-14T13:05:06.452467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot for another id where fractured bones are C4 and C5\nid2 = '1.2.826.0.1.3680043.9940'\nplot_16_first_slices(id2)","metadata":{"execution":{"iopub.status.busy":"2022-09-14T13:05:06.455136Z","iopub.execute_input":"2022-09-14T13:05:06.456246Z","iopub.status.idle":"2022-09-14T13:05:09.073432Z","shell.execute_reply.started":"2022-09-14T13:05:06.456201Z","shell.execute_reply":"2022-09-14T13:05:09.072385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### Now, what about segmentation ?<br>\n##### These files are very special as they contain many images. We can get data from them as array like this :","metadata":{}},{"cell_type":"code","source":"path = '../input/rsna-2022-cervical-spine-fracture-detection/segmentations/'\nid3 = '1.2.826.0.1.3680043.10633'\nsegmentations = nib.load( path + id3 + '.nii').get_fdata()","metadata":{"execution":{"iopub.status.busy":"2022-09-14T13:05:09.074846Z","iopub.execute_input":"2022-09-14T13:05:09.075246Z","iopub.status.idle":"2022-09-14T13:05:10.892989Z","shell.execute_reply.started":"2022-09-14T13:05:09.075213Z","shell.execute_reply":"2022-09-14T13:05:10.891647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"segmentations.shape","metadata":{"execution":{"iopub.status.busy":"2022-09-14T13:05:10.894687Z","iopub.execute_input":"2022-09-14T13:05:10.895075Z","iopub.status.idle":"2022-09-14T13:05:10.90156Z","shell.execute_reply.started":"2022-09-14T13:05:10.89504Z","shell.execute_reply":"2022-09-14T13:05:10.9005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### But to get the right orientation of the image we must do this transformation :\n[See this discution of Harshit Sheoran.](https://www.kaggle.com/competitions/rsna-2022-cervical-spine-fracture-detection/discussion/340612)","metadata":{}},{"cell_type":"code","source":"segmentations = segmentations[:, ::-1, ::-1] # invserse orientation of dim 1 and 2\n\nsegmentations = segmentations.transpose(2, 1, 0) # then switch dim 0 with dim 2\n\nprint(segmentations.shape)","metadata":{"execution":{"iopub.status.busy":"2022-09-14T13:05:10.903038Z","iopub.execute_input":"2022-09-14T13:05:10.90347Z","iopub.status.idle":"2022-09-14T13:05:10.918644Z","shell.execute_reply.started":"2022-09-14T13:05:10.903429Z","shell.execute_reply":"2022-09-14T13:05:10.917266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"[](http://)","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(1,1, figsize=(8,8))\nax.imshow(segmentations[0], cmap=plt.cm.bone)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-09-14T13:05:10.920813Z","iopub.execute_input":"2022-09-14T13:05:10.922058Z","iopub.status.idle":"2022-09-14T13:05:11.204044Z","shell.execute_reply.started":"2022-09-14T13:05:10.92201Z","shell.execute_reply":"2022-09-14T13:05:11.202797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### There is noting in this image ! \n##### and we can see that many of the first images contains nothing\n##### Let's plot other ones at random begining from position 200:","metadata":{"execution":{"iopub.status.busy":"2022-09-09T14:21:17.087063Z","iopub.execute_input":"2022-09-09T14:21:17.08992Z","iopub.status.idle":"2022-09-09T14:21:17.099795Z","shell.execute_reply.started":"2022-09-09T14:21:17.089871Z","shell.execute_reply":"2022-09-09T14:21:17.098516Z"}}},{"cell_type":"code","source":"seg = [i for i in range(200, 300)]  #random.sample(range(200,300), 16)\n\nfig, ax = plt.subplots(4,4, figsize=(20,20))\n\nfor i in range(16):\n    r = i // 4\n    c = i % 4\n    ax[r, c].imshow(segmentations[seg[i],:,:], cmap=plt.cm.bone)\n    \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-09-14T13:05:11.205666Z","iopub.execute_input":"2022-09-14T13:05:11.206146Z","iopub.status.idle":"2022-09-14T13:05:14.007356Z","shell.execute_reply.started":"2022-09-14T13:05:11.2061Z","shell.execute_reply":"2022-09-14T13:05:14.005967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### What we'll see if we sum up all these sementations :","metadata":{}},{"cell_type":"code","source":"temp = segmentations[0,:,:]\nfor i in range(segmentations.shape[0]):\n    temp = temp + segmentations[i,:,:]\n\n\nfig, ax = plt.subplots(1,1, figsize=(10,10))\nax.imshow(temp, cmap=plt.cm.bone)\n\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-09-14T13:05:14.009255Z","iopub.execute_input":"2022-09-14T13:05:14.009705Z","iopub.status.idle":"2022-09-14T13:05:14.510327Z","shell.execute_reply.started":"2022-09-14T13:05:14.009655Z","shell.execute_reply":"2022-09-14T13:05:14.509307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### Let's define a function so we can try another ids","metadata":{}},{"cell_type":"code","source":"def plot_nii_files(patient_id, seg_order = None, plot = True):\n    path = '../input/rsna-2022-cervical-spine-fracture-detection/segmentations/'\n    segmentations = nib.load( path + patient_id + '.nii').get_fdata()[:, ::-1, ::-1].transpose(2, 1, 0)\n    \n    if seg_order == None:\n        img = segmentations[0,:,:]\n        for i in range(segmentations.shape[0]):\n            img = img + segmentations[i,:,:]\n    else:\n        img = segmentations[seg_order,:,:]\n        \n    if plot :\n        # plot the image\n        fig, ax = plt.subplots(1,1, figsize=(8,8))\n        ax.imshow(img, cmap=plt.cm.bone)\n        plt.show()\n    else:\n        return img\n    \n    ","metadata":{"execution":{"iopub.status.busy":"2022-09-14T13:05:14.511427Z","iopub.execute_input":"2022-09-14T13:05:14.511803Z","iopub.status.idle":"2022-09-14T13:05:14.521695Z","shell.execute_reply.started":"2022-09-14T13:05:14.511768Z","shell.execute_reply":"2022-09-14T13:05:14.519954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id4 = '1.2.826.0.1.3680043.19333'\nplot_nii_files(id4)","metadata":{"execution":{"iopub.status.busy":"2022-09-14T13:05:14.523745Z","iopub.execute_input":"2022-09-14T13:05:14.524255Z","iopub.status.idle":"2022-09-14T13:05:15.832273Z","shell.execute_reply.started":"2022-09-14T13:05:14.524207Z","shell.execute_reply":"2022-09-14T13:05:15.830997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Let's prepare data for a Conv2D model","metadata":{}},{"cell_type":"markdown","source":"##### First define a function that return a number of image for a number of ids and there labels :","metadata":{}},{"cell_type":"code","source":"cols = ['patient_overall', 'C1', 'C2', 'C3', 'C4', 'C5', 'C6', 'C7'] # for labels\n\ndef get_train_images(train_df, n_ids = 10, n_img = 10, img_size = (32, 32)):\n    \n    train_images = []\n    train_labels = []\n    \n    # get n_ids ids at random\n    ids = random.sample(list(train_df['StudyInstanceUID'].values), n_ids)\n    \n    \n    for id_ in ids:\n        # get labels for tha id_\n        labels = train_df.loc[train_df['StudyInstanceUID'] == id_, cols].values[0]\n       \n        # file names of that id_\n        id_files = os.listdir(path_ti + '/' + id_)\n        # read each of the dcm files\n        for i in range(min(len(id_files), n_ids)):\n            image = pydicom.dcmread(path_ti + '/' + id_ + '/' + id_files[i]).pixel_array\n            # resize image\n            image = np.array(cv2.resize(image, dsize=img_size))\n            # rescale image\n            image = (image - image.min())/image.max()\n            # add image to train_image\n            train_images.append(image)\n            # add labels train_labels\n            train_labels.append(labels)\n    return np.array(train_images), np.array(train_labels)\n\n            \n            ","metadata":{"execution":{"iopub.status.busy":"2022-09-14T13:05:15.833829Z","iopub.execute_input":"2022-09-14T13:05:15.834195Z","iopub.status.idle":"2022-09-14T13:05:15.844117Z","shell.execute_reply.started":"2022-09-14T13:05:15.834161Z","shell.execute_reply":"2022-09-14T13:05:15.842908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test the function\nget_train_images(train, 2, 2)[1].shape","metadata":{"execution":{"iopub.status.busy":"2022-09-14T13:05:15.84948Z","iopub.execute_input":"2022-09-14T13:05:15.849818Z","iopub.status.idle":"2022-09-14T13:05:16.134318Z","shell.execute_reply.started":"2022-09-14T13:05:15.849787Z","shell.execute_reply":"2022-09-14T13:05:16.133123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### Now a function that yields this images (will be needed later)","metadata":{}},{"cell_type":"code","source":"def generate_train_images(train_df, n_ids = 10, n_img = 10, img_size = (32, 32)):\n    while 1:\n        yield get_train_images(train_df, n_ids , n_img , img_size )","metadata":{"execution":{"iopub.status.busy":"2022-09-14T13:05:16.136927Z","iopub.execute_input":"2022-09-14T13:05:16.138272Z","iopub.status.idle":"2022-09-14T13:05:16.14422Z","shell.execute_reply.started":"2022-09-14T13:05:16.13823Z","shell.execute_reply":"2022-09-14T13:05:16.142978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### The Model (a simple one !) ","metadata":{}},{"cell_type":"code","source":"# import libraries\nfrom tensorflow.keras import layers\nfrom tensorflow.keras import Model\nfrom tensorflow.keras.optimizers import RMSprop","metadata":{"execution":{"iopub.status.busy":"2022-09-14T13:05:16.145643Z","iopub.execute_input":"2022-09-14T13:05:16.146132Z","iopub.status.idle":"2022-09-14T13:05:22.559544Z","shell.execute_reply.started":"2022-09-14T13:05:16.146083Z","shell.execute_reply":"2022-09-14T13:05:22.558506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_model():\n    \n    img_input = layers.Input(shape=(32, 32,1))\n\n    # First convolution extracts 16 filters that are 3x3\n    # Convolution is followed by max-pooling layer with a 2x2 window\n    x = layers.Conv2D(16, 3, activation='relu')(img_input)\n    x = layers.MaxPooling2D(2)(x)\n\n    # Flatten feature map to a 1-dim tensor so we can add fully connected layers\n    x = layers.Flatten()(x)\n\n    # Create output layer with a single node and sigmoid activation\n    output = layers.Dense(8, activation='sigmoid')(x)\n\n\n    # Create model:\n    model = Model(img_input, output)\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2022-09-14T13:05:22.560943Z","iopub.execute_input":"2022-09-14T13:05:22.562348Z","iopub.status.idle":"2022-09-14T13:05:22.570032Z","shell.execute_reply.started":"2022-09-14T13:05:22.562305Z","shell.execute_reply":"2022-09-14T13:05:22.56804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = get_model()","metadata":{"execution":{"iopub.status.busy":"2022-09-14T13:05:22.571727Z","iopub.execute_input":"2022-09-14T13:05:22.572165Z","iopub.status.idle":"2022-09-14T13:05:22.913236Z","shell.execute_reply.started":"2022-09-14T13:05:22.572102Z","shell.execute_reply":"2022-09-14T13:05:22.912083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(loss='binary_crossentropy',\n              optimizer=RMSprop(lr=0.001),\n              metrics=['acc'])","metadata":{"execution":{"iopub.status.busy":"2022-09-14T13:05:22.914642Z","iopub.execute_input":"2022-09-14T13:05:22.914998Z","iopub.status.idle":"2022-09-14T13:05:22.929965Z","shell.execute_reply.started":"2022-09-14T13:05:22.914964Z","shell.execute_reply":"2022-09-14T13:05:22.929095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# count the total number of image\n# n_train_images = 0\n# for id_ in train['StudyInstanceUID']:\n#     n_train_images = n_train_images + len(os.listdir(path_ti + '/' + id_))\n# print(n_train_images)\n\nn_train_images = 711601","metadata":{"execution":{"iopub.status.busy":"2022-09-14T13:05:22.931258Z","iopub.execute_input":"2022-09-14T13:05:22.932135Z","iopub.status.idle":"2022-09-14T13:05:22.936324Z","shell.execute_reply.started":"2022-09-14T13:05:22.9321Z","shell.execute_reply":"2022-09-14T13:05:22.935311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(generate_train_images(train,100,100), steps_per_epoch = 10, epochs = 1)","metadata":{"execution":{"iopub.status.busy":"2022-09-14T11:42:32.966995Z","iopub.execute_input":"2022-09-14T11:42:32.967654Z","iopub.status.idle":"2022-09-14T12:07:51.513729Z","shell.execute_reply.started":"2022-09-14T11:42:32.967618Z","shell.execute_reply":"2022-09-14T12:07:51.510873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_test_images(img_size = (32, 32)):\n    path = '../input/rsna-2022-cervical-spine-fracture-detection/test_images'\n    test_images = []\n    test_ids = []\n    # test ids\n    ids = os.listdir('../input/rsna-2022-cervical-spine-fracture-detection/test_images')\n    \n    \n    for id_ in ids:\n        # file names of that id_\n        id_files = os.listdir(path + '/' + id_)\n        # read each of the dcm files\n        for i in range(len(id_files)):\n            image = pydicom.dcmread(path + '/' + id_ + '/' + id_files[i]).pixel_array\n            # resize image\n            image = np.array(cv2.resize(image, dsize=img_size))\n            # rescale image\n            image = (image - image.min())/image.max()\n            # add image to train_image\n            test_images.append(image)\n            # add id_ to test_ids\n            test_ids.append(id_)\n    \n    return np.array(test_images), np.array(test_ids)\n    ","metadata":{"execution":{"iopub.status.busy":"2022-09-14T12:07:51.519799Z","iopub.execute_input":"2022-09-14T12:07:51.520348Z","iopub.status.idle":"2022-09-14T12:07:51.534585Z","shell.execute_reply.started":"2022-09-14T12:07:51.520263Z","shell.execute_reply":"2022-09-14T12:07:51.533249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test the function\ntest_images, test_ids = get_test_images()","metadata":{"execution":{"iopub.status.busy":"2022-09-14T12:07:51.536402Z","iopub.execute_input":"2022-09-14T12:07:51.537412Z","iopub.status.idle":"2022-09-14T12:08:14.872386Z","shell.execute_reply.started":"2022-09-14T12:07:51.537373Z","shell.execute_reply":"2022-09-14T12:08:14.871194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results = model.predict(test_images)","metadata":{"execution":{"iopub.status.busy":"2022-09-14T12:08:14.87411Z","iopub.execute_input":"2022-09-14T12:08:14.87447Z","iopub.status.idle":"2022-09-14T12:08:15.15185Z","shell.execute_reply.started":"2022-09-14T12:08:14.87444Z","shell.execute_reply":"2022-09-14T12:08:15.150408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Prepare the submission file","metadata":{}},{"cell_type":"code","source":"results_df = pd.DataFrame(data=results, columns=cols)\nresults_df.insert(0, 'patient_id', test_ids)\nresults_df = results_df.melt(id_vars='patient_id', value_vars=cols, var_name='bone', value_name='fractured')\nresults_df.insert(0, 'row_id', results_df.patient_id + '_' + results_df.bone)\nresults_df.drop(['patient_id', 'bone'], axis=1, inplace=True)\nresults_df = pd.DataFrame(results_df.groupby(['row_id']).max().reset_index())\nresults_df.head(30)","metadata":{"execution":{"iopub.status.busy":"2022-09-14T12:08:15.154477Z","iopub.execute_input":"2022-09-14T12:08:15.155014Z","iopub.status.idle":"2022-09-14T12:08:15.212196Z","shell.execute_reply.started":"2022-09-14T12:08:15.154963Z","shell.execute_reply":"2022-09-14T12:08:15.210845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-09-14T12:08:15.214116Z","iopub.execute_input":"2022-09-14T12:08:15.214535Z","iopub.status.idle":"2022-09-14T12:08:15.235499Z","shell.execute_reply.started":"2022-09-14T12:08:15.214499Z","shell.execute_reply":"2022-09-14T12:08:15.234263Z"},"trusted":true},"execution_count":null,"outputs":[]}]}