{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nos.chdir(\"../input/rsna-2022-cervical-spine-fracture-detection\")\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-10-17T22:56:31.726654Z","iopub.execute_input":"2022-10-17T22:56:31.727103Z","iopub.status.idle":"2022-10-17T22:56:32.128481Z","shell.execute_reply.started":"2022-10-17T22:56:31.727025Z","shell.execute_reply":"2022-10-17T22:56:32.127242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('train.csv')","metadata":{"execution":{"iopub.status.busy":"2022-10-17T22:56:32.717348Z","iopub.execute_input":"2022-10-17T22:56:32.71772Z","iopub.status.idle":"2022-10-17T22:56:32.726616Z","shell.execute_reply.started":"2022-10-17T22:56:32.717692Z","shell.execute_reply":"2022-10-17T22:56:32.725426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Findings\n1. The \"patient_overall\" tag corresponds perfectly with whether there's a spinal fracture in 1 or more of the columns\n2. There's roughly 400 patients who have two fractures or more (a significant proportion of the 2k total patients)","metadata":{}},{"cell_type":"code","source":"!pip install gdcm pylibjpeg pylibjpeg-libjpeg","metadata":{"execution":{"iopub.status.busy":"2022-10-17T22:56:10.593429Z","iopub.execute_input":"2022-10-17T22:56:10.593826Z","iopub.status.idle":"2022-10-17T22:56:23.550565Z","shell.execute_reply.started":"2022-10-17T22:56:10.593796Z","shell.execute_reply":"2022-10-17T22:56:23.549379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMAGE_FOLDER = 'train_images/'\n\nimport pydicom\nfrom pydicom import dcmread\nfrom pydicom.data import get_testdata_file\n\ndef get_image_slices(name: str):\n    image_slices = []\n    for image_path in sorted(os.listdir(IMAGE_FOLDER + name), key = lambda x: int(x.split('.')[0])):\n        ds = dcmread(IMAGE_FOLDER + name + \"/\" + image_path)\n        image_slice = ds.pixel_array; image_slices.append(image_slice)\n    return np.array(image_slices)\n\ndef get_metadata(name: str):\n    for image_path in sorted(os.listdir(IMAGE_FOLDER + name), key = lambda x: int(x.split('.')[0])):\n        ds = dcmread(IMAGE_FOLDER + name + \"/\" + image_path); break;\n    return (str(ds.file_meta), ds.WindowWidth, ds.WindowCenter, ds.PixelSpacing, ds.SliceThickness, ds.Rows, ds.Columns)","metadata":{"execution":{"iopub.status.busy":"2022-10-17T22:56:35.578275Z","iopub.execute_input":"2022-10-17T22:56:35.578865Z","iopub.status.idle":"2022-10-17T22:56:35.767949Z","shell.execute_reply.started":"2022-10-17T22:56:35.578827Z","shell.execute_reply":"2022-10-17T22:56:35.766951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Metadata is basically identical for a scan of a single patient. But are some constants (window center, slice thickness, scan size) the same for all patients?","metadata":{}},{"cell_type":"code","source":"import tqdm\ncollected_meta = []\nfor i in tqdm.trange(len(train_df)):\n    collected_meta.append(get_metadata(train_df['StudyInstanceUID'][i]))\nmetadata = pd.DataFrame()\nmetadata[\"Filedata\"], metadata[\"Window Width\"], metadata['Window Level'], metadata[\"Pixel Spacing\"], metadata[\"Slice Thickness\"], metadata[\"Rows\"], metadata[\"Columns\"] = zip(*collected_meta)","metadata":{"execution":{"iopub.status.busy":"2022-10-17T22:56:37.182943Z","iopub.execute_input":"2022-10-17T22:56:37.183275Z","iopub.status.idle":"2022-10-17T22:58:22.564658Z","shell.execute_reply.started":"2022-10-17T22:56:37.183247Z","shell.execute_reply":"2022-10-17T22:58:22.563621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metadata[\"Pixel Spacing X\"] = [i[0] for i in metadata[\"Pixel Spacing\"]]\nmetadata[\"Pixel Spacing Y\"] = [i[1] for i in metadata[\"Pixel Spacing\"]]\ndef parse_window(X):\n    if (type(X) == pydicom.multival.MultiValue): return abs(X[0]-X[1])\n    else: return X\nmetadata[\"Window Width\"] = [parse_window(i) for i in metadata[\"Window Width\"]]\nmetadata[\"Window Level\"] = [parse_window(i) for i in metadata[\"Window Level\"]]\nmetadata.drop(\"Pixel Spacing\", axis=1, inplace = True)\nmetadata[\"Volume\"] = [metadata[\"Pixel Spacing X\"][i] * metadata[\"Pixel Spacing Y\"][i] * metadata[\"Rows\"][i] * metadata[\"Columns\"][i] * metadata[\"Slice Thickness\"][i] for i in metadata.index]","metadata":{"execution":{"iopub.status.busy":"2022-10-17T22:58:22.566937Z","iopub.execute_input":"2022-10-17T22:58:22.567188Z","iopub.status.idle":"2022-10-17T22:58:22.649393Z","shell.execute_reply.started":"2022-10-17T22:58:22.567164Z","shell.execute_reply":"2022-10-17T22:58:22.648493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metadata","metadata":{"execution":{"iopub.status.busy":"2022-10-16T08:29:07.617756Z","iopub.execute_input":"2022-10-16T08:29:07.6182Z","iopub.status.idle":"2022-10-16T08:29:07.643232Z","shell.execute_reply.started":"2022-10-16T08:29:07.618165Z","shell.execute_reply":"2022-10-16T08:29:07.642314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axs = plt.subplots(2, 3, figsize=(15, 8))\nsns.histplot(metadata[\"Window Width\"],kde=True, ax = axs[0][0])\nsns.histplot(metadata[\"Window Level\"], kde = True, ax = axs[0][1])\nsns.histplot(metadata[\"Slice Thickness\"], kde=True, ax=axs[0][2])\nsns.histplot(metadata[\"Pixel Spacing X\"], kde=True, ax=axs[1][0])\nsns.histplot(metadata[\"Pixel Spacing Y\"], kde=True, ax=axs[1][1])\nsns.histplot(metadata[\"Volume\"], kde=True, ax=axs[1][2])","metadata":{"execution":{"iopub.status.busy":"2022-10-16T08:29:17.743935Z","iopub.execute_input":"2022-10-16T08:29:17.745046Z","iopub.status.idle":"2022-10-16T08:29:19.342577Z","shell.execute_reply.started":"2022-10-16T08:29:17.744999Z","shell.execute_reply":"2022-10-16T08:29:19.341074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"set(metadata[\"Rows\"]), set(metadata[\"Columns\"])","metadata":{"execution":{"iopub.status.busy":"2022-10-16T07:18:00.237906Z","iopub.execute_input":"2022-10-16T07:18:00.238345Z","iopub.status.idle":"2022-10-16T07:18:00.246501Z","shell.execute_reply.started":"2022-10-16T07:18:00.238309Z","shell.execute_reply":"2022-10-16T07:18:00.24515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sum(metadata[\"Rows\"] == 768), sum(metadata[\"Columns\"] == 519), sum(metadata[\"Columns\"] == 768)","metadata":{"execution":{"iopub.status.busy":"2022-10-16T07:23:04.961041Z","iopub.execute_input":"2022-10-16T07:23:04.96147Z","iopub.status.idle":"2022-10-16T07:23:04.972179Z","shell.execute_reply.started":"2022-10-16T07:23:04.961424Z","shell.execute_reply":"2022-10-16T07:23:04.970689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Looks like there are images that have different widths and levels as well as different sizes. That's not a problem, we can rescale and transform our volume to a standard scale.\n\nI wonder... do the level and spacing of the images correlate significantly with the target outcomes? ","metadata":{}},{"cell_type":"code","source":"combined_df = pd.concat([train_df, metadata], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-10-16T08:29:27.730296Z","iopub.execute_input":"2022-10-16T08:29:27.730725Z","iopub.status.idle":"2022-10-16T08:29:27.740707Z","shell.execute_reply.started":"2022-10-16T08:29:27.730689Z","shell.execute_reply":"2022-10-16T08:29:27.739225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(15,8))\nsns.heatmap(combined_df.corr(), annot=True, ax = ax)","metadata":{"execution":{"iopub.status.busy":"2022-10-16T08:29:27.993665Z","iopub.execute_input":"2022-10-16T08:29:27.994075Z","iopub.status.idle":"2022-10-16T08:29:29.64011Z","shell.execute_reply.started":"2022-10-16T08:29:27.994042Z","shell.execute_reply":"2022-10-16T08:29:29.638898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# What about monotonic relationships?\nfig, ax = plt.subplots(figsize=(15,8))\nsns.heatmap(combined_df.corr(method = \"spearman\"), annot=True, ax = ax)","metadata":{"execution":{"iopub.status.busy":"2022-10-16T08:30:52.932267Z","iopub.execute_input":"2022-10-16T08:30:52.93278Z","iopub.status.idle":"2022-10-16T08:30:54.58085Z","shell.execute_reply.started":"2022-10-16T08:30:52.932738Z","shell.execute_reply":"2022-10-16T08:30:54.579664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Good, they do not (there'd be a serious problem if they did!!). We can also observe that the pixel spacing in the X axis correlates perfectly with the pixel spacing in the Y axis - in this case, it's because they're always equal.","metadata":{}},{"cell_type":"code","source":"from os.path import dirname, join\nfrom pprint import pprint\nimport numpy as np\nimport ipywidgets as ipyw\nimport matplotlib.pyplot as plt\nimport nibabel as nib\nfrom skimage.transform import resize\nimport cv2\n\nclass ImageSliceViewer3D:\n    \"\"\" \n    ImageSliceViewer3D is for viewing volumetric image slices in jupyter or\n    ipython notebooks. \n    \n    User can interactively change the slice plane selection for the image and \n    the slice plane being viewed.\nArguments:\n    Volume = 3D input image\n    figsize = default(8,8), to set the size of the figure\n    cmap = default('gray'), string for the matplotlib colormap. You can find \n    more matplotlib colormaps on the following link:\n    https://matplotlib.org/users/colormaps.html\n    \n    \"\"\"\n    \n    def __init__(self, volume, figsize=(20,20), cmap='gray'):\n        self.volume = volume\n        self.figsize = figsize\n        self.cmap = cmap\n        self.v = [np.min(volume), np.max(volume)]\n        \n        # Call to select slice plane\n        ipyw.interact(self.views)\n    \n    def views(self):\n        self.vol1 = np.transpose(self.volume, [1,2,0])\n        self.vol2 = np.rot90(np.transpose(self.volume, [2,0,1]), 3) #rotate 270 degrees\n        self.vol3 = np.transpose(self.volume, [0,1,2])\n        maxZ1 = self.vol1.shape[2] - 1\n        maxZ2 = self.vol2.shape[2] - 1\n        maxZ3 = self.vol3.shape[2] - 1\n        ipyw.interact(self.plot_slice, \n            z1=ipyw.IntSlider(min=0, max=maxZ1, step=1, continuous_update=False, \n            description='Axial:'), \n            z2=ipyw.IntSlider(min=0, max=maxZ2, step=1, continuous_update=False, \n            description='Coronal:'),\n            z3=ipyw.IntSlider(min=0, max=maxZ3, step=1, continuous_update=False, \n            description='Sagittal:'))\n    def plot_slice(self, z1, z2, z3):\n            # Plot slice for the given plane and slice\n            f,ax = plt.subplots(1,3, figsize=self.figsize)\n            #print(self.figsize)\n            #self.fig = plt.figure(figsize=self.figsize)\n            #f(figsize = self.figsize)\n            ax[0].imshow(self.vol1[:,:,z1], cmap=plt.get_cmap(self.cmap), \n                vmin=self.v[0], vmax=self.v[1])\n            ax[1].imshow(self.vol2[:,:,z2], cmap=plt.get_cmap(self.cmap), \n                vmin=self.v[0], vmax=self.v[1])\n            sagittal_projection = resize(self.vol3[:,:,z3], (700, 500), preserve_range = True)\n            #sagittal_projection /= ((np.max(sagittal_projection) - np.min(sagittal_projection)) / 255)\n            #sagittal_projection -= np.min(sagittal_projection)\n            #sagittal_projection = cv2.filter2D(sagittal_projection, -1, kernel)\n            ax[2].imshow(sagittal_projection, cmap=plt.get_cmap(self.cmap), \n                vmin=self.v[0], vmax=self.v[1])\n            plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-16T07:37:25.969444Z","iopub.execute_input":"2022-10-16T07:37:25.969883Z","iopub.status.idle":"2022-10-16T07:37:26.551017Z","shell.execute_reply.started":"2022-10-16T07:37:25.969851Z","shell.execute_reply":"2022-10-16T07:37:26.549694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vol = get_image_slices(train_df['StudyInstanceUID'][0])","metadata":{"execution":{"iopub.status.busy":"2022-10-16T07:37:27.925979Z","iopub.execute_input":"2022-10-16T07:37:27.92638Z","iopub.status.idle":"2022-10-16T07:37:33.479843Z","shell.execute_reply.started":"2022-10-16T07:37:27.926348Z","shell.execute_reply":"2022-10-16T07:37:33.478559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"slicer = ImageSliceViewer3D(vol)","metadata":{"execution":{"iopub.status.busy":"2022-10-16T07:37:33.483952Z","iopub.execute_input":"2022-10-16T07:37:33.485117Z","iopub.status.idle":"2022-10-16T07:37:34.230265Z","shell.execute_reply.started":"2022-10-16T07:37:33.485064Z","shell.execute_reply":"2022-10-16T07:37:34.22886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.histplot(np.random.choice(vol.flatten(), 1000000))","metadata":{"execution":{"iopub.status.busy":"2022-10-16T00:09:58.017658Z","iopub.execute_input":"2022-10-16T00:09:58.018077Z","iopub.status.idle":"2022-10-16T00:10:00.0986Z","shell.execute_reply.started":"2022-10-16T00:09:58.018042Z","shell.execute_reply":"2022-10-16T00:10:00.097443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.boxplot(np.random.choice(vol.flatten(), 1000000))","metadata":{"execution":{"iopub.status.busy":"2022-10-16T00:10:22.20821Z","iopub.execute_input":"2022-10-16T00:10:22.208998Z","iopub.status.idle":"2022-10-16T00:10:22.511578Z","shell.execute_reply.started":"2022-10-16T00:10:22.208961Z","shell.execute_reply":"2022-10-16T00:10:22.510321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There's a crap ton of outliers!!!\nHere's some reading on what the pixel values mean. Normally, pixel values are cropped between 0 and 255 or 0 and 1; however, DICOM voxel values use the [Hounsfield scale](https://en.wikipedia.org/wiki/Hounsfield_scale), which can range from -1000 to 10000. However, the outliers aren't informative - they probably rather describe errors from the scanner. A Hounsfield value of -1000 denotes air, so any values less than -1000 denote parts of the body with a density less than air and should be removed (really? is there a vacuum somewhere in your head?).\n\nLikewise, Hounsfield values of 5000 or above denote heavy metals (gold/steel/brass/copper), which can come from implants but ultimately aren't useful in the detection of broken bones (bones should have a value of around 2000). Let's crop the Hounsfield values of the voxels to something more reasonable.","metadata":{}},{"cell_type":"code","source":"def clip_outliers(data, m=2):\n    data_mean = np.median(data)\n    std = np.std(data)\n    return np.clip(data, data_mean - m * std, data_mean + m * std)\nvol_no_outliers = clip_outliers(vol)","metadata":{"execution":{"iopub.status.busy":"2022-10-16T07:37:37.527515Z","iopub.execute_input":"2022-10-16T07:37:37.528007Z","iopub.status.idle":"2022-10-16T07:37:38.733711Z","shell.execute_reply.started":"2022-10-16T07:37:37.52797Z","shell.execute_reply":"2022-10-16T07:37:38.732456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"slice_no_outliers = ImageSliceViewer3D(vol_no_outliers)","metadata":{"execution":{"iopub.status.busy":"2022-10-16T07:37:39.693873Z","iopub.execute_input":"2022-10-16T07:37:39.69432Z","iopub.status.idle":"2022-10-16T07:37:40.434735Z","shell.execute_reply.started":"2022-10-16T07:37:39.694273Z","shell.execute_reply":"2022-10-16T07:37:40.43367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axs = plt.subplots(2,1, figsize=(15, 6))\nsns.histplot(np.random.choice(vol_no_outliers.flatten(), 1000000), ax = axs[0])\nsns.boxplot(np.random.choice(vol_no_outliers.flatten(), 1000000), ax = axs[1])","metadata":{"execution":{"iopub.status.busy":"2022-10-16T07:39:28.006526Z","iopub.execute_input":"2022-10-16T07:39:28.006974Z","iopub.status.idle":"2022-10-16T07:39:29.818307Z","shell.execute_reply.started":"2022-10-16T07:39:28.006939Z","shell.execute_reply":"2022-10-16T07:39:29.816984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"That's much better. Four distinct modalities appear, the first at -2200, the second at -1000, the third around 0, and the fourth at 700. My hypothesis is that the first probably corresponds to the empty space outside the skull or \"crop space\" (which is usually infilled by a large negative value). The second is probably air, the third water / flesh, and the fourth bone. Let's color the data by cluster and take a look.","metadata":{}},{"cell_type":"code","source":"CHANNELS = [(-5000, -2000), (-2000, -500), (-500, 500), (500, 1000)]\nvol_channels = []\nfor channel in CHANNELS:\n    tmp = vol_no_outliers.copy()\n    criterion = (tmp >= channel[0]) & (tmp < channel[1])\n    tmp -= np.mean(tmp); tmp /= np.std(tmp)\n    tmp[np.invert(criterion)] = -3\n    vol_channels.append(tmp)\nvol_channels = np.array(vol_channels)","metadata":{"execution":{"iopub.status.busy":"2022-10-16T08:03:13.854474Z","iopub.execute_input":"2022-10-16T08:03:13.855696Z","iopub.status.idle":"2022-10-16T08:03:18.293966Z","shell.execute_reply.started":"2022-10-16T08:03:13.855644Z","shell.execute_reply":"2022-10-16T08:03:18.292746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ImageSliceViewer3D(vol_channels[0])","metadata":{"execution":{"iopub.status.busy":"2022-10-16T08:03:45.153332Z","iopub.execute_input":"2022-10-16T08:03:45.153774Z","iopub.status.idle":"2022-10-16T08:03:45.844822Z","shell.execute_reply.started":"2022-10-16T08:03:45.153737Z","shell.execute_reply":"2022-10-16T08:03:45.843294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ImageSliceViewer3D(vol_channels[1])","metadata":{"execution":{"iopub.status.busy":"2022-10-16T08:03:56.026635Z","iopub.execute_input":"2022-10-16T08:03:56.027095Z","iopub.status.idle":"2022-10-16T08:03:56.735618Z","shell.execute_reply.started":"2022-10-16T08:03:56.027054Z","shell.execute_reply":"2022-10-16T08:03:56.734125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ImageSliceViewer3D(vol_channels[2])","metadata":{"execution":{"iopub.status.busy":"2022-10-16T08:04:29.957167Z","iopub.execute_input":"2022-10-16T08:04:29.957968Z","iopub.status.idle":"2022-10-16T08:04:30.667013Z","shell.execute_reply.started":"2022-10-16T08:04:29.957928Z","shell.execute_reply":"2022-10-16T08:04:30.665651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ImageSliceViewer3D(vol_channels[3])","metadata":{"execution":{"iopub.status.busy":"2022-10-16T08:04:38.132056Z","iopub.execute_input":"2022-10-16T08:04:38.132465Z","iopub.status.idle":"2022-10-16T08:04:38.817656Z","shell.execute_reply.started":"2022-10-16T08:04:38.132431Z","shell.execute_reply":"2022-10-16T08:04:38.816847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Okay, let's not feed the entire volume into the model. Let's find the sagittal plane slices that show the best views of the spinal column.","metadata":{}},{"cell_type":"code","source":"transposed = np.transpose(vol_channels[3], [0,1,2])","metadata":{"execution":{"iopub.status.busy":"2022-10-16T08:38:21.661795Z","iopub.execute_input":"2022-10-16T08:38:21.662263Z","iopub.status.idle":"2022-10-16T08:38:21.669191Z","shell.execute_reply.started":"2022-10-16T08:38:21.662227Z","shell.execute_reply":"2022-10-16T08:38:21.667834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sums = np.sum(transposed, axis=(0, 1))\nmost_bones = sorted(range(len(sums)), key = lambda x: -sums[x])\nfig, axs = plt.subplots(5, 5, figsize = (15, 25))\nfor i in range(5):\n    for j in range(5):\n        sagittal_projection = resize(transposed[:,:,most_bones[i*5+j]], (700, 500), preserve_range = True)\n        axs[i][j].imshow(sagittal_projection, cmap='gray')","metadata":{"execution":{"iopub.status.busy":"2022-10-16T08:44:52.488869Z","iopub.execute_input":"2022-10-16T08:44:52.489302Z","iopub.status.idle":"2022-10-16T08:44:57.061243Z","shell.execute_reply.started":"2022-10-16T08:44:52.489263Z","shell.execute_reply":"2022-10-16T08:44:57.06037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"What if we take the median of the 49 samples showing the most bone and then sample the slices around them? ","metadata":{}},{"cell_type":"code","source":"sums = np.sum(transposed, axis=(0, 1))\nmost_bones = sorted(range(len(sums)), key = lambda x: -sums[x])\nmedian = np.around(np.median(most_bones[:49])).astype(np.int); \nslices_to_select = range(median-10, median+10)\nfig, axs = plt.subplots(4, 5, figsize = (15, 15))\nfor i in range(4):\n    for j in range(5):\n        sagittal_projection = resize(transposed[:,:,slices_to_select[i*5+j]], (700, 500), preserve_range = True)\n        axs[i][j].imshow(sagittal_projection, cmap='gray')","metadata":{"execution":{"iopub.status.busy":"2022-10-16T08:50:14.121464Z","iopub.execute_input":"2022-10-16T08:50:14.121904Z","iopub.status.idle":"2022-10-16T08:50:18.664869Z","shell.execute_reply.started":"2022-10-16T08:50:14.121859Z","shell.execute_reply":"2022-10-16T08:50:18.663683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Hmm, okay. The first approach - using a mixture of influential slices - will lead to more robustness and a greater diversity of information. The downside is that it destroys important spacial information about the slices (using the second approach, we know that the slices will be one over another).\n\nI think it's probably best to use the first approach for a 2d image classifier approach, and the second for a narrow slice 3d volume classifier approach. We can try both.","metadata":{}},{"cell_type":"code","source":"sns.heatmap(train_df.drop([\"StudyInstanceUID\",\"patient_overall\"], axis=1).corr(),annot=True)","metadata":{"execution":{"iopub.status.busy":"2022-10-16T00:10:57.268812Z","iopub.execute_input":"2022-10-16T00:10:57.269219Z","iopub.status.idle":"2022-10-16T00:10:57.808279Z","shell.execute_reply.started":"2022-10-16T00:10:57.269183Z","shell.execute_reply":"2022-10-16T00:10:57.806985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Males are affected more commonly than females with a median age of injury of 56 years. Falls, motor vehicle collisions, pedestrian accidents, cycling and diving are common causes of injury [: reference](https://radiopaedia.org/articles/cervical-spine-fractures?lang=us)\n\nIt also seems like a few fractures seem to occur together. Adjacent neck vertebra from C3 to C6 tend to fracture together (e.g if C4 fractures, C5 has a 34% chance of also being fractured).","metadata":{}},{"cell_type":"code","source":"bb_df = pd.read_csv('train_bounding_boxes.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bb_df[bb_df['StudyInstanceUID'] == \"1.2.826.0.1.3680043.10051\"]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta[30]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['multi_input'] = (train_df['C1'] + train_df['C2'] + train_df['C3'] + train_df['C4'] + train_df['C5'] + train_df['C6'] + train_df['C7'])\nprint(train_df['multi_input'].describe())\nsns.histplot(train_df['multi_input'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}