{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-10-13T21:17:33.136173Z","iopub.execute_input":"2022-10-13T21:17:33.137312Z","iopub.status.idle":"2022-10-13T21:17:33.143386Z","shell.execute_reply.started":"2022-10-13T21:17:33.137245Z","shell.execute_reply":"2022-10-13T21:17:33.142328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nimport albumentations as Alb\nimport pydicom as dicom\nfrom warnings import filterwarnings\nfrom tensorflow import keras\n\nfilterwarnings(action='ignore', category=DeprecationWarning, message='`np.bool` is a deprecated alias')","metadata":{"execution":{"iopub.status.busy":"2022-10-13T21:17:33.145875Z","iopub.execute_input":"2022-10-13T21:17:33.14659Z","iopub.status.idle":"2022-10-13T21:17:33.154569Z","shell.execute_reply.started":"2022-10-13T21:17:33.146553Z","shell.execute_reply":"2022-10-13T21:17:33.153331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nbounding_boxes = pd.read_csv('../input/rsna-2022-cervical-spine-fracture-detection/train_bounding_boxes.csv')\nbounding_boxes","metadata":{"execution":{"iopub.status.busy":"2022-10-13T21:17:33.15632Z","iopub.execute_input":"2022-10-13T21:17:33.157068Z","iopub.status.idle":"2022-10-13T21:17:33.196409Z","shell.execute_reply.started":"2022-10-13T21:17:33.157032Z","shell.execute_reply":"2022-10-13T21:17:33.195362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_ids = list(bounding_boxes[\"StudyInstanceUID\"].unique())\nunique_ids","metadata":{"execution":{"iopub.status.busy":"2022-10-13T21:17:33.197854Z","iopub.execute_input":"2022-10-13T21:17:33.198504Z","iopub.status.idle":"2022-10-13T21:17:33.213036Z","shell.execute_reply.started":"2022-10-13T21:17:33.198466Z","shell.execute_reply":"2022-10-13T21:17:33.212061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_ids[0]","metadata":{"execution":{"iopub.status.busy":"2022-10-13T21:17:33.216034Z","iopub.execute_input":"2022-10-13T21:17:33.216699Z","iopub.status.idle":"2022-10-13T21:17:33.223297Z","shell.execute_reply.started":"2022-10-13T21:17:33.216661Z","shell.execute_reply":"2022-10-13T21:17:33.222341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import random\nlabels = []\nall_dcm_files = []\ndir = \"../input/rsna-csfd-256x256-jpg-dataset/train_images/\"\n\nfor uid in unique_ids:\n    patient_scans_with_fractures = bounding_boxes[bounding_boxes['StudyInstanceUID'] == uid]\n    scans_with_fractures = list(patient_scans_with_fractures['slice_number'])\n    files = []\n    scan_directory = dir + uid\n    neg_scans = []\n    count = 0\n    for file in os.listdir(scan_directory):\n        if file.endswith('.jpg'):\n            # the list of all dicom files in the patient folder\n            files.append(os.path.join(scan_directory, file))\n            # get the scan number and check if it is in the list of scans with fractures\n            scan_number = int(file.split('.')[0])\n            # if it is in the list, append a 1 to the list of labels otherwise append a 1\n            if scan_number in scans_with_fractures:\n                labels.append(1)\n                all_dcm_files.append(os.path.join(scan_directory, file))\n                count += 1\n            else:\n                # labels.append(0)   \n                neg_scans.append(os.path.join(scan_directory, file))# file shoul rep the full path\n    sublist = random.sample(neg_scans, k = count)\n    all_dcm_files.extend(sublist)\n    labels.extend([0] * count)\n    \nsublist","metadata":{"execution":{"iopub.status.busy":"2022-10-13T21:17:33.224938Z","iopub.execute_input":"2022-10-13T21:17:33.225773Z","iopub.status.idle":"2022-10-13T21:17:34.013389Z","shell.execute_reply.started":"2022-10-13T21:17:33.225573Z","shell.execute_reply":"2022-10-13T21:17:34.012225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(all_dcm_files)","metadata":{"execution":{"iopub.status.busy":"2022-10-13T21:17:34.014987Z","iopub.execute_input":"2022-10-13T21:17:34.016214Z","iopub.status.idle":"2022-10-13T21:17:34.023615Z","shell.execute_reply.started":"2022-10-13T21:17:34.01617Z","shell.execute_reply":"2022-10-13T21:17:34.022306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from collections import Counter\nCounter(labels)","metadata":{"execution":{"iopub.status.busy":"2022-10-13T21:17:34.025244Z","iopub.execute_input":"2022-10-13T21:17:34.025639Z","iopub.status.idle":"2022-10-13T21:17:34.039559Z","shell.execute_reply.started":"2022-10-13T21:17:34.025602Z","shell.execute_reply":"2022-10-13T21:17:34.038522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_dcm_files[0]","metadata":{"execution":{"iopub.status.busy":"2022-10-13T21:17:34.04114Z","iopub.execute_input":"2022-10-13T21:17:34.042094Z","iopub.status.idle":"2022-10-13T21:17:34.049466Z","shell.execute_reply.started":"2022-10-13T21:17:34.042056Z","shell.execute_reply":"2022-10-13T21:17:34.048348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\n\nfig, loc = plt.subplots(nrows = 1, ncols = 1, figsize=(10,10))\nfig.suptitle(\"imageshow\", weight = \"bold\", size=20)\n\nexample_jpg = mpimg.imread(all_dcm_files[14433])\nexample_plot = plt.imshow(example_jpg)","metadata":{"execution":{"iopub.status.busy":"2022-10-13T21:17:34.050953Z","iopub.execute_input":"2022-10-13T21:17:34.051467Z","iopub.status.idle":"2022-10-13T21:17:34.52268Z","shell.execute_reply.started":"2022-10-13T21:17:34.051432Z","shell.execute_reply":"2022-10-13T21:17:34.521547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from PIL import Image\nim = Image.open(\"../input/rsna-csfd-256x256-jpg-dataset/train_images/1.2.826.0.1.3680043.10051/130.jpg\")\n","metadata":{"execution":{"iopub.status.busy":"2022-10-13T21:17:34.524306Z","iopub.execute_input":"2022-10-13T21:17:34.527402Z","iopub.status.idle":"2022-10-13T21:17:34.561672Z","shell.execute_reply.started":"2022-10-13T21:17:34.52735Z","shell.execute_reply":"2022-10-13T21:17:34.557557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\ndef process_scan(scan_path):\n    try:\n        img = Image.open(scan_path)\n        data=np.array(img)\n        data= tf.expand_dims(data, axis=2)\n        return data\n    #         print(data[10]) #ensure we are not feeding empty images\n    #         data=data-np.min(data)\n    #         if np.max(data) != 0:\n    #             data=data/np.max(data)\n    #         data=(data*255).astype(np.uint8)        \n    #         return cv2.cvtColor(data.reshape(512, 512), cv2.COLOR_GRAY2RGB)\n    except:\n        return np.array(256,256,1)\n        \n    ","metadata":{"execution":{"iopub.status.busy":"2022-10-13T21:42:55.015637Z","iopub.execute_input":"2022-10-13T21:42:55.016007Z","iopub.status.idle":"2022-10-13T21:42:55.025538Z","shell.execute_reply.started":"2022-10-13T21:42:55.015975Z","shell.execute_reply":"2022-10-13T21:42:55.024552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"singleScan = process_scan(all_dcm_files[0])","metadata":{"execution":{"iopub.status.busy":"2022-10-13T21:19:56.142985Z","iopub.execute_input":"2022-10-13T21:19:56.143374Z","iopub.status.idle":"2022-10-13T21:19:56.150896Z","shell.execute_reply.started":"2022-10-13T21:19:56.143339Z","shell.execute_reply":"2022-10-13T21:19:56.149791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class DataGenerator(keras.utils.Sequence):\n    'Generates data for Keras'\n    def __init__(self, dicom_files, labels, batch_size, dim, n_classes, shuffle,  n_channels=1):\n        'Initialization'\n        self.dim = dim\n        self.batch_size = batch_size\n        self.labels = labels\n        self.dicom_files = dicom_files\n        self.n_classes = n_classes\n        self.n_channels = n_channels\n        self.shuffle = shuffle\n        self.on_epoch_end()\n\n    def __len__(self):\n        'Denotes the number of batches per epoch'\n        return int(np.floor(len(self.dicom_files) / self.batch_size))\n\n    def __getitem__(self, index):\n        'Generate one batch of data'\n        # Generate indexes of the batch\n        indexes = self.indexes[index*self.batch_size:(index+1)*self.batch_size]\n\n        dicom_files_temp = [self.dicom_files[k] for k in indexes]\n        labels_temp = [self.labels[k] for k in indexes]\n\n        # Generate data\n        X, y = self.__data_generation(dicom_files_temp, labels_temp)\n        \n        return X, y\n\n    def on_epoch_end(self):\n        'Updates indexes after each epoch'\n        self.indexes = np.arange(len(self.dicom_files))\n        if self.shuffle == True:\n            np.random.shuffle(self.indexes)\n\n    def __data_generation(self, dicom_files_temp, labels_temp):\n        'Generates data containing batch_size samples'\n        # Initialization\n        X = np.empty((self.batch_size, *self.dim, self.n_channels))\n        y = np.empty((self.batch_size), dtype=int)\n\n        # Generate data\n        for i, dicom_file in enumerate(dicom_files_temp):\n            X[i] = process_scan(dicom_file)\n            y[i] = labels_temp[i]\n\n        return X, y","metadata":{"execution":{"iopub.status.busy":"2022-10-13T21:37:16.171174Z","iopub.execute_input":"2022-10-13T21:37:16.172327Z","iopub.status.idle":"2022-10-13T21:37:16.187747Z","shell.execute_reply.started":"2022-10-13T21:37:16.172258Z","shell.execute_reply":"2022-10-13T21:37:16.186304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_dcm_files","metadata":{"execution":{"iopub.status.busy":"2022-10-13T22:09:18.330522Z","iopub.execute_input":"2022-10-13T22:09:18.330883Z","iopub.status.idle":"2022-10-13T22:09:18.354384Z","shell.execute_reply.started":"2022-10-13T22:09:18.330852Z","shell.execute_reply":"2022-10-13T22:09:18.352869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nn_classes = 1\nimg_width = 256\nimg_height = 256\n\nparams = {'dim': (img_width, img_height),\n          'batch_size': 4,\n          'n_classes': n_classes,\n          'shuffle': True}\n\n# x,listidx=np.unique(all_dcm_files,return_index=True)\n# indices = np.random.permutation(len(listidx))\n# training_idx, test_idx = indices[:round(0.8*len(indices))], indices[round(0.8*len(indices))+1:]\n\n# x = all_dcm_files\n# y = np.array(labels)\n# y.expand_dims(1)\n# x_train, x_val = x[training_idx], x[test_idx]\n# # THIS IS FOR LABELS\n# y_train, y_val = y[training_idx], y[test_idx]\n# # print(\"Number of samples in train and validation are %d and %d.\"\n# #     % (x_train.shape[0], x_val.shape[0])\n# #         )\n\nx_train, x_val, y_train, y_val = train_test_split(all_dcm_files, labels, test_size=0.2, random_state=42)\ntraining_generator = DataGenerator(x_train, y_train, **params)\nval_generator = DataGenerator(x_val, y_val, **params)\n","metadata":{"execution":{"iopub.status.busy":"2022-10-13T22:19:31.008368Z","iopub.execute_input":"2022-10-13T22:19:31.00873Z","iopub.status.idle":"2022-10-13T22:19:31.026543Z","shell.execute_reply.started":"2022-10-13T22:19:31.0087Z","shell.execute_reply":"2022-10-13T22:19:31.025183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x, y = training_generator.__getitem__(20)","metadata":{"execution":{"iopub.status.busy":"2022-10-13T21:37:28.037813Z","iopub.execute_input":"2022-10-13T21:37:28.038197Z","iopub.status.idle":"2022-10-13T21:37:28.076328Z","shell.execute_reply.started":"2022-10-13T21:37:28.038165Z","shell.execute_reply":"2022-10-13T21:37:28.07529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x.shape","metadata":{"execution":{"iopub.status.busy":"2022-10-14T00:11:47.296339Z","iopub.execute_input":"2022-10-14T00:11:47.29713Z","iopub.status.idle":"2022-10-14T00:11:47.323825Z","shell.execute_reply.started":"2022-10-14T00:11:47.297091Z","shell.execute_reply":"2022-10-14T00:11:47.322549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nprint(y)","metadata":{"execution":{"iopub.status.busy":"2022-10-13T21:43:11.437164Z","iopub.execute_input":"2022-10-13T21:43:11.437849Z","iopub.status.idle":"2022-10-13T21:43:11.44406Z","shell.execute_reply.started":"2022-10-13T21:43:11.437814Z","shell.execute_reply":"2022-10-13T21:43:11.442554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.applications.densenet import DenseNet121\nfrom keras.models import Model\nfrom tensorflow.keras import layers\nfrom tensorflow import keras\n\n\ndensenet = DenseNet121(\n    weights=None,\n    input_tensor=None,\n    include_top=False,\n    input_shape = (256,256,1),\n    classes=1\n)\n\noutput = densenet.layers[-1].output\noutput = keras.layers.Flatten()(output)\noutput = keras.layers.Dense(1,activation='sigmoid')(output)\n\ndensenet = Model(densenet.input, output)\n\ndensenet.summary()","metadata":{"execution":{"iopub.status.busy":"2022-10-13T21:47:21.894501Z","iopub.execute_input":"2022-10-13T21:47:21.894862Z","iopub.status.idle":"2022-10-13T21:47:24.447241Z","shell.execute_reply.started":"2022-10-13T21:47:21.89483Z","shell.execute_reply":"2022-10-13T21:47:24.44634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras import optimizers\n\n# class_weights = {0: 1.,\n#                  1: 9.76426264801}\n\nmetrics = [\n          keras.metrics.TruePositives(name='tp'),\n          keras.metrics.FalsePositives(name='fp'),\n          keras.metrics.TrueNegatives(name='tn'),\n          keras.metrics.FalseNegatives(name='fn'), \n          keras.metrics.BinaryAccuracy(name='accuracy'),\n          keras.metrics.Precision(name='precision'),\n          keras.metrics.Recall(name='recall'),\n          keras.metrics.AUC(name='auc'),\n          keras.metrics.AUC(name='prc', curve='PR'), # precision-recall curve\n        ]\n\nadam_opt = optimizers.Adam(learning_rate = 0.0001)\ndensenet.compile(\n    optimizer = adam_opt,\n    loss = \"binary_crossentropy\",\n    metrics=metrics,\n)\n\ncheckpoint_cb = keras.callbacks.ModelCheckpoint(\"2d_fracturedetector_OCT13.h5\", save_best_only=True)\nearly_stopping_cb = keras.callbacks.EarlyStopping(monitor=\"val_loss\", patience=10)\n\ndensenet.fit(\n    training_generator,\n    validation_data=val_generator,\n    #callbacks=[checkpoint_cb, early_stopping_cb, WandbCallback()],\n    callbacks=[checkpoint_cb, early_stopping_cb],\n    #class_weight=class_weights,\n    epochs=300\n)\n\n","metadata":{"execution":{"iopub.status.busy":"2022-10-13T22:24:43.289156Z","iopub.execute_input":"2022-10-13T22:24:43.28983Z","iopub.status.idle":"2022-10-14T00:11:47.294073Z","shell.execute_reply.started":"2022-10-13T22:24:43.289793Z","shell.execute_reply":"2022-10-14T00:11:47.293058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2022-10-14T06:11:42.389121Z","iopub.execute_input":"2022-10-14T06:11:42.390112Z","iopub.status.idle":"2022-10-14T06:11:42.597926Z","shell.execute_reply.started":"2022-10-14T06:11:42.390071Z","shell.execute_reply":"2022-10-14T06:11:42.596482Z"},"trusted":true},"execution_count":null,"outputs":[]}]}