{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#install pydicom requirements\n\n%load_ext autoreload\n%autoreload 2\n!conda install '/kaggle/input/pydicom-conda-helper/libjpeg-turbo-2.1.0-h7f98852_0.tar.bz2' --offline -y\n!conda install '/kaggle/input/pydicom-conda-helper/libgcc-ng-9.3.0-h2828fa1_19.tar.bz2' --offline -y\n#!conda install '/kaggle/input/pydicom-conda-helper/gdcm-2.8.9-py37h500ead1_1.Truetar.bz2' --offline -y \n!cp ../input/gdcm-conda-install/gdcm.tar .\n!tar -xvzf gdcm.tar\n!conda install --offline ./gdcm/gdcm-2.8.9-py37h71b2a6d_0.tar.bz2\n!conda install '/kaggle/input/pydicom-conda-helper/conda-4.10.1-py37h89c1867_0.tar.bz2' --offline -y\n!conda install '/kaggle/input/pydicom-conda-helper/certifi-2020.12.5-py37h89c1867_1.tar.bz2' --offline -y\n!conda install '/kaggle/input/pydicom-conda-helper/openssl-1.1.1k-h7f98852_0.tar.bz2' --offline -y","metadata":{"execution":{"iopub.status.busy":"2022-08-23T00:54:58.525314Z","iopub.execute_input":"2022-08-23T00:54:58.525893Z","iopub.status.idle":"2022-08-23T00:55:48.308533Z","shell.execute_reply.started":"2022-08-23T00:54:58.525824Z","shell.execute_reply":"2022-08-23T00:55:48.307812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport glob\nimport random\nimport collections\nimport gc\nimport math\n\nimport numpy as np\nimport pandas as pd\n\nfrom joblib import Parallel, delayed\nfrom tqdm.notebook import tqdm\n\nimport pydicom\nimport matplotlib.pyplot as plt\nimport cv2\nimport scipy\nimport tensorflow as tf\nimport tensorflow_addons as tfa\nfrom tensorflow.keras import backend as K\nfrom tensorflow import keras\nfrom tensorflow.keras import layers as L\n\nfrom sklearn.model_selection import KFold, StratifiedKFold","metadata":{"execution":{"iopub.status.busy":"2022-08-23T00:55:48.309788Z","iopub.execute_input":"2022-08-23T00:55:48.310037Z","iopub.status.idle":"2022-08-23T00:55:54.037436Z","shell.execute_reply.started":"2022-08-23T00:55:48.310013Z","shell.execute_reply":"2022-08-23T00:55:54.036423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#set desired image size and depth (number of patient's images to load)\nclass Config:\n    img_size = 256\n    depth = 128\n    train_one_fold = True\n\n\n\nIMG_PATH_TRAIN = '../input/rsna-2022-cervical-spine-fracture-detection/train_images/'\nIMG_PATH_TEST = '../input/rsna-2022-cervical-spine-fracture-detection/test_images/'\nTRAIN_CSV_PATH = '../input/rsna-2022-cervical-spine-fracture-detection/train.csv'\nTEST_CSV_PATH = '../input/rsna-2022-cervical-spine-fracture-detection/test.csv'\n\ntrain_images = os.listdir(IMG_PATH_TRAIN)\ntest_images = os.listdir(IMG_PATH_TEST)\n\ntrain=pd.read_csv(TRAIN_CSV_PATH)\ntest=pd.read_csv(TEST_CSV_PATH)","metadata":{"execution":{"iopub.status.busy":"2022-08-23T00:55:54.038705Z","iopub.execute_input":"2022-08-23T00:55:54.039288Z","iopub.status.idle":"2022-08-23T00:55:54.246851Z","shell.execute_reply.started":"2022-08-23T00:55:54.039263Z","shell.execute_reply":"2022-08-23T00:55:54.246105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_dicom(path):\n    dicom = pydicom.read_file(path)\n    data = dicom.pixel_array\n    data = data - np.min(data)\n    if np.max(data) != 0:\n        data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n    data = cv2.resize(data, (Config.img_size,Config.img_size), interpolation = cv2.INTER_AREA)\n    return data\n     \n\ndef load_dicom_line_par(path, indices:list = None):\n    t_paths = sorted(glob.glob(os.path.join(path, \"*\")),\n       key=lambda x: int(x.split('/')[-1].split(\".\")[0]))\n    \n    if indices is not None:\n        t_paths = [t_paths[i] for i in indices]\n        \n    images = Parallel(n_jobs=-1)(delayed(load_dicom)(filename) for filename in t_paths)\n    \n    return np.array(images)","metadata":{"execution":{"iopub.status.busy":"2022-08-23T00:55:54.248319Z","iopub.execute_input":"2022-08-23T00:55:54.248865Z","iopub.status.idle":"2022-08-23T00:55:54.298354Z","shell.execute_reply.started":"2022-08-23T00:55:54.248842Z","shell.execute_reply":"2022-08-23T00:55:54.297709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_output_path = './train_arrays/'\ntest_output_path = './test_arrays/'\n\nif not os.path.exists(train_output_path): os.mkdir(train_output_path)\nif not os.path.exists(test_output_path): os.mkdir(test_output_path)    \n    \ntest_patients = sorted(os.listdir(IMG_PATH_TEST))\n\ndef save_3d_voxels(dicom_path, output_path):\n    \n    n_scans=len(os.listdir(dicom_path))\n    \n    #instead of zooming whole dicom series, load only part of the images\n    ind = np.quantile(list(range(n_scans)), np.linspace(0.1, 0.9, Config.depth)).astype(int)\n    image = load_dicom_line_par(dicom_path, indices = ind)\n    \n    if image.ndim <4:\n        image = np.expand_dims(image, -1)\n    \n    np.save(f\"{output_path}{dicom_path.split('/')[-1]}.npy\", image)\n    \n    del image\n    return None\n \n\nfor i in tqdm(range(len(test_patients))):\n    case = IMG_PATH_TEST + test_patients[i]\n    save_3d_voxels(case, test_output_path)\n    \ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-23T00:55:54.300165Z","iopub.execute_input":"2022-08-23T00:55:54.300657Z","iopub.status.idle":"2022-08-23T00:55:58.340712Z","shell.execute_reply.started":"2022-08-23T00:55:54.300583Z","shell.execute_reply":"2022-08-23T00:55:58.339688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.DataFrame({'StudyInstanceUID': test_patients})\n\ntest['StudyInstanceUID'] = test_patients\ntest['numpy_path'] = test['StudyInstanceUID'].apply(lambda x: f'{test_output_path}{x}.npy')\ntest","metadata":{"execution":{"iopub.status.busy":"2022-08-23T00:55:58.342465Z","iopub.execute_input":"2022-08-23T00:55:58.342822Z","iopub.status.idle":"2022-08-23T00:55:58.41191Z","shell.execute_reply.started":"2022-08-23T00:55:58.342789Z","shell.execute_reply":"2022-08-23T00:55:58.410724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class SampleGenerator(tf.keras.utils.Sequence):\n    def __init__(self, df: pd.DataFrame, batch_size, resample_rate: float = None, steps_per_epoch: int = 10000, is_train=True, shuffle=True):\n        self.is_train      = is_train\n        self.numpy_path  = df.numpy_path\n        self.df  = df\n        self.batch_size = batch_size\n        self.length = len(df)\n        self.resample = resample_rate\n        self.shuffle = shuffle\n        self.steps_per_epoch= steps_per_epoch\n        \n    def __len__(self):\n        return  min(int(np.ceil(self.length / float(self.batch_size))), self.steps_per_epoch)\n    \n    def on_epoch_end(self):\n        if self.shuffle:\n            self.df = self.df.sample(frac=1).reset_index(drop=True)\n            self.numpy_path  = self.df.numpy_path\n    \n    def __getitem__(self, index):\n                  \n        if self.is_train:         \n            \n            batch_x = []\n            batch_y = []\n            \n            targets = self.df[['patient_overall', 'C1', 'C2', 'C3', 'C4', 'C5', 'C6', 'C7']]\n            \n            for i in range(self.batch_size):\n                cur_ind = self.batch_size*index + i\n                if cur_ind < self.length:\n                    batch_x.append(np.load(self.numpy_path.iloc[cur_ind]))\n                    batch_y.append(targets.iloc[cur_ind])\n              \n            if self.resample is not None:\n                n_images = batch_x[0].shape[0]\n                im_ids = sorted(np.random.choice(list(range(n_images)), int(n_images * self.resample), replace=False))\n                batch_x = np.array(batch_x)[:,im_ids]\n                   \n            #return np.array(batch_x), np.expand_dims(np.array(batch_y), -1).astype(np.float32)\n            return np.array(batch_x), np.array(batch_y).astype(np.float32)\n\n        else:\n            batch_x = []\n            for i in range(self.batch_size):\n                cur_ind = self.batch_size*index + i\n                if cur_ind < self.length:\n                    batch_x.append(np.load(self.numpy_path.iloc[cur_ind]))\n            \n            return np.array(batch_x)","metadata":{"execution":{"iopub.status.busy":"2022-08-23T00:55:58.413255Z","iopub.execute_input":"2022-08-23T00:55:58.413608Z","iopub.status.idle":"2022-08-23T00:55:58.471574Z","shell.execute_reply.started":"2022-08-23T00:55:58.413574Z","shell.execute_reply":"2022-08-23T00:55:58.470683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def competiton_loss(y_true, y_pred):\n\n    competition_weights = {\n        '-' : tf.constant([7, 1, 1, 1, 1, 1, 1, 1], dtype=tf.float32),\n        '+' : tf.constant([14, 2, 2, 2, 2, 2, 2, 2], dtype=tf.float32)\n    }\n    \n    loss = tf.keras.losses.BinaryCrossentropy(reduction=tf.keras.losses.Reduction.NONE)(tf.expand_dims(y_true, -1),tf.expand_dims(y_pred,-1))\n    weights  = y_true*competition_weights['+'] + (1-y_true)*competition_weights['-'] \n    \n    loss = tf.reduce_mean(tf.reduce_sum(loss * weights, axis=1)) / tf.reduce_sum(weights)\n    return loss\n    ","metadata":{"execution":{"iopub.status.busy":"2022-08-23T00:56:15.328809Z","iopub.execute_input":"2022-08-23T00:56:15.329124Z","iopub.status.idle":"2022-08-23T00:56:15.381373Z","shell.execute_reply.started":"2022-08-23T00:56:15.329099Z","shell.execute_reply":"2022-08-23T00:56:15.379925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = tf.keras.models.load_model('../input/seresnet-weights/seresnet50_best_fold_1.hdf5', custom_objects = {'competiton_loss': competiton_loss})","metadata":{"execution":{"iopub.status.busy":"2022-08-23T00:56:17.35122Z","iopub.execute_input":"2022-08-23T00:56:17.351752Z","iopub.status.idle":"2022-08-23T00:56:25.911641Z","shell.execute_reply.started":"2022-08-23T00:56:17.351728Z","shell.execute_reply":"2022-08-23T00:56:25.910678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_test = SampleGenerator(test, 4, shuffle = False, is_train = False)","metadata":{"execution":{"iopub.status.busy":"2022-08-23T00:56:25.913058Z","iopub.execute_input":"2022-08-23T00:56:25.9133Z","iopub.status.idle":"2022-08-23T00:56:25.964043Z","shell.execute_reply.started":"2022-08-23T00:56:25.913277Z","shell.execute_reply":"2022-08-23T00:56:25.96281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = model.predict(data_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-23T00:56:25.965575Z","iopub.execute_input":"2022-08-23T00:56:25.96622Z","iopub.status.idle":"2022-08-23T00:56:38.683349Z","shell.execute_reply.started":"2022-08-23T00:56:25.966191Z","shell.execute_reply":"2022-08-23T00:56:38.682559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def proccess_test(df, preds):\n    cols = ['C1', 'C2', 'C3', 'C4', 'C5', 'C6', 'C7']\n    cols = ['patient_overall', 'C1', 'C2', 'C3', 'C4', 'C5', 'C6', 'C7']\n    patients = df.StudyInstanceUID.to_list()\n    \n    df_sub = pd.DataFrame()\n    \n    for i, p in enumerate(patients):\n        scores = list(preds[i])\n        if len(scores) < 8:\n            scores.append(preds[i].max() + preds[i].mean())\n        \n        df_temp = pd.DataFrame({'StudyInstanceUID': [p]*len(cols), 'prediction_type': cols, 'fractured': scores})\n        df_sub = pd.concat([df_sub, df_temp])\n        \n        del df_temp\n    \n    df_sub['row_id'] = df_sub['StudyInstanceUID'] + '_' + df_sub['prediction_type']\n    \n    return df_sub[['row_id', 'fractured']].reset_index(drop = True)\n    ","metadata":{"execution":{"iopub.status.busy":"2022-08-23T00:56:38.685353Z","iopub.execute_input":"2022-08-23T00:56:38.685848Z","iopub.status.idle":"2022-08-23T00:56:38.739Z","shell.execute_reply.started":"2022-08-23T00:56:38.685821Z","shell.execute_reply":"2022-08-23T00:56:38.738146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sub = proccess_test(test, pred)","metadata":{"execution":{"iopub.status.busy":"2022-08-23T00:56:38.740122Z","iopub.execute_input":"2022-08-23T00:56:38.740356Z","iopub.status.idle":"2022-08-23T00:56:38.79878Z","shell.execute_reply.started":"2022-08-23T00:56:38.740333Z","shell.execute_reply":"2022-08-23T00:56:38.79803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sub.to_csv('submission.csv', index=False)\ndf_sub","metadata":{"execution":{"iopub.status.busy":"2022-08-23T00:56:38.80005Z","iopub.execute_input":"2022-08-23T00:56:38.800973Z","iopub.status.idle":"2022-08-23T00:56:38.857472Z","shell.execute_reply.started":"2022-08-23T00:56:38.800942Z","shell.execute_reply":"2022-08-23T00:56:38.856859Z"},"trusted":true},"execution_count":null,"outputs":[]}]}