{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install pydicom\n!pip install pylibjpeg\n!pip install pylibjpeg-libjpeg\n!pip install pylibjpeg pylibjpeg-libjpeg pylibjpeg-openjpeg","metadata":{"execution":{"iopub.status.busy":"2023-03-24T07:39:33.662134Z","iopub.execute_input":"2023-03-24T07:39:33.662552Z","iopub.status.idle":"2023-03-24T07:40:22.513038Z","shell.execute_reply.started":"2023-03-24T07:39:33.662517Z","shell.execute_reply":"2023-03-24T07:40:22.510836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport sys\nimport numpy as np\nimport pandas as pd\nfrom glob import glob\nfrom PIL import Image\nimport cv2\nimport re\nimport gc\nfrom tqdm import tqdm\nimport math\n\nimport matplotlib.pyplot as plt\nfrom matplotlib.patches import Rectangle\n\nimport skimage.transform as skTrans\nfrom skimage import exposure\n\nimport albumentations as alb\nfrom albumentations.pytorch import ToTensorV2\n\nimport pydicom as dicom\n\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nimport torch.nn.functional as F\nfrom torch.utils.data import DataLoader, Dataset\nfrom torch.cuda.amp import autocast, GradScaler\nfrom torch.optim.lr_scheduler import OneCycleLR\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.model_selection import KFold, StratifiedKFold\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2023-03-24T07:40:22.516009Z","iopub.execute_input":"2023-03-24T07:40:22.516485Z","iopub.status.idle":"2023-03-24T07:40:22.53023Z","shell.execute_reply.started":"2023-03-24T07:40:22.516437Z","shell.execute_reply":"2023-03-24T07:40:22.527996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Configs","metadata":{}},{"cell_type":"code","source":"SEED = 1927550\nIMG_SIZE = 512\nBATCH = 5\nEPOCH = 7\nCLASS = 13 # from 0 to 12\nfolds = 5\nhidden1 = 128\nhidden2 = 64\ntot_slice = 29\nmid_slice = 13\nbest_acc = 0\ndevice = 'cuda' if torch.cuda.is_available() else 'cpu'\nbase_path = '/kaggle/input'\n\ntrainlosslog = []\ntrainacclog = []\nvalidlosslog = []\nvalidacclog = []","metadata":{"execution":{"iopub.status.busy":"2023-03-24T07:40:22.532448Z","iopub.execute_input":"2023-03-24T07:40:22.533943Z","iopub.status.idle":"2023-03-24T07:40:22.551634Z","shell.execute_reply.started":"2023-03-24T07:40:22.53389Z","shell.execute_reply":"2023-03-24T07:40:22.550546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vert_class = {\n    '0': 0,\n    '1': 1,\n    '2': 2,\n    '3': 3,\n    '4': 4,\n    '5': 5,\n    '6': 6,\n    '7': 7,\n    '8': 8,\n    '9': 9,\n    '10': 10,\n    '11': 11,\n    '12': 12\n}","metadata":{"execution":{"iopub.status.busy":"2023-03-24T07:40:22.554864Z","iopub.execute_input":"2023-03-24T07:40:22.555265Z","iopub.status.idle":"2023-03-24T07:40:22.567977Z","shell.execute_reply.started":"2023-03-24T07:40:22.555226Z","shell.execute_reply":"2023-03-24T07:40:22.566856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Preprocessing","metadata":{}},{"cell_type":"code","source":"vert_df = pd.read_csv(f'{base_path}/sagittal-preprocess/vert_list.csv')\nvert_df['StudyInstanceUID'] = 0\n\nfor idx in range(len(vert_df)):\n    vert_id = vert_df.loc[idx]['id']\n    studyuid = vert_id.split('_')[0]\n    vert_df['StudyInstanceUID'][idx] = studyuid\n\nfor idx in range(len(vert_df)):\n    loc = vert_df.loc[idx]\n    vert_df['vertebrae'][idx] = vert_df['vertebrae'][idx][1:len(loc['vertebrae'])-1].split(' ')","metadata":{"execution":{"iopub.status.busy":"2023-03-24T07:40:22.569518Z","iopub.execute_input":"2023-03-24T07:40:22.569981Z","iopub.status.idle":"2023-03-24T07:40:43.396194Z","shell.execute_reply.started":"2023-03-24T07:40:22.56994Z","shell.execute_reply":"2023-03-24T07:40:43.394189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vert_df.head(199)","metadata":{"execution":{"iopub.status.busy":"2023-03-24T07:40:43.397966Z","iopub.execute_input":"2023-03-24T07:40:43.398388Z","iopub.status.idle":"2023-03-24T07:40:43.419447Z","shell.execute_reply.started":"2023-03-24T07:40:43.39835Z","shell.execute_reply":"2023-03-24T07:40:43.418164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"studyuidlist = list(set(list(vert_df['StudyInstanceUID'])))\nnew_vert = []\nnew_id = []\n\nfor studyuid in studyuidlist:\n    each_df = vert_df[vert_df['StudyInstanceUID'] == studyuid].reset_index(drop=True)\n    \n    for idx in range(len(each_df)):\n        loc = each_df.loc[len(each_df)-1-idx]\n        new_id.append(loc['id'])\n        if loc['vertebrae'] == ['0.', '1.', '2.']:\n            count = 0j\n            for before_list in each_df['vertebrae'][len(each_df)-1-idx:]:\n                if before_list == ['0.', '2.']: count = count + 1\n            \n            if count == 0:\n                new_vert.append('1')\n            else:\n                new_vert.append('2')\n        else:\n            leave = loc['vertebrae'][-1].split('.')[0]\n            new_vert.append(f'{leave}')","metadata":{"execution":{"iopub.status.busy":"2023-03-24T07:40:43.421375Z","iopub.execute_input":"2023-03-24T07:40:43.422687Z","iopub.status.idle":"2023-03-24T07:40:46.885608Z","shell.execute_reply.started":"2023-03-24T07:40:43.42264Z","shell.execute_reply":"2023-03-24T07:40:46.884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_df = pd.DataFrame(list(zip(new_id, new_vert)), columns=['id', 'vertebrae'])\nvert_df = vert_df.merge(new_df, on='id', how='left')\nvert_df['vertebrae'] = vert_df['vertebrae_y']\nvert_df.drop(['vertebrae_x', 'vertebrae_y'], axis=1, inplace=True)\nvert_df","metadata":{"execution":{"iopub.status.busy":"2023-03-24T07:40:46.88792Z","iopub.execute_input":"2023-03-24T07:40:46.888296Z","iopub.status.idle":"2023-03-24T07:40:46.952319Z","shell.execute_reply.started":"2023-03-24T07:40:46.888262Z","shell.execute_reply":"2023-03-24T07:40:46.950623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# cropping voxel can only be done when the slice of image index is known.\n# which means, with my stage 1 prediction, it is unable to get cropped voxel preprocessed images\n# thus, to solve this,\n# no cropping voxel, but predicting on the labels is done\n\ntrain_df = pd.read_csv(f'{base_path}/rsna-2022-cervical-spine-fracture-detection/train.csv')\ndf = vert_df.merge(train_df, on='StudyInstanceUID', how='left')","metadata":{"execution":{"iopub.status.busy":"2023-03-24T07:40:46.954839Z","iopub.execute_input":"2023-03-24T07:40:46.955198Z","iopub.status.idle":"2023-03-24T07:40:46.97977Z","shell.execute_reply.started":"2023-03-24T07:40:46.955165Z","shell.execute_reply":"2023-03-24T07:40:46.977652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2023-03-24T07:40:46.983856Z","iopub.execute_input":"2023-03-24T07:40:46.984233Z","iopub.status.idle":"2023-03-24T07:40:47.007598Z","shell.execute_reply.started":"2023-03-24T07:40:46.984201Z","shell.execute_reply":"2023-03-24T07:40:47.005595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.unique(df['vertebrae'])","metadata":{"execution":{"iopub.status.busy":"2023-03-24T07:40:47.010068Z","iopub.execute_input":"2023-03-24T07:40:47.01054Z","iopub.status.idle":"2023-03-24T07:40:47.037165Z","shell.execute_reply.started":"2023-03-24T07:40:47.0105Z","shell.execute_reply":"2023-03-24T07:40:47.035471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Drop Bad Scans","metadata":{}},{"cell_type":"code","source":"bad_scans = ['1.2.826.0.1.3680043.20574','1.2.826.0.1.3680043.29952']\n\nfor uid in bad_scans:\n    df.drop(df[df['StudyInstanceUID']==uid].index, axis=0, inplace=True)\ndf.reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2023-03-24T07:40:47.038637Z","iopub.execute_input":"2023-03-24T07:40:47.039176Z","iopub.status.idle":"2023-03-24T07:40:47.085576Z","shell.execute_reply.started":"2023-03-24T07:40:47.039124Z","shell.execute_reply":"2023-03-24T07:40:47.084426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Leave vertebraes only between 1 and 7","metadata":{}},{"cell_type":"code","source":"verts = np.unique(df['vertebrae'])\nclass_verts = [1, 2, 3, 4, 5, 6, 7]\nremove_verts = []\n\nfor vert in verts:\n    if vert_class[vert] not in class_verts:\n        remove_verts.append(vert)\n    \nremove_verts = np.unique(remove_verts)\nfor vert in remove_verts:\n    df.drop(df[df['vertebrae']==vert].index, axis=0, inplace=True)\ndf.reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2023-03-24T07:40:47.088423Z","iopub.execute_input":"2023-03-24T07:40:47.088902Z","iopub.status.idle":"2023-03-24T07:40:47.172669Z","shell.execute_reply.started":"2023-03-24T07:40:47.088866Z","shell.execute_reply":"2023-03-24T07:40:47.170579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Save cropped voxel images","metadata":{}},{"cell_type":"code","source":"work_path = '/kaggle/input/rsna-2022-cervical-spine-fracture-detection'","metadata":{"execution":{"iopub.status.busy":"2023-03-24T07:40:47.175875Z","iopub.execute_input":"2023-03-24T07:40:47.176571Z","iopub.status.idle":"2023-03-24T07:40:47.184505Z","shell.execute_reply.started":"2023-03-24T07:40:47.176505Z","shell.execute_reply":"2023-03-24T07:40:47.18211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def slice_len(ID):\n    train_path = f'{work_path}/train_images'\n    path = os.path.join(train_path, ID)\n    return len([file for file in os.listdir(path)])","metadata":{"execution":{"iopub.status.busy":"2023-03-24T07:40:47.186093Z","iopub.execute_input":"2023-03-24T07:40:47.186537Z","iopub.status.idle":"2023-03-24T07:40:47.197912Z","shell.execute_reply.started":"2023-03-24T07:40:47.186501Z","shell.execute_reply":"2023-03-24T07:40:47.195963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_dicom(uid, idx, size=IMG_SIZE):\n    filename = f'{work_path}/train_images/{uid}/{idx}.dcm'\n    \n    img = dicom.read_file(filename)\n    img = img.pixel_array\n    img = cv2.resize(img, (size, size), interpolation=cv2.INTER_LINEAR)\n    img = img - np.min(img)\n    img = img / (np.max(img) + 1e-7)\n    img = (img * 255).astype(np.uint8)\n    \n    return img","metadata":{"execution":{"iopub.status.busy":"2023-03-24T07:40:47.199259Z","iopub.execute_input":"2023-03-24T07:40:47.199653Z","iopub.status.idle":"2023-03-24T07:40:47.211662Z","shell.execute_reply.started":"2023-03-24T07:40:47.199613Z","shell.execute_reply":"2023-03-24T07:40:47.210184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_voxel_df = []\ncount = 0\ntrain_df_uid_list = list(set(list(df['StudyInstanceUID'])))\n\ntry: os.mkdir('voxel_image')\nexcept: pass\n\nfor uid in tqdm(train_df_uid_list):\n    this_df = df[df['StudyInstanceUID']==uid].reset_index(drop=True)\n    \n    for each_vert in class_verts:\n        each_vert_df = this_df[this_df['vertebrae']==f'{each_vert}'].reset_index(drop=True)\n        if len(each_vert_df) < 1:\n            break\n        slices = np.linspace(0, len(each_vert_df)-1, tot_slice, dtype=int)\n\n        cropped_voxel = []\n        for each_slice in slices:\n            index = each_vert_df.loc[each_slice]['id'].split('_')[1]\n            image = load_dicom(uid, index)\n            cropped_voxel.append(image)\n        mask_id = each_vert_df.loc[slices[mid_slice]]['id']\n        mask =  np.load(f'{base_path}/axial-preprocessed-windowing/segmentations/{mask_id}.npz')['arr_0'][:,:,-1]\n        cropped_voxel.append(mask)\n\n        save_path = f'voxel_image/{uid}_C{each_vert}.npz'\n        with open(f'/kaggle/working/{save_path}', 'wb') as file:\n            np.savez_compressed(file, cropped_voxel)\n\n        del(cropped_voxel)\n        new_voxel_df.append([f'{uid}_C{each_vert}', uid, f'C{each_vert}', save_path])","metadata":{"execution":{"iopub.status.busy":"2023-03-24T07:42:46.91179Z","iopub.execute_input":"2023-03-24T07:42:46.912405Z","iopub.status.idle":"2023-03-24T08:05:37.099226Z","shell.execute_reply.started":"2023-03-24T07:42:46.912335Z","shell.execute_reply":"2023-03-24T08:05:37.096805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import shutil\n\noutput_filename = 'voxel_image'\ndir_name = 'voxel_image'\nshutil.make_archive(output_filename, 'zip', dir_name)\n\npath = '/kaggle/working/voxel_image'\nfor file_name in os.listdir(path):\n    file = path + '/' + file_name\n    if os.path.isfile(file):\n        os.remove(file)\n\nos.rmdir('voxel_image')","metadata":{"execution":{"iopub.status.busy":"2023-03-24T08:10:45.678869Z","iopub.execute_input":"2023-03-24T08:10:45.679493Z","iopub.status.idle":"2023-03-24T08:12:55.517804Z","shell.execute_reply.started":"2023-03-24T08:10:45.679438Z","shell.execute_reply":"2023-03-24T08:12:55.515943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"save_voxel_df = pd.DataFrame(new_voxel_df, columns=['id', 'StudyInstanceUID', 'vertebrae', 'iamge_path'])\nsave_voxel_df.to_csv('voxel_crop_df.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-03-24T08:12:55.520596Z","iopub.execute_input":"2023-03-24T08:12:55.521556Z","iopub.status.idle":"2023-03-24T08:12:55.542344Z","shell.execute_reply.started":"2023-03-24T08:12:55.52151Z","shell.execute_reply":"2023-03-24T08:12:55.540912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}