{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from IPython.display import clear_output, FileLink\nimport torch.nn as nn\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nimport numpy as np\nimport torch.nn.functional as F\nfrom PIL import Image\nfrom tqdm import tqdm\n\nimport pydicom as dicom\nimport pandas as pd\nclear_output()\nNUM_CLASS = 2\nfrom pathlib import Path\nfrom joblib import Parallel, delayed\n\nimport os, shutil\nfrom glob import glob\n\nimport pydicom\nimport math\nimport cv2\nimport matplotlib.pyplot as plt\n\nimport torch\nimport json","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-09-10T14:36:03.87166Z","iopub.execute_input":"2023-09-10T14:36:03.872292Z","iopub.status.idle":"2023-09-10T14:36:07.625825Z","shell.execute_reply.started":"2023-09-10T14:36:03.872258Z","shell.execute_reply":"2023-09-10T14:36:07.624527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROOT_PATH = '/kaggle/input/rsna-2023-abdominal-trauma-detection'\nIMG_DIR = '/tmp/Dataset/rsna-atd'\n","metadata":{"execution":{"iopub.status.busy":"2023-09-10T14:36:07.627503Z","iopub.execute_input":"2023-09-10T14:36:07.628299Z","iopub.status.idle":"2023-09-10T14:36:07.63258Z","shell.execute_reply.started":"2023-09-10T14:36:07.628262Z","shell.execute_reply":"2023-09-10T14:36:07.631451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"resize_dim = 512\nimg_size = [resize_dim, resize_dim]\nIDX = 0\nPARTS = 7\nprint(f\"Image Size: {img_size}\")\nprint(f\"Dim: {np.prod(img_size)**0.5: 0.2f}\")","metadata":{"execution":{"iopub.status.busy":"2023-09-10T14:36:07.633951Z","iopub.execute_input":"2023-09-10T14:36:07.634555Z","iopub.status.idle":"2023-09-10T14:36:07.651368Z","shell.execute_reply.started":"2023-09-10T14:36:07.634524Z","shell.execute_reply":"2023-09-10T14:36:07.650043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Meta Data","metadata":{}},{"cell_type":"code","source":"\nimport os\nimport glob\nfrom tqdm import tqdm\n\n# column_list = [\"patient_id\",\"bowel_healthy\",\"bowel_injury\",\"extravasation_healthy\",\"extravasation_injury\",\"kidney_healthy\",\"kidney_low\",\"kidney_high\",\"liver_healthy\",\"liver_low\",\"liver_high\",\"spleen_healthy\",\"spleen_low\",\"spleen_high\",\"any_injury\",\"image_path\",\"series_id\"]\ncolumn_list = [\"patient_id\",\"series_id\", \"image_path\"]\ndata_df = pd.read_csv(f'{ROOT_PATH}/train.csv')\ntrain = []\nfor patient_id in tqdm(data_df['patient_id']):\n    series = os.listdir(f'{ROOT_PATH}/train_images/' + str(patient_id))\n    for serie in series:\n        path = f'{ROOT_PATH}/train_images/' + str(patient_id) + '/' + serie\n        for file in glob.glob(path + '/*'):\n            train.append([patient_id] + [serie] + [file])\ntrain_df = pd.DataFrame.from_dict(train)\ntrain_df = train_df.rename(columns={0: 'patient_id', 1: 'series_id', 2: 'image_path'})\ntrain_df = data_df.merge(train_df, on=['patient_id'], how='left')\ntrain_df = train_df.sort_values('image_path')\n\n#'''","metadata":{"execution":{"iopub.status.busy":"2023-09-10T14:36:07.654759Z","iopub.execute_input":"2023-09-10T14:36:07.655237Z","iopub.status.idle":"2023-09-10T14:40:55.219592Z","shell.execute_reply.started":"2023-09-10T14:36:07.655204Z","shell.execute_reply":"2023-09-10T14:40:55.21835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(train_df))","metadata":{"execution":{"iopub.status.busy":"2023-09-10T14:40:55.221165Z","iopub.execute_input":"2023-09-10T14:40:55.221549Z","iopub.status.idle":"2023-09-10T14:40:55.229728Z","shell.execute_reply.started":"2023-09-10T14:40:55.221516Z","shell.execute_reply":"2023-09-10T14:40:55.228355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\nimport glob\n\ntest_paths = glob.glob('/kaggle/input/rsna-2023-abdominal-trauma-detection/test_images/*/*/*dcm')\n\ntest_df = pd.DataFrame(test_paths, columns=[\"image_path\"])\ntest_df['patient_id'] = test_df.image_path.map(lambda x: x.split('/')[-3]).astype(int)\ntest_df['series_id'] = test_df.image_path.map(lambda x: x.split('/')[-2]).astype(int)\ntest_df['instance_number'] = test_df.image_path.map(lambda x: x.split('/')[-1].replace('.dcm','')).astype(int)\nprint('Test:')\nprint(f'# Size: {test_df.shape}')\ndisplay(test_df.head())\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2023-09-10T14:40:55.231153Z","iopub.execute_input":"2023-09-10T14:40:55.231661Z","iopub.status.idle":"2023-09-10T14:40:55.246527Z","shell.execute_reply.started":"2023-09-10T14:40:55.231621Z","shell.execute_reply":"2023-09-10T14:40:55.245062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm -r {IMG_DIR}\nos.makedirs(f'{IMG_DIR}/train_images', exist_ok = True)\n#os.makedirs(f'{IMG_DIR}/test_images', exist_ok = True)\n","metadata":{"execution":{"iopub.status.busy":"2023-09-10T14:40:55.249313Z","iopub.execute_input":"2023-09-10T14:40:55.249741Z","iopub.status.idle":"2023-09-10T14:40:56.295149Z","shell.execute_reply.started":"2023-09-10T14:40:55.249708Z","shell.execute_reply":"2023-09-10T14:40:56.293589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef standardize_pixel_array(dcm, scale='window'):\n    # Correct DICOM pixel_array if PixelRepresentation == 1.\n    pixel_array = dcm.pixel_array\n    if dcm.PixelRepresentation == 1:\n        bit_shift = dcm.BitsAllocated - dcm.BitsStored\n        dtype = pixel_array.dtype \n        new_array = (pixel_array << bit_shift).astype(dtype) >>  bit_shift\n        pixel_array = pydicom.pixel_data_handlers.util.apply_modality_lut(new_array, dcm)\n    \n    pixel_array = hounsfield_units(pixel_array, dcm.RescaleIntercept , dcm.RescaleSlope)\n    \n    if scale is True: \n        pixel_array = window_scale(pixel_array)\n    elif scale in ['window']: \n        pixel_array = window(pixel_array, dcm.WindowCenter, dcm.WindowWidth)\n\n    return pixel_array\n\ndef hounsfield_units(pixel_array, intercept, slope):\n    pixel_array = pixel_array * slope + intercept\n    return pixel_array\n\ndef window_scale(px, min_px=-1100, max_px=None, n_bins=100):\n    if min_px is not None: px[px<min_px] = min_px\n    if max_px is not None: px[px>max_px] = max_px\n        \n    \"A function to split the range of pixel values into groups, such that each group has around the same number of pixels\"\n    imsd = np.sort(px.flatten())\n    t = np.concatenate((np.array([0.001]), np.arange(n_bins) / n_bins + (1 / (2 * n_bins)), np.array([0.999])))\n    t = (len(imsd) * t).astype(int)\n    brks = np.unique(imsd[t])\n    \n    \"Scales a tensor using `freqhist_bins` to values between 0 and 1\"\n    ys = np.linspace(0., 1., len(brks))\n    x = px.flatten()\n    x = np.interp(x, np.array(brks), ys)\n    \n    return x.reshape(px.shape).clip(0., 1.)\n\ndef window(px, WC, WW):\n    upper, lower = WC+WW//2, WC-WW//2\n    px = np.clip(px.copy(), lower, upper)\n    #px[px<lower] = lower\n    #px[px>upper] = upper\n    return px\n\ndef read_xray(path, fix_monochrome = True):\n    dicom = pydicom.dcmread(path)\n    data = standardize_pixel_array(dicom)\n    data = (data - np.min(data)) / (np.max(data) - np.min(data) + 1e-6)\n    \n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = 1.0 - data\n    data = (data * 255.0).astype(np.uint8)    \n    \n    return data\n\ndef resize_and_save(file_path):\n    img = read_xray(file_path)\n    h, w = img.shape[:2]  # orig hw\n    img = cv2.resize(img, (resize_dim, resize_dim), cv2.INTER_LINEAR)\n    #img = (img * 255).astype(np.uint8)\n    \n    sub_path = file_path.split(\"/\",4)[-1].split('.dcm')[0] + '.png'\n    infos = sub_path.split('/')\n    pid = infos[-3]\n    sid = infos[-2]\n    iid = infos[-1]; iid = iid.replace('.png','')\n    new_path = os.path.join(IMG_DIR, sub_path)\n    os.makedirs(new_path.rsplit('/',1)[0], exist_ok=True)\n    cv2.imwrite(new_path, img,\n#                 [cv2.IMWRITE_PNG_COMPRESSION, 1],\n               )\n    return pid,sid,iid,w,h, img\n#'''","metadata":{"execution":{"iopub.status.busy":"2023-09-10T14:40:56.297256Z","iopub.execute_input":"2023-09-10T14:40:56.298382Z","iopub.status.idle":"2023-09-10T14:40:56.318604Z","shell.execute_reply.started":"2023-09-10T14:40:56.298307Z","shell.execute_reply":"2023-09-10T14:40:56.317716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfrom joblib import Parallel, delayed\nSIZE = -(-len(train_df) // PARTS)\nfile_paths = train_df.image_path.tolist()[IDX*SIZE:(IDX+1)*SIZE]","metadata":{"execution":{"iopub.status.busy":"2023-09-10T14:40:56.320356Z","iopub.execute_input":"2023-09-10T14:40:56.321142Z","iopub.status.idle":"2023-09-10T14:40:56.487103Z","shell.execute_reply.started":"2023-09-10T14:40:56.321102Z","shell.execute_reply":"2023-09-10T14:40:56.486001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img = read_xray(file_paths[0])\n#img = read_xray('/kaggle/input/rsna-2023-abdominal-trauma-detection/test_images/63706/39279/30.dcm')\n#pid,sid,iid,w,h, img = resize_and_save(file_paths[0])\n\nplt.imshow(img, cmap=plt.cm.bone)","metadata":{"execution":{"iopub.status.busy":"2023-09-10T14:40:56.490612Z","iopub.execute_input":"2023-09-10T14:40:56.491232Z","iopub.status.idle":"2023-09-10T14:40:56.900483Z","shell.execute_reply.started":"2023-09-10T14:40:56.491194Z","shell.execute_reply":"2023-09-10T14:40:56.899442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SIZE = -(-len(train_df) // PARTS)\nfile_paths = train_df.image_path.tolist()[IDX*SIZE:(IDX+1)*SIZE]\n\nfor file_path in tqdm(file_paths, leave=True, position=0):\n    resize_and_save(file_path)","metadata":{"execution":{"iopub.status.busy":"2023-09-10T14:40:56.90163Z","iopub.execute_input":"2023-09-10T14:40:56.902136Z","iopub.status.idle":"2023-09-10T16:22:37.73853Z","shell.execute_reply.started":"2023-09-10T14:40:56.902106Z","shell.execute_reply":"2023-09-10T16:22:37.735172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.to_csv(f'{IMG_DIR}/train.csv',index = False)\n#test_df.to_csv(f'{IMG_DIR}/test.csv',index = False)\n\nshutil.copy(f'{ROOT_PATH}/train_series_meta.csv',f'{IMG_DIR}/')\n#shutil.copy(f'{ROOT_PATH}/test_series_meta.csv',f'{IMG_DIR}/')\n#shutil.copy(f'{ROOT_PATH}/sample_submission.csv',f'{IMG_DIR}/')","metadata":{"execution":{"iopub.status.busy":"2023-09-10T16:22:37.749253Z","iopub.execute_input":"2023-09-10T16:22:37.749884Z","iopub.status.idle":"2023-09-10T16:22:47.57358Z","shell.execute_reply.started":"2023-09-10T16:22:37.749817Z","shell.execute_reply":"2023-09-10T16:22:47.572534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from zipfile import ZipFile\n\nzipObj = ZipFile(f'/tmp/Dataset/rsna-2023-atd.zip', 'w')\n\nfile_paths = glob.glob(f'{IMG_DIR}/**/*',recursive = True)\nprint(f'Total Files:{len(file_paths)}')\nprint('Zippping...')\nfor file_path in tqdm(file_paths,leave=True, position=0):\n    zipObj.write(file_path, file_path[len(IMG_DIR):])\n    os.remove(file_path) if os.path.isfile(file_path) else None\nzipObj.close()","metadata":{"execution":{"iopub.status.busy":"2023-09-10T16:22:47.575232Z","iopub.execute_input":"2023-09-10T16:22:47.575679Z","iopub.status.idle":"2023-09-10T16:29:00.162992Z","shell.execute_reply.started":"2023-09-10T16:22:47.575645Z","shell.execute_reply":"2023-09-10T16:29:00.161442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -qq kaggle --upgrade\n!mkdir ~/.kaggle/\n!cp -b /kaggle/input/longpml/kaggle.json  ~/.kaggle/\n!chmod 600 ~/.kaggle/kaggle.json","metadata":{"execution":{"iopub.status.busy":"2023-09-10T16:29:00.165776Z","iopub.execute_input":"2023-09-10T16:29:00.166151Z","iopub.status.idle":"2023-09-10T16:29:17.974194Z","shell.execute_reply.started":"2023-09-10T16:29:00.166121Z","shell.execute_reply":"2023-09-10T16:29:17.972918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!kaggle --version","metadata":{"execution":{"iopub.status.busy":"2023-09-10T16:29:17.976671Z","iopub.execute_input":"2023-09-10T16:29:17.97717Z","iopub.status.idle":"2023-09-10T16:29:19.558231Z","shell.execute_reply.started":"2023-09-10T16:29:17.977122Z","shell.execute_reply":"2023-09-10T16:29:19.557104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from kaggle_secrets import UserSecretsClient\nuser_secrets = UserSecretsClient()\nsecret_value_0 = user_secrets.get_secret(\"KAGGLE_KEY\")\nsecret_value_1 = user_secrets.get_secret(\"KAGGLE_USERNAME\")","metadata":{"execution":{"iopub.status.busy":"2023-09-10T16:29:19.560294Z","iopub.execute_input":"2023-09-10T16:29:19.560685Z","iopub.status.idle":"2023-09-10T16:29:19.856484Z","shell.execute_reply.started":"2023-09-10T16:29:19.560647Z","shell.execute_reply":"2023-09-10T16:29:19.855354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os, json\nos.makedirs('/kaggle/dataset/', exist_ok=True)\n!cp -b '/tmp/Dataset/rsna-2023-atd.zip' '/kaggle/dataset/'\nos.listdir('/kaggle/dataset/')","metadata":{"execution":{"iopub.status.busy":"2023-09-10T16:29:19.85791Z","iopub.execute_input":"2023-09-10T16:29:19.858231Z","iopub.status.idle":"2023-09-10T16:31:15.568194Z","shell.execute_reply.started":"2023-09-10T16:29:19.858204Z","shell.execute_reply":"2023-09-10T16:31:15.566488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta = dict(\n    id=\"hundred3421/rsna-2023-atd-window\",\n    title=\"RSNA_2023_ATD_Window\",\n    isPrivate=True,\n    licenses=[dict(name=\"other\")]\n)\n\nwith open('/kaggle/dataset/dataset-metadata.json', 'w') as f:\n    json.dump(meta, f)","metadata":{"execution":{"iopub.status.busy":"2023-09-10T16:31:15.571635Z","iopub.execute_input":"2023-09-10T16:31:15.572663Z","iopub.status.idle":"2023-09-10T16:31:15.579798Z","shell.execute_reply.started":"2023-09-10T16:31:15.572614Z","shell.execute_reply":"2023-09-10T16:31:15.578906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!kaggle datasets version -p \"/kaggle/dataset\" -m \"1/7 dataset version 1\" --dir-mode skip","metadata":{"execution":{"iopub.status.busy":"2023-09-10T16:31:15.581415Z","iopub.execute_input":"2023-09-10T16:31:15.582136Z","iopub.status.idle":"2023-09-10T16:32:57.892737Z","shell.execute_reply.started":"2023-09-10T16:31:15.582095Z","shell.execute_reply":"2023-09-10T16:32:57.891169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}],"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}}