{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":52254,"databundleVersionId":6863140,"sourceType":"competition"}],"dockerImageVersionId":30664,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"raw","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"pip install -qU python-gdcm pydicom pylibjpeg","metadata":{"execution":{"iopub.status.busy":"2024-03-09T04:57:23.021342Z","iopub.execute_input":"2024-03-09T04:57:23.021754Z","iopub.status.idle":"2024-03-09T04:57:38.578048Z","shell.execute_reply.started":"2024-03-09T04:57:23.021723Z","shell.execute_reply":"2024-03-09T04:57:38.576526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport cv2\nfrom glob import glob\nimport gdcm\nimport pydicom\nimport zipfile\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nfrom tqdm import tqdm\nfrom joblib import Parallel, delayed\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut","metadata":{"execution":{"iopub.status.busy":"2024-03-09T04:57:49.626787Z","iopub.execute_input":"2024-03-09T04:57:49.627236Z","iopub.status.idle":"2024-03-09T04:57:51.531574Z","shell.execute_reply.started":"2024-03-09T04:57:49.627201Z","shell.execute_reply":"2024-03-09T04:57:51.530348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## From https://www.kaggle.com/code/haqishen/rsna-2022-1st-place-solution-inference/input\n# data_dir = '../input/rsna-2022-cervical-spine-fracture-detection/'\nimage_size_seg = (128, 128, 128)\n# msk_size = image_size_seg[0]\n# image_size_cls = 224\n# n_slice_per_c = 15\n# n_ch = 5\n\n# batch_size_seg = 1\n# num_workers = 2","metadata":{"execution":{"iopub.status.busy":"2024-03-09T04:58:12.171841Z","iopub.execute_input":"2024-03-09T04:58:12.172552Z","iopub.status.idle":"2024-03-09T04:58:12.178999Z","shell.execute_reply.started":"2024-03-09T04:58:12.172511Z","shell.execute_reply":"2024-03-09T04:58:12.177371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def standardize_pixel_array(dcm: pydicom.dataset.FileDataset) -> np.ndarray:\n    \"\"\"\n    Source : https://www.kaggle.com/competitions/rsna-2023-abdominal-trauma-detection/discussion/427217\n    \"\"\"\n    # Correct DICOM pixel_array if PixelRepresentation == 1.\n    pixel_array = dcm.pixel_array\n    if dcm.PixelRepresentation == 1:\n        bit_shift = dcm.BitsAllocated - dcm.BitsStored\n        dtype = pixel_array.dtype \n        pixel_array = (pixel_array << bit_shift).astype(dtype) >>  bit_shift\n#         pixel_array = pydicom.pixel_data_handlers.util.apply_modality_lut(new_array, dcm)\n\n    intercept = float(dcm.RescaleIntercept)\n    slope = float(dcm.RescaleSlope)\n    center = int(dcm.WindowCenter)\n    width = int(dcm.WindowWidth)\n    low = center - width / 2\n    high = center + width / 2    \n    \n    pixel_array = (pixel_array * slope) + intercept\n    pixel_array = np.clip(pixel_array, low, high)\n\n    return pixel_array","metadata":{"execution":{"iopub.status.busy":"2024-03-09T04:58:52.581257Z","iopub.execute_input":"2024-03-09T04:58:52.581695Z","iopub.status.idle":"2024-03-09T04:58:52.592615Z","shell.execute_reply.started":"2024-03-09T04:58:52.581655Z","shell.execute_reply":"2024-03-09T04:58:52.591204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN_PATH = \"/kaggle/input/rsna-2023-abdominal-trauma-detection/train_images/\"\n\nprint('Number of training patients :', len(os.listdir(TRAIN_PATH)))","metadata":{"execution":{"iopub.status.busy":"2024-03-09T05:01:52.436794Z","iopub.execute_input":"2024-03-09T05:01:52.437283Z","iopub.status.idle":"2024-03-09T05:01:52.7796Z","shell.execute_reply.started":"2024-03-09T05:01:52.437248Z","shell.execute_reply":"2024-03-09T05:01:52.778417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from collections import Counter\ndef study_path_to_3D_image(path, image_size_seg=(128, 128, 128), plot_image=False, z_df=None):\n    t_paths = sorted(glob(os.path.join(path, \"*\")), key=lambda x: int(x.split('/')[-1].split(\".\")[0]))\n    n_scans = len(t_paths)\n#     print(n_scans)\n    indices = np.quantile(list(range(n_scans)), np.linspace(0., 1., image_size_seg[2])).round().astype(int)\n    print(len(indices))\n    t_paths = [t_paths[i] for i in indices]\n\n    imgs = {}\n    pos_zs = []\n    for filename in t_paths:\n        dicom = pydicom.dcmread(filename)\n        pos_z = dicom[(0x20, 0x32)].value[-1]  # to retrieve the order of frames\n#         dicom_numbers.append(filename.split('/')[-1].split(\".\")[0])\n        pos_zs.append(pos_z)\n        img = standardize_pixel_array(dicom)\n        img = cv2.resize(img, (image_size_seg[0], image_size_seg[1]), interpolation = cv2.INTER_AREA)\n        if dicom.PhotometricInterpretation == \"MONOCHROME1\":\n            img = 1 - img\n        imgs[pos_z] = img\n        \n#     print(dicom_numbers, pos_zs)\n    if z_df is not None: \n        study = path.split('/')[-1]\n        z_df.loc[study, 'pos_z_ascending'] = pos_zs[-1] > pos_zs[0]\n        \n    images = []\n    cnt = Counter(pos_zs)\n    for i, k in enumerate(sorted(imgs.keys())):\n        img = imgs[k]\n#         images.append(img)\n        images.extend([img] * cnt[k]) # to make sure we have same dimensions\n        if not (i % 100) and plot_image:\n            plt.figure(figsize=(5, 5))\n            plt.imshow(img, cmap=\"gray\")\n            plt.title(f\"Patient {patient} - Study {study} - Frame {i}/{len(imgs)}\")\n            plt.axis(False)\n            plt.show()\n    images = np.stack(images, -1)\n    \n    images = images - np.min(images)\n    images = images / (np.max(images) + 1e-4)\n    images = (images * 255).astype(np.uint8)\n    \n    if plot_image: \n        print('plotting image normalized accross all images')\n        for i in range(images.shape[-1]): \n            if i % 100: continue\n            img = images[:, :, i]\n            plt.figure(figsize=(5, 5))\n            plt.imshow(img, cmap=\"gray\")\n            plt.title(f\"Patient {patient} - Study {study} - Frame {i}/{len(imgs)}\")\n            plt.axis(False)\n            plt.show()\n\n    return images","metadata":{"execution":{"iopub.status.busy":"2024-03-09T05:02:36.822747Z","iopub.execute_input":"2024-03-09T05:02:36.823188Z","iopub.status.idle":"2024-03-09T05:02:36.842644Z","shell.execute_reply.started":"2024-03-09T05:02:36.823155Z","shell.execute_reply":"2024-03-09T05:02:36.841312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"study_path_103 = '/kaggle/input/rsna-2023-abdominal-trauma-detection/train_images/10005/18667'\nimages = study_path_to_3D_image(study_path_103, image_size_seg, plot_image=False)\nimages.shape","metadata":{"execution":{"iopub.status.busy":"2024-03-09T05:03:02.291386Z","iopub.execute_input":"2024-03-09T05:03:02.291878Z","iopub.status.idle":"2024-03-09T05:03:06.153387Z","shell.execute_reply.started":"2024-03-09T05:03:02.291842Z","shell.execute_reply":"2024-03-09T05:03:06.152197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"z_df = pd.DataFrame()\nfor patient in sorted(os.listdir(TRAIN_PATH)):\n    for study in os.listdir(TRAIN_PATH + patient):\n        study_path = os.path.join(TRAIN_PATH + patient, study)\n        images = study_path_to_3D_image(study_path, image_size_seg, plot_image=True, z_df=z_df)\n    break\ndisplay(z_df)","metadata":{"execution":{"iopub.status.busy":"2024-03-09T05:03:19.751103Z","iopub.execute_input":"2024-03-09T05:03:19.751552Z","iopub.status.idle":"2024-03-09T05:03:28.885048Z","shell.execute_reply.started":"2024-03-09T05:03:19.751489Z","shell.execute_reply":"2024-03-09T05:03:28.883531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ram_size_bytes = images.nbytes\n\n# Convert bytes to a more human-readable format (e.g., megabytes)\nram_size_megabytes = ram_size_bytes / (1024 * 1024)\n\nprint(f\"RAM size of the NumPy array: {ram_size_megabytes:.2f} megabytes\")\nprint(f\"approx RAM size of all NumPy array: {ram_size_megabytes * 4700:.2f} megabytes\")","metadata":{"execution":{"iopub.status.busy":"2024-03-09T05:06:10.731779Z","iopub.execute_input":"2024-03-09T05:06:10.732227Z","iopub.status.idle":"2024-03-09T05:06:10.740453Z","shell.execute_reply.started":"2024-03-09T05:06:10.732195Z","shell.execute_reply":"2024-03-09T05:06:10.738968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process(patient, image_size_seg, save_folder=\"\", data_path=\"\", z_df=None):\n    for study in sorted(os.listdir(data_path + patient)):\n        study_path = os.path.join(data_path + patient, study)\n        images = study_path_to_3D_image(study_path, image_size_seg, z_df=z_df)\n        np.save(save_folder + f\"{patient}_{study}.npy\", images)\n        \n        \n#         if isinstance(save_folder, str):\n#             cv2.imwrite(save_folder + f\"{patient}_{study}_{i}.png\", (img * 255).astype(np.uint8))\n#         else:\n#             im = cv2.imencode('.png', (img * 255).astype(np.uint8))[1]\n#             save_folder.writestr(f'{patient}_{study}_{i:04d}.png', im)","metadata":{"execution":{"iopub.status.busy":"2024-03-09T05:06:27.650873Z","iopub.execute_input":"2024-03-09T05:06:27.651296Z","iopub.status.idle":"2024-03-09T05:06:27.659302Z","shell.execute_reply.started":"2024-03-09T05:06:27.651265Z","shell.execute_reply":"2024-03-09T05:06:27.657983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"patients = os.listdir(TRAIN_PATH)\nos.makedirs('train_images')\nz_df = pd.DataFrame()\nfor patient in tqdm(patients):\n    process(patient, image_size_seg, save_folder='train_images', data_path=TRAIN_PATH, z_df=z_df)\nz_df.to_csv('z_df.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-03-09T05:34:08.151833Z","iopub.execute_input":"2024-03-09T05:34:08.152349Z","iopub.status.idle":"2024-03-09T10:29:03.18894Z","shell.execute_reply.started":"2024-03-09T05:34:08.152312Z","shell.execute_reply":"2024-03-09T10:29:03.186158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"z_df.pos_z_ascending.value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-03-09T10:29:26.595113Z","iopub.execute_input":"2024-03-09T10:29:26.596639Z","iopub.status.idle":"2024-03-09T10:29:26.624742Z","shell.execute_reply.started":"2024-03-09T10:29:26.596572Z","shell.execute_reply":"2024-03-09T10:29:26.623735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}