{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport pydicom \nimport nibabel as nib\nimport matplotlib.pyplot as plt\nimport seaborn as sn\nimport os\nfrom glob import glob","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.status.busy":"2023-08-29T05:52:44.652363Z","iopub.execute_input":"2023-08-29T05:52:44.653018Z","iopub.status.idle":"2023-08-29T05:52:46.42531Z","shell.execute_reply.started":"2023-08-29T05:52:44.652979Z","shell.execute_reply":"2023-08-29T05:52:46.424177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Overview of Train Image Data and Segmentation Data\n\nThe training data consists of two main folders: 'train_images' and 'segmentations'. The data is used as features for training machine learning models. Let's take a closer look:\n\n1. 'train_images' Folder:\n   - This folder contains 3087 or more individual patient folders.\n   - Each patient folder is named using a unique Patient ID.\n   - Inside each patient folder, there are subfolders named with Series IDs.\n   - Each Series ID subfolder contains several DICOM image files.\n   - These DICOM files hold the CT scan data in DICOM format.\n   - The DICOM images provide detailed cross-sectional views of the patients' anatomy.\n\n2. 'segmentations' Folder:\n   - The 'segmentations' folder contains 206 model-generated pixel-level annotations.\n   - These annotations are available for a subset of the scans in the training set.\n   - The annotations are generated for relevant organs and some major bones.\n   - The annotation data is stored in the NIfTI file format.\n   - Each annotation file is named using a Series ID.\n   - Note that not all Series IDs in the 'train_images' folder have corresponding segmentation files.\n\nOverall, the 'train_images' folder contains CT scan data that captures patient anatomy, while the 'segmentations' folder provides annotations that highlight specific organs and bones for a subset of the scans.\n","metadata":{}},{"cell_type":"markdown","source":"### Visualizing Sample Train Image Files: Patient ID 10004, Series ID 21057\"\n\n","metadata":{}},{"cell_type":"code","source":"image_file = glob(\"/kaggle/input/rsna-2023-abdominal-trauma-detection/train_images/10004/21057/*.dcm\")\nprint(len(image_file))\nplt.figure(figsize=(20, 20))\n\nfor i in range(8):\n    ax = plt.subplot(4, 4, i + 1)\n    image_path = image_file[i]\n    ds = pydicom.dcmread(image_path)\n    \n    plt.axis('off')\n    plt.imshow(ds.pixel_array,cmap='gray')\n","metadata":{"execution":{"iopub.status.busy":"2023-08-29T05:52:46.427502Z","iopub.execute_input":"2023-08-29T05:52:46.427863Z","iopub.status.idle":"2023-08-29T05:52:48.033744Z","shell.execute_reply.started":"2023-08-29T05:52:46.42783Z","shell.execute_reply":"2023-08-29T05:52:48.032559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This partcular seried ID contains 1022 CT scan data files","metadata":{}},{"cell_type":"markdown","source":"### Visualizing Sample of Segmenatation images ","metadata":{}},{"cell_type":"code","source":"seg_image_file = glob(\"/kaggle/input/rsna-2023-abdominal-trauma-detection/segmentations/*.nii\")\nprint(len(seg_image_file))\nplt.figure(figsize=(20, 20))\n\nfor i in range(8):\n    ax = plt.subplot(4, 4, i + 1)\n    image_path = seg_image_file[i]\n    nii_img = nib.load(image_path).get_fdata()\n    nib_image = nii_img[:,:,nii_img.shape[2]//2]\n    \n    plt.axis('off')\n    plt.imshow(nib_image,cmap='gray')","metadata":{"execution":{"iopub.status.busy":"2023-08-29T05:52:48.035075Z","iopub.execute_input":"2023-08-29T05:52:48.035418Z","iopub.status.idle":"2023-08-29T05:53:07.950889Z","shell.execute_reply.started":"2023-08-29T05:52:48.035387Z","shell.execute_reply":"2023-08-29T05:53:07.949889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The 'segmentations' folder contains 206 model-generated pixel-level annotations","metadata":{}},{"cell_type":"markdown","source":"### DICOM file to 3D numpy segmetation and Visualization","metadata":{}},{"cell_type":"markdown","source":"#### Checking whether the random series_id has segmentation image","metadata":{}},{"cell_type":"code","source":"PATH = '/kaggle/input/rsna-2023-abdominal-trauma-detection'\nid = 21057 #input series_id\n\na = glob(f'{PATH}/segmentations/*.nii')\n\nif f'{PATH}/segmentations/{id}.nii' in a:\n    print('The series id exists')\nelse:\n    print('The series id does not exist')","metadata":{"execution":{"iopub.status.busy":"2023-08-29T05:53:07.953966Z","iopub.execute_input":"2023-08-29T05:53:07.954487Z","iopub.status.idle":"2023-08-29T05:53:07.961831Z","shell.execute_reply.started":"2023-08-29T05:53:07.954445Z","shell.execute_reply":"2023-08-29T05:53:07.960628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(f'{PATH}/train.csv')\ntrain_series_meta = pd.read_csv(f'{PATH}/train_series_meta.csv')\n\ndf = pd.merge(train_series_meta, train, how='inner', on='patient_id')\ndf","metadata":{"execution":{"iopub.status.busy":"2023-08-29T05:53:07.96527Z","iopub.execute_input":"2023-08-29T05:53:07.96564Z","iopub.status.idle":"2023-08-29T05:53:08.039749Z","shell.execute_reply.started":"2023-08-29T05:53:07.96561Z","shell.execute_reply":"2023-08-29T05:53:08.038566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Retrieve the patient_id associated with a given series_id = 21057","metadata":{}},{"cell_type":"code","source":"patient_id = df[df['series_id'] == id]['patient_id'].iloc[0]\npatient_id","metadata":{"execution":{"iopub.status.busy":"2023-08-29T05:53:08.040952Z","iopub.execute_input":"2023-08-29T05:53:08.041292Z","iopub.status.idle":"2023-08-29T05:53:08.050779Z","shell.execute_reply.started":"2023-08-29T05:53:08.041254Z","shell.execute_reply":"2023-08-29T05:53:08.049447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Getting the train image and segmentation file paths","metadata":{}},{"cell_type":"code","source":"seg_filepath = f'{PATH}/segmentations/{id}.nii'\nimg_filepath = f'{PATH}/train_images/{patient_id}/{id}'\n\nprint(\"Segmentation File Path:\", seg_filepath)\nprint(\"Image File Path:\", img_filepath)","metadata":{"execution":{"iopub.status.busy":"2023-08-29T05:53:08.052034Z","iopub.execute_input":"2023-08-29T05:53:08.052348Z","iopub.status.idle":"2023-08-29T05:53:08.061201Z","shell.execute_reply.started":"2023-08-29T05:53:08.05232Z","shell.execute_reply":"2023-08-29T05:53:08.060187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### create_3D_scans function\nThis function called create_3D_scans() that aims to create a 3D volume from a folder containing DICOM files. It performs several steps including reading DICOM files, rescaling pixel values, clipping pixel values based on windowing parameters, and creating a 3D volume by stacking 2D slices. The purpose of the function is to create 3D scan data.","metadata":{}},{"cell_type":"code","source":"def create_3D_scans(folder, downsample_rate=1):\n    \n    #reads the filenames in the folder, extracts the IDs from the filenames, sorts them, and constructs a list of filenames to read in order.\n    filenames = os.listdir(folder)\n    filenames = [int(filename.split('.')[0]) for filename in filenames]\n    filenames = sorted(filenames)\n    filenames = [str(filename) + '.dcm' for filename in filenames]\n        \n    volume = []\n    for filename in filenames[::downsample_rate]:\n        filepath = os.path.join(folder, filename)\n        ds = pydicom.dcmread(filepath)\n        #This extracts the pixel array data from the DICOM object. The pixel array represents the 2D image slice\n        image = ds.pixel_array\n        \n        # find rescale params\n        if (\"RescaleIntercept\" in ds) and (\"RescaleSlope\" in ds):\n            intercept = float(ds.RescaleIntercept)\n            slope = float(ds.RescaleSlope)\n    \n        # find clipping params\n        center = int(ds.WindowCenter)\n        width = int(ds.WindowWidth)\n        low = center - width / 2\n        high = center + width / 2    \n        image = (image * slope) + intercept\n        image = np.clip(image, low, high)\n\n        image = (image / np.max(image) * 255).astype(np.int16)#Normalize the pixel values to a range between 0 and 255 and convert them to 16-bit integers. This step is usually done for visualization purpose\n        image = image[::downsample_rate, ::downsample_rate]\n        volume.append( image )\n    \n    volume = np.stack(volume, axis=0)\n    return volume","metadata":{"execution":{"iopub.status.busy":"2023-08-29T05:53:08.062574Z","iopub.execute_input":"2023-08-29T05:53:08.062973Z","iopub.status.idle":"2023-08-29T05:53:08.073437Z","shell.execute_reply.started":"2023-08-29T05:53:08.062923Z","shell.execute_reply":"2023-08-29T05:53:08.072228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### create_3D_segmentations Function\n\nThe function create_3D_segmentations() is designed to process NIfTI files (segmentation files) and create a 3D segmentation volume from them. It involves several operations to orient and downsample the data.\n","metadata":{}},{"cell_type":"code","source":"def create_3D_segmentations(filepath, downsample_rate=1):\n    img = nib.load(filepath).get_fdata()\n    img = np.transpose(img, [2, 1, 0])\n    img = np.rot90(img, -1, (1,2))\n    img = img[::-1,:,:]\n    img = np.transpose(img, [2, 1, 0])\n    img = img[::downsample_rate, ::downsample_rate, ::downsample_rate]\n    return img","metadata":{"execution":{"iopub.status.busy":"2023-08-29T05:53:08.074841Z","iopub.execute_input":"2023-08-29T05:53:08.075177Z","iopub.status.idle":"2023-08-29T05:53:08.08772Z","shell.execute_reply.started":"2023-08-29T05:53:08.075147Z","shell.execute_reply":"2023-08-29T05:53:08.086753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Visualize the preprocessed 3D train image scan data for series_id = 21057","metadata":{}},{"cell_type":"code","source":"volume = create_3D_scans(img_filepath)\nvolume = volume.transpose(1, 2, 0)\nplt.imshow(volume[:, :,volume.shape[2]//2],cmap='gray')\nplt.axis(\"off\")  \nplt.show()\nprint(volume.shape)\n","metadata":{"execution":{"iopub.status.busy":"2023-08-29T05:54:06.801814Z","iopub.execute_input":"2023-08-29T05:54:06.802173Z","iopub.status.idle":"2023-08-29T05:54:21.148702Z","shell.execute_reply.started":"2023-08-29T05:54:06.802144Z","shell.execute_reply":"2023-08-29T05:54:21.147526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Visualize the preprocessed 3D segemantation image data for series_id = 21057","metadata":{}},{"cell_type":"code","source":"volume_seg = create_3D_segmentations(seg_filepath)\nplt.imshow(volume_seg[:, :,volume_seg.shape[2]//2],cmap='gray')\nplt.axis(\"off\")  \nplt.show()\nprint(volume_seg.shape)","metadata":{"execution":{"iopub.status.busy":"2023-08-29T05:54:27.961971Z","iopub.execute_input":"2023-08-29T05:54:27.962419Z","iopub.status.idle":"2023-08-29T05:54:36.322488Z","shell.execute_reply.started":"2023-08-29T05:54:27.962383Z","shell.execute_reply":"2023-08-29T05:54:36.32135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Masked images\n\ndisplaying a slice from a 3D volume, a corresponding slice from a segmented volume, and an overlay of the two","metadata":{}},{"cell_type":"code","source":"fig = plt.figure(figsize=(16,16))\n\nax1 = fig.add_subplot(131)\nax1.imshow(volume[:,:,volume.shape[2]//2], cmap = 'gray')\nax1.set_title('Original Image', fontsize=14)\n\nax2 = fig.add_subplot(132)\nax2.imshow(volume_seg[:,:,volume_seg.shape[2]//2], cmap = 'gray')\nax2.set_title('Segmented Image', fontsize=14)\n\nax3 = fig.add_subplot(133)\nax3.imshow(volume[:,:,volume.shape[2]//2]*np.where(volume_seg[:,:,volume_seg.shape[2]//2]>0,1,0), cmap = 'gray')\nax3.set_title('Overlay of Original and Segmented', fontsize=14)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-08-29T05:54:39.462204Z","iopub.execute_input":"2023-08-29T05:54:39.463446Z","iopub.status.idle":"2023-08-29T05:54:40.463915Z","shell.execute_reply.started":"2023-08-29T05:54:39.463401Z","shell.execute_reply":"2023-08-29T05:54:40.462629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}