{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<img src=\"https://i.ibb.co/JCvbJjJ/image.png\" alt=\"banner\" border=\"0\">\n\nThe goal of this competition is to identify fractures in CT scans of the cervical spine (neck) at both the level of a single vertebrae and the entire patient. Quickly detecting and determining the location of any vertebral fractures is essential to prevent neurologic deterioration and paralysis after trauma.\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"markdown","source":"# Usual imports and configurations","metadata":{}},{"cell_type":"code","source":"!pip install -qU \"python-gdcm\" pydicom pylibjpeg \"opencv-python-headless\"","metadata":{"_kg_hide-input":false,"_kg_hide-output":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2022-08-21T05:54:42.662308Z","iopub.execute_input":"2022-08-21T05:54:42.663371Z","iopub.status.idle":"2022-08-21T05:55:01.60982Z","shell.execute_reply.started":"2022-08-21T05:54:42.663242Z","shell.execute_reply":"2022-08-21T05:55:01.608693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport sys\nimport cv2\nimport glob\nimport random\nimport pydicom\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport nibabel as nib\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\nfrom matplotlib import cm\nfrom PIL import Image as Img\nfrom sklearn.preprocessing import MinMaxScaler\nfrom sklearn.model_selection import train_test_split\nfrom IPython.display import display_html, display, Image\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut # When the dataset requires a modality LUT or rescale operation as part of the Modality LUT module then that must be applied before any windowing operation.\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\nplt.rcParams.update({'font.size': 16})","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2022-08-21T05:55:01.613524Z","iopub.execute_input":"2022-08-21T05:55:01.614041Z","iopub.status.idle":"2022-08-21T05:55:02.759157Z","shell.execute_reply.started":"2022-08-21T05:55:01.613991Z","shell.execute_reply":"2022-08-21T05:55:02.757357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG :\n    root_dir = \"../input/rsna-2022-cervical-spine-fracture-detection/\"\n    train_dir = \"../input/rsna-2022-cervical-spine-fracture-detection/train_images\"\n    train_csv_path = \"../input/rsna-2022-cervical-spine-fracture-detection/train.csv\"\n    test_csv_path = \"../input/rsna-2022-cervical-spine-fracture-detection/test.csv\"\n    sample_csv_path = \"../input/rsna-2022-cervical-spine-fracture-detection/sample_submission.csv\"\n    segmentations = \"../input/rsna-2022-cervical-spine-fracture-detection/segmentations\"\n    seed = 42","metadata":{"execution":{"iopub.status.busy":"2022-08-21T05:55:02.760757Z","iopub.execute_input":"2022-08-21T05:55:02.761102Z","iopub.status.idle":"2022-08-21T05:55:02.768893Z","shell.execute_reply.started":"2022-08-21T05:55:02.76107Z","shell.execute_reply":"2022-08-21T05:55:02.767338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class clr:\n    S = '\\033[1m' + '\\033[94m'\n    E = '\\033[0m'\n    \nmy_colors = [\"#5EAFD9\", \"#449DD1\", \"#3977BB\", \n             \"#2D51A5\", \"#5C4C8F\", \"#8B4679\",\n             \"#C53D4C\", \"#E23836\", \"#FF4633\", \"#FF5746\"]\n\nprint(clr.S+\"Notebook Color Schemes:\"+clr.E)\nsns.palplot(sns.color_palette(my_colors))\nplt.show()","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2022-08-21T05:55:02.77159Z","iopub.execute_input":"2022-08-21T05:55:02.77219Z","iopub.status.idle":"2022-08-21T05:55:02.90773Z","shell.execute_reply.started":"2022-08-21T05:55:02.772156Z","shell.execute_reply":"2022-08-21T05:55:02.906139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"lot in this notebook is learnt from https://www.kaggle.com/code/andradaolteanu/rsna-fracture-detection-exploratory-baseline","metadata":{}},{"cell_type":"markdown","source":"## Helper functions","metadata":{}},{"cell_type":"code","source":"def set_seed(seed=42):\n    \"\"\"\n    sets the integer starting value used in generating random numbers.\n    comment out packages/frameworks not in use as per your choice \n    \"\"\"\n    np.random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    random.seed(seed)\n#     tf.random.set_seed(seed)\n#     torch.manual_seed(seed)\n#     torch.cuda.manual_seed(seed)\n#     torch.backends.cudnn.deterministic = True\n    \ndef read_csvs(train_csv_path, test_csv_path, sample_csv):\n    '''Reads in all .csv files.'''\n    \n    train = pd.read_csv(train_csv_path)\n    test = pd.read_csv(test_csv_path)\n    ss = pd.read_csv(sample_csv)\n    \n    return train, test, ss\n\ndef get_csv_info(df, name=\"Default\"):\n    '''Prints main information for the specified dataframe'''\n    \n    print(clr.S+f\"=== {name} ===\"+clr.E)\n    print(clr.S+f\"Shape:\"+clr.E, df.shape)\n    print(clr.S+f\"Missing Values:\"+clr.E, df.isna().sum().sum(), \"total missing datapoints.\")\n    print(clr.S+\"Columns:\"+clr.E, list(df.columns), \"\\n\")\n    \n    display_html(df.head())\n    print(\"\\n\")\n    \ndef show_values_on_bars(axs, h_v=\"v\", space=0.4):\n    '''Plots the value at the end of the a seaborn barplot.\n    axs: the ax of the plot\n    h_v: weather or not the barplot is vertical/ horizontal'''\n    \n    def _show_on_single_plot(ax):\n        if h_v == \"v\":\n            for p in ax.patches:\n                _x = p.get_x() + p.get_width() / 2\n                _y = p.get_y() + p.get_height()\n                value = int(p.get_height())\n                ax.text(_x, _y, format(value, ','), ha=\"center\") \n        elif h_v == \"h\":\n            for p in ax.patches:\n                _x = p.get_x() + p.get_width() + float(space)\n                _y = p.get_y() + p.get_height()\n                value = int(p.get_width())\n                ax.text(_x, _y, format(value, ','), ha=\"left\")\n\n    if isinstance(axs, np.ndarray):\n        for idx, ax in np.ndenumerate(axs):\n            _show_on_single_plot(ax)\n    else:\n        _show_on_single_plot(axs)\n        \ndef show_dcm_images(StudyInstanceUID, rows=4, cols=8, random=0):\n    '''Show .dcm images based on id.'''\n    \n    N = rows*cols\n    fig, axes = plt.subplots(nrows=rows, ncols=cols, figsize=(24,15))\n    fig.suptitle(f'ID: {StudyInstanceUID}', weight=\"bold\", size=16)\n    \n    # Get .dcm paths\n    sample_study_path = glob.glob(os.path.join(CFG.train_dir, train_df[train_df[\"StudyInstanceUID\"] == StudyInstanceUID].values[0,0], \"**.dcm\"))\n    sample_study_path = sorted(sample_study_path, key=lambda x : int(x.split('/')[-1].split('.')[0]))\n    print(clr.S+\"Number of TOTAL Slices:\"+clr.E, len(sample_study_path))\n    \n    # Get corresponging datasets and images\n    datasets = [pydicom.dcmread(path) for path in sample_study_path]\n    images = [apply_voi_lut(dataset.pixel_array, dataset) for dataset in datasets]\n    images = images[random:random+N]\n\n    # Loop through the information\n    for data, img, i in zip(datasets, images, range(N)):\n        slice_no = data.SOPInstanceUID.split('.')[-1]\n    \n        # Plot the image\n        x = i // cols\n        y = i % cols\n\n        axes[x, y].imshow(img, cmap=\"gray\")\n        axes[x, y].set_title(f\"Slice: {slice_no}\", \n                  fontsize=14, weight='bold')\n        axes[x, y].axis('off');\n    plt.show()\n    \ndef show_study_animation(StudyInstanceID):\n    \"\"\"creates a animation of all scans and saves it in a .gif file\n    \"\"\"\n    sample_study_path = glob.glob(os.path.join(CFG.train_dir, train_df[train_df[\"StudyInstanceUID\"] == StudyInstanceID].values[0,0], \"**.dcm\"))\n    sample_study_path = sorted(sample_study_path, key=lambda x : int(x.split('/')[-1].split('.')[0]))\n    \n    # Get corresponging datasets and images\n    datasets = [pydicom.dcmread(path) for path in sample_study_path]\n    images = [Img.fromarray(apply_voi_lut(dataset.pixel_array, dataset)) for dataset in tqdm(datasets)]\n    \n    images[0].save(f\"{StudyInstanceID}.gif\", format='GIF', append_images=images[1:], save_all=True, duration=35, loop=0)\n    \n\ndef multi_dim_plot(multi_dim_array, num_slices=64):\n    \"\"\"\n    For plotting segmentations\n    \"\"\"\n    fig = plt.figure(figsize=(30, 30))\n    plt.title(f'Plotting first {num_slices} slices out of {multi_dim_array.shape}')\n    plt.yticks([])\n    plt.xticks([])\n    \n    for i in range(num_slices):\n        ax = fig.add_subplot(8, 8, i + 1)\n        plt.imshow(multi_dim_array[..., :num_slices][..., i], cmap=\"bone\")\n        plt.axis(\"off\")\n    plt.show()","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2022-08-21T05:55:02.910186Z","iopub.execute_input":"2022-08-21T05:55:02.911261Z","iopub.status.idle":"2022-08-21T05:55:02.977978Z","shell.execute_reply.started":"2022-08-21T05:55:02.911181Z","shell.execute_reply":"2022-08-21T05:55:02.976591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"set_seed(CFG.seed)","metadata":{"execution":{"iopub.status.busy":"2022-08-21T05:55:02.979841Z","iopub.execute_input":"2022-08-21T05:55:02.980313Z","iopub.status.idle":"2022-08-21T05:55:02.993844Z","shell.execute_reply.started":"2022-08-21T05:55:02.980268Z","shell.execute_reply":"2022-08-21T05:55:02.992737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Directory structure","metadata":{}},{"cell_type":"code","source":"!ls ../input/rsna-2022-cervical-spine-fracture-detection","metadata":{"execution":{"iopub.status.busy":"2022-08-21T05:55:02.995333Z","iopub.execute_input":"2022-08-21T05:55:02.995933Z","iopub.status.idle":"2022-08-21T05:55:04.157002Z","shell.execute_reply.started":"2022-08-21T05:55:02.995882Z","shell.execute_reply":"2022-08-21T05:55:04.15523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"train.csv Metadata for the train test set.\n\n* `StudyInstanceUID` - The study ID. There is one unique study ID for each patient scan.\n* `patient_overall` - One of the target columns. The patient level outcome, i.e. if any of the vertebrae are fractured.\n* `C[1-7]` - The other target columns. Whether the given vertebrae is fractured. See this diagram for the real location of each vertbrae in the spine.\n\ntest.csv Metadata for the test set prediction structure. Only the first few rows of the test set are available for download.\n\n* `row_id` - The row ID. This will match the same column in the sample submission file.\n* `StudyInstanceUID` - The study ID.\n* `prediction_type` - Which one of the eight target columns needs a prediction in this row.","metadata":{}},{"cell_type":"markdown","source":"# Lets explore train_images/","metadata":{}},{"cell_type":"code","source":"!ls \"../input/rsna-2022-cervical-spine-fracture-detection/train_images\" | head ","metadata":{"execution":{"iopub.status.busy":"2022-08-21T05:55:04.159506Z","iopub.execute_input":"2022-08-21T05:55:04.160364Z","iopub.status.idle":"2022-08-21T05:55:05.418058Z","shell.execute_reply.started":"2022-08-21T05:55:04.16031Z","shell.execute_reply":"2022-08-21T05:55:05.416484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls \"../input/rsna-2022-cervical-spine-fracture-detection/train_images/1.2.826.0.1.3680043.10001\" ","metadata":{"execution":{"iopub.status.busy":"2022-08-21T05:55:05.420336Z","iopub.execute_input":"2022-08-21T05:55:05.420832Z","iopub.status.idle":"2022-08-21T05:55:06.584628Z","shell.execute_reply.started":"2022-08-21T05:55:05.420782Z","shell.execute_reply":"2022-08-21T05:55:06.58268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Okay so, in `train_images/` we have folders named with `StudyInstanceUID` and inside each folder we have sequential dicom images.","metadata":{}},{"cell_type":"markdown","source":"# CSV info","metadata":{}},{"cell_type":"code","source":"train_df, test_df, sample_sub_df = read_csvs(CFG.train_csv_path, CFG.test_csv_path, CFG.sample_csv_path)","metadata":{"execution":{"iopub.status.busy":"2022-08-21T05:55:06.588934Z","iopub.execute_input":"2022-08-21T05:55:06.589403Z","iopub.status.idle":"2022-08-21T05:55:06.624705Z","shell.execute_reply.started":"2022-08-21T05:55:06.589363Z","shell.execute_reply":"2022-08-21T05:55:06.62375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"get_csv_info(train_df, \"train.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-08-21T05:55:06.626779Z","iopub.execute_input":"2022-08-21T05:55:06.627681Z","iopub.status.idle":"2022-08-21T05:55:06.651117Z","shell.execute_reply.started":"2022-08-21T05:55:06.627629Z","shell.execute_reply":"2022-08-21T05:55:06.649273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"get_csv_info(test_df, \"test.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-08-21T05:55:06.652932Z","iopub.execute_input":"2022-08-21T05:55:06.653298Z","iopub.status.idle":"2022-08-21T05:55:06.666786Z","shell.execute_reply.started":"2022-08-21T05:55:06.653266Z","shell.execute_reply":"2022-08-21T05:55:06.665082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Lets check how many patients have a fractured vertbrae","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(24, 12))\nax = sns.countplot(data=train_df, x=\"patient_overall\", palette=[my_colors[1], my_colors[-1]]);\nshow_values_on_bars(ax, h_v=\"v\", space=0.4)\nax.set_title(\"Fractured patient count\", weight=\"bold\", size=19)\nsns.despine(right=True, top=True, left=True);","metadata":{"execution":{"iopub.status.busy":"2022-08-21T05:55:06.668685Z","iopub.execute_input":"2022-08-21T05:55:06.669167Z","iopub.status.idle":"2022-08-21T05:55:06.968065Z","shell.execute_reply.started":"2022-08-21T05:55:06.669121Z","shell.execute_reply":"2022-08-21T05:55:06.966506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Fairly balanced...","metadata":{}},{"cell_type":"markdown","source":"# Count of Type of Fractured Vertebra","metadata":{}},{"cell_type":"code","source":"dt = pd.melt(train_df, \n             id_vars=['StudyInstanceUID', 'patient_overall'],\n             var_name=\"Vertebra\",\n             value_name=\"Flag\")\n\nplt.figure(figsize=(24, 12))\nax = sns.countplot(data=dt, x=\"Vertebra\", hue=\"Flag\", palette=[my_colors[1], my_colors[-1]])\nshow_values_on_bars(ax, h_v=\"v\", space=0.4)\nax.set_title(\"Fractured Vertebra type count\", weight=\"bold\", size=19)\nsns.despine(right=True, top=True, left=True);","metadata":{"execution":{"iopub.status.busy":"2022-08-21T05:55:06.969892Z","iopub.execute_input":"2022-08-21T05:55:06.970512Z","iopub.status.idle":"2022-08-21T05:55:07.354473Z","shell.execute_reply.started":"2022-08-21T05:55:06.970469Z","shell.execute_reply":"2022-08-21T05:55:07.353148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**C7** is the type of vertebra having most fractured bones than the others","metadata":{}},{"cell_type":"markdown","source":"# Number of Scans per Study","metadata":{}},{"cell_type":"code","source":"no_of_scans = []\nfor study in tqdm(os.listdir(CFG.train_dir)) :\n    no_of_scans.append(len(glob.glob(os.path.join(CFG.train_dir, study,  \"**.dcm\"))))","metadata":{"execution":{"iopub.status.busy":"2022-08-21T05:55:07.355984Z","iopub.execute_input":"2022-08-21T05:55:07.356366Z","iopub.status.idle":"2022-08-21T05:56:16.623244Z","shell.execute_reply.started":"2022-08-21T05:55:07.356334Z","shell.execute_reply":"2022-08-21T05:56:16.622121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.boxplot(x=no_of_scans).set_title(\"No of scans per StudyInstanceUID\");","metadata":{"execution":{"iopub.status.busy":"2022-08-21T05:56:16.624757Z","iopub.execute_input":"2022-08-21T05:56:16.625616Z","iopub.status.idle":"2022-08-21T05:56:16.79775Z","shell.execute_reply.started":"2022-08-21T05:56:16.625581Z","shell.execute_reply":"2022-08-21T05:56:16.796899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"min(no_of_scans)","metadata":{"execution":{"iopub.status.busy":"2022-08-21T05:56:16.799288Z","iopub.execute_input":"2022-08-21T05:56:16.799885Z","iopub.status.idle":"2022-08-21T05:56:16.806576Z","shell.execute_reply.started":"2022-08-21T05:56:16.799852Z","shell.execute_reply":"2022-08-21T05:56:16.805451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"so its quite clear that we don't have consistent number of scans per `StudyInstanceID`  \nmost of the scans per study are in the range `200-400`","metadata":{}},{"cell_type":"markdown","source":"## Visualizing train_images","metadata":{}},{"cell_type":"code","source":"study_id= random.choice(train_df[\"StudyInstanceUID\"])\nshow_dcm_images(study_id)","metadata":{"execution":{"iopub.status.busy":"2022-08-21T05:56:16.808113Z","iopub.execute_input":"2022-08-21T05:56:16.808872Z","iopub.status.idle":"2022-08-21T05:56:37.042331Z","shell.execute_reply.started":"2022-08-21T05:56:16.808839Z","shell.execute_reply":"2022-08-21T05:56:37.040629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_study_animation(study_id)\nImage(f'{study_id}.gif')","metadata":{"execution":{"iopub.status.busy":"2022-08-21T05:56:37.045545Z","iopub.execute_input":"2022-08-21T05:56:37.046334Z","iopub.status.idle":"2022-08-21T05:56:44.955407Z","shell.execute_reply.started":"2022-08-21T05:56:37.046289Z","shell.execute_reply":"2022-08-21T05:56:44.954299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"not that good \n> GIF can be see in the `Notebook Output`, if you know how to make it more clearer lemme know in the comments !","metadata":{}},{"cell_type":"code","source":"!ls ../input/rsna-2022-cervical-spine-fracture-detection","metadata":{"execution":{"iopub.status.busy":"2022-08-21T05:56:44.957471Z","iopub.execute_input":"2022-08-21T05:56:44.959281Z","iopub.status.idle":"2022-08-21T05:56:46.208918Z","shell.execute_reply.started":"2022-08-21T05:56:44.959229Z","shell.execute_reply":"2022-08-21T05:56:46.207437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Segmentations\n\nQuick note :   \nPixel level annotations for a subset of the training set. This data is provided in the nifti file format.  \nA portion of the imaging datasets have been segmented automatically using a 3D UNET model, and radiologists modified and approved the segmentations. The provided segmentation labels have values of 1 to 7 for C1 to C7 (seven cervical vertebrae) and 8 to 19 for T1 to T12 (twelve thoracic vertebrae are located in the center of your upper and middle back), and 0 for everything else. As we focused on the cervical spine, all scans have C1 to C7 labels but not all thoracic labels.    \n\nPlease be aware that the NIFTI files consist of segmentation in the sagittal plane, while the DICOM files are in the axial plane. Please use the NIFTI header information to determine the appropriate orientation such that the DICOM images and segmentation match. Otherwise, you run the risk of having the segmentations flipped in the Z axis and mirrored in the X axis.  ","metadata":{}},{"cell_type":"code","source":"segmented_scans = os.listdir(CFG.segmentations)","metadata":{"execution":{"iopub.status.busy":"2022-08-21T05:56:46.211575Z","iopub.execute_input":"2022-08-21T05:56:46.212062Z","iopub.status.idle":"2022-08-21T05:56:46.230572Z","shell.execute_reply.started":"2022-08-21T05:56:46.212016Z","shell.execute_reply":"2022-08-21T05:56:46.229321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(segmented_scans) / train_df.shape[0] * 100","metadata":{"execution":{"iopub.status.busy":"2022-08-21T06:08:37.87297Z","iopub.execute_input":"2022-08-21T06:08:37.873434Z","iopub.status.idle":"2022-08-21T06:08:37.881154Z","shell.execute_reply.started":"2022-08-21T06:08:37.873399Z","shell.execute_reply":"2022-08-21T06:08:37.880284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Only ~4% of the `StudyInstanceUID` has segmentations   ","metadata":{}},{"cell_type":"code","source":"os.path.join(CFG.segmentations, study,  \"**.nii\")","metadata":{"execution":{"iopub.status.busy":"2022-08-21T06:11:12.637029Z","iopub.execute_input":"2022-08-21T06:11:12.638604Z","iopub.status.idle":"2022-08-21T06:11:12.645636Z","shell.execute_reply.started":"2022-08-21T06:11:12.638533Z","shell.execute_reply":"2022-08-21T06:11:12.644827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"no_of_segmentations = []\nfor study in tqdm(segmented_scans) :\n    img_path = os.path.join(CFG.segmentations, study)\n    nimg = nib.load(img_path)\n    nimg_array = np.asanyarray(nimg.dataobj)\n    no_of_segmentations.append(nimg_array.shape[-1])","metadata":{"execution":{"iopub.status.busy":"2022-08-21T06:13:24.858967Z","iopub.execute_input":"2022-08-21T06:13:24.859439Z","iopub.status.idle":"2022-08-21T06:13:26.228398Z","shell.execute_reply.started":"2022-08-21T06:13:24.859404Z","shell.execute_reply":"2022-08-21T06:13:26.226755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.boxplot(x=no_of_segmentations).set_title(\"No of segmentations per StudyInstanceUID\");","metadata":{"execution":{"iopub.status.busy":"2022-08-21T06:13:27.392542Z","iopub.execute_input":"2022-08-21T06:13:27.393588Z","iopub.status.idle":"2022-08-21T06:13:27.591784Z","shell.execute_reply.started":"2022-08-21T06:13:27.393541Z","shell.execute_reply":"2022-08-21T06:13:27.59073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sum(no_of_segmentations)","metadata":{"execution":{"iopub.status.busy":"2022-08-21T06:21:34.611592Z","iopub.execute_input":"2022-08-21T06:21:34.612027Z","iopub.status.idle":"2022-08-21T06:21:34.621675Z","shell.execute_reply.started":"2022-08-21T06:21:34.611994Z","shell.execute_reply":"2022-08-21T06:21:34.620324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Each study whose segmentations are available contains `250-400` segmentations","metadata":{}},{"cell_type":"code","source":"img_path = os.path.join(CFG.segmentations, segmented_scans[0])\nnimg = nib.load(img_path)\nnimg_array = np.asanyarray(nimg.dataobj)","metadata":{"execution":{"iopub.status.busy":"2022-08-20T08:51:10.033542Z","iopub.execute_input":"2022-08-20T08:51:10.034375Z","iopub.status.idle":"2022-08-20T08:51:10.059493Z","shell.execute_reply.started":"2022-08-20T08:51:10.03434Z","shell.execute_reply":"2022-08-20T08:51:10.058281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.unique(nimg_array)","metadata":{"execution":{"iopub.status.busy":"2022-08-20T08:51:10.060838Z","iopub.execute_input":"2022-08-20T08:51:10.061398Z","iopub.status.idle":"2022-08-20T08:51:12.299101Z","shell.execute_reply.started":"2022-08-20T08:51:10.061363Z","shell.execute_reply":"2022-08-20T08:51:12.297808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"`1-7` stands for `C1-C7` but `8-9` idk ","metadata":{}},{"cell_type":"code","source":"nimg_array.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-20T08:51:12.300622Z","iopub.execute_input":"2022-08-20T08:51:12.301091Z","iopub.status.idle":"2022-08-20T08:51:12.309278Z","shell.execute_reply.started":"2022-08-20T08:51:12.301049Z","shell.execute_reply":"2022-08-20T08:51:12.308055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"multi_dim_plot(nimg_array)","metadata":{"execution":{"iopub.status.busy":"2022-08-20T08:51:12.311097Z","iopub.execute_input":"2022-08-20T08:51:12.311549Z","iopub.status.idle":"2022-08-20T08:51:16.829216Z","shell.execute_reply.started":"2022-08-20T08:51:12.311515Z","shell.execute_reply":"2022-08-20T08:51:16.828175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"looks like segmentations are \"flipped in the Z axis and mirrored in the X axis\" as mentioned before but I was not able to find anything on how to correct so lemme know if you've any info about it","metadata":{}},{"cell_type":"markdown","source":"# Bounding Boxes","metadata":{}},{"cell_type":"code","source":"boxes_df =  pd.read_csv(CFG.root_dir + 'train_bounding_boxes.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-20T08:51:16.831167Z","iopub.execute_input":"2022-08-20T08:51:16.831829Z","iopub.status.idle":"2022-08-20T08:51:16.864341Z","shell.execute_reply.started":"2022-08-20T08:51:16.831785Z","shell.execute_reply":"2022-08-20T08:51:16.86312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"boxes_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-20T08:51:16.869455Z","iopub.execute_input":"2022-08-20T08:51:16.870151Z","iopub.status.idle":"2022-08-20T08:51:16.886001Z","shell.execute_reply.started":"2022-08-20T08:51:16.870115Z","shell.execute_reply":"2022-08-20T08:51:16.885027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(np.unique(boxes_df[\"StudyInstanceUID\"]))","metadata":{"execution":{"iopub.status.busy":"2022-08-20T08:51:16.8874Z","iopub.execute_input":"2022-08-20T08:51:16.88771Z","iopub.status.idle":"2022-08-20T08:51:16.897033Z","shell.execute_reply.started":"2022-08-20T08:51:16.887682Z","shell.execute_reply":"2022-08-20T08:51:16.896051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"boxes_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-20T08:51:16.898469Z","iopub.execute_input":"2022-08-20T08:51:16.898804Z","iopub.status.idle":"2022-08-20T08:51:16.910235Z","shell.execute_reply.started":"2022-08-20T08:51:16.898775Z","shell.execute_reply":"2022-08-20T08:51:16.909024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"boxes_df[\"StudyInstanceUID\"].nunique() / train_df[\"StudyInstanceUID\"].nunique() * 100","metadata":{"execution":{"iopub.status.busy":"2022-08-20T08:51:16.912093Z","iopub.execute_input":"2022-08-20T08:51:16.912946Z","iopub.status.idle":"2022-08-20T08:51:16.927166Z","shell.execute_reply.started":"2022-08-20T08:51:16.912902Z","shell.execute_reply":"2022-08-20T08:51:16.925979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"so for subset of the scans(~12%), we've respective bounding boxes ","metadata":{}},{"cell_type":"markdown","source":"# 3D Segmentations\n> This part is completely takens from [@andradaolteanu](https://www.kaggle.com/andradaolteanu) notebook which was again inspired by Seungwon Song's notebook: [Break Time☕️ Just Enjoy drawing 3D cervical spine](https://www.kaggle.com/code/songseungwon/break-time-just-enjoy-drawing-3d-cervical-spine/notebook)","metadata":{}},{"cell_type":"code","source":"import itertools","metadata":{"execution":{"iopub.status.busy":"2022-08-20T08:51:16.928836Z","iopub.execute_input":"2022-08-20T08:51:16.929464Z","iopub.status.idle":"2022-08-20T08:51:16.933476Z","shell.execute_reply.started":"2022-08-20T08:51:16.929431Z","shell.execute_reply":"2022-08-20T08:51:16.932589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nii_paths = glob.glob(\"../input/rsna-2022-cervical-spine-fracture-detection/segmentations/*\")\nprint(clr.S+\"Total paths in [segmentations] folder:\"+clr.E, len(nii_paths))","metadata":{"execution":{"iopub.status.busy":"2022-08-20T08:51:16.934969Z","iopub.execute_input":"2022-08-20T08:51:16.935609Z","iopub.status.idle":"2022-08-20T08:51:16.947126Z","shell.execute_reply.started":"2022-08-20T08:51:16.935577Z","shell.execute_reply":"2022-08-20T08:51:16.945917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nib.load(nii_paths[0]).get_fdata().shape","metadata":{"execution":{"iopub.status.busy":"2022-08-20T08:51:16.9489Z","iopub.execute_input":"2022-08-20T08:51:16.949583Z","iopub.status.idle":"2022-08-20T08:51:17.179602Z","shell.execute_reply.started":"2022-08-20T08:51:16.949541Z","shell.execute_reply":"2022-08-20T08:51:17.177116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class DrawMaskSample():\n    \n    def __init__(self, nii_paths, i, ax, nii=True, scans=None ):\n        self.i = i\n        self.ax = ax\n        if nii : \n            self.nii_sample = nib.load(nii_paths[self.i]).get_fdata()\n        else :\n            self.nii_sample = scans\n        self.get_xyz()\n        self.draw_sample_3d()\n        \n    def get_xyz(self):\n        self.xyz_li = []\n        cnt = 0\n        max_cnt = self.nii_sample.shape[-1]\n        for iter_z, (iter_img) in enumerate(self.nii_sample.transpose(2,0,1)):\n            for iter_x, iter_arr_y in enumerate(iter_img):\n                iter_arr_y = np.where(iter_arr_y)[0]\n                if len(iter_arr_y) >= 1:\n                    iter_arr_y = list(set([iter_arr_y.max(), iter_arr_y.min()]))\n                    xyz = [(iter_x, iter_y, iter_z) for iter_y in iter_arr_y if np.any(iter_y)]\n                    self.xyz_li.append(xyz)\n            cnt += 1\n\n    def draw_sample_3d(self):\n        xyz_matrix = np.array(list(itertools.chain.from_iterable(self.xyz_li)))\n        X = xyz_matrix[:,0]\n        Y = xyz_matrix[:,1]\n        Z = xyz_matrix[:,2]\n\n        self.ax.scatter(X, Y, Z, s=1, alpha=0.05, color='#e3dac9')\n        xlim, ylim, zlim = self.nii_sample.shape\n        self.ax.set_xlim(0, xlim)\n        self.ax.set_ylim(0, ylim)\n        self.ax.set_zlim(0, zlim)\n        self.ax.set_title(f'sample - ({self.i+1})')\n        self.ax.set_yticklabels([])\n        self.ax.set_xticklabels([])\n        self.ax.set_zticklabels([])\n        self.ax.set_facecolor('#081921')","metadata":{"execution":{"iopub.status.busy":"2022-08-20T08:51:17.182296Z","iopub.execute_input":"2022-08-20T08:51:17.183189Z","iopub.status.idle":"2022-08-20T08:51:17.202364Z","shell.execute_reply.started":"2022-08-20T08:51:17.18314Z","shell.execute_reply":"2022-08-20T08:51:17.200878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(24,24))\nfor i in range(9):\n    ax = fig.add_subplot(int(f'33{i+1}'), projection='3d')\n    DrawMaskSample(nii_paths, i, ax)\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-20T08:51:17.204383Z","iopub.execute_input":"2022-08-20T08:51:17.204942Z","iopub.status.idle":"2022-08-20T08:51:57.440676Z","shell.execute_reply.started":"2022-08-20T08:51:17.204894Z","shell.execute_reply":"2022-08-20T08:51:57.439522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"![noise](https://c.tenor.com/N66me6-wOwMAAAAC/nice-good.gif)","metadata":{}}]}