{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":36363,"databundleVersionId":4050810,"sourceType":"competition"}],"dockerImageVersionId":30839,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Pipeline","metadata":{}},{"cell_type":"markdown","source":"## Create small dataset ","metadata":{}},{"cell_type":"code","source":"# -------------------------------\n# Import Necessary Libraries\n# -------------------------------\nimport pandas as pd\nimport os\nimport shutil\nimport zipfile\nimport random\nfrom IPython.display import FileLink, display\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport pydicom\nimport numpy as np\nimport warnings\n\nwarnings.filterwarnings('ignore')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-17T17:10:26.908195Z","iopub.execute_input":"2025-01-17T17:10:26.908709Z","iopub.status.idle":"2025-01-17T17:10:29.01923Z","shell.execute_reply.started":"2025-01-17T17:10:26.908662Z","shell.execute_reply":"2025-01-17T17:10:29.018176Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# -------------------------------\n# Configuration Parameters\n# -------------------------------\n\n# Total number of samples to extract (train + test)\nNUM_SAMPLES = 20  # Adjust as needed\n\n# Define the split ratio (train:test)\nTRAIN_RATIO = 0.7  # 70% train, 30% test\n\nRANDOM_SEED = 42  # For reproducibility\n\n# Kaggle input paths\nTRAIN_CSV_PATH = '/kaggle/input/rsna-2022-cervical-spine-fracture-detection/train.csv'\nBBOXES_CSV_PATH = '/kaggle/input/rsna-2022-cervical-spine-fracture-detection/train_bounding_boxes.csv'\nTRAIN_IMAGES_DIR = '/kaggle/input/rsna-2022-cervical-spine-fracture-detection/train_images'\nTEST_IMAGES_DIR = '/kaggle/input/rsna-2022-cervical-spine-fracture-detection/test_images'\nSEGMENTATIONS_DIR = '/kaggle/input/rsna-2022-cervical-spine-fracture-detection/segmentations'\n\n# Working directories\nWORKING_DIR = '/kaggle/working'\nPARTIAL_TRAIN_IMAGES_DIR = os.path.join(WORKING_DIR, 'train_images_partial')\nPARTIAL_TEST_IMAGES_DIR = os.path.join(WORKING_DIR, 'test_images_partial')\nPARTIAL_SEGMENTATIONS_DIR = os.path.join(WORKING_DIR, 'segmentations_partial')\nPARTIAL_BBOXES_PATH = os.path.join(WORKING_DIR, 'train_bounding_boxes_partial.csv')\nDATASET_DIR = os.path.join(WORKING_DIR, f'RSNA2022_Spine_Dataset_{NUM_SAMPLES}_Samples')\nZIP_FILENAME = f'RSNA2022_Spine_Dataset_{NUM_SAMPLES}_Samples.zip'\n\n# Create necessary directories\nos.makedirs(PARTIAL_TRAIN_IMAGES_DIR, exist_ok=True)\nos.makedirs(PARTIAL_TEST_IMAGES_DIR, exist_ok=True)\nos.makedirs(PARTIAL_SEGMENTATIONS_DIR, exist_ok=True)\nos.makedirs(DATASET_DIR, exist_ok=True)\n\n# -------------------------------\n# Step 1: Load CSV Files\n# -------------------------------\nprint(\"Loading train.csv...\")\ntrain_df = pd.read_csv(TRAIN_CSV_PATH)\nprint(\"First 5 rows of train.csv:\")\ndisplay(train_df.head())\n\nprint(\"\\nLoading train_bounding_boxes.csv...\")\nbboxes_df = pd.read_csv(BBOXES_CSV_PATH)\nprint(\"First 5 rows of train_bounding_boxes.csv:\")\ndisplay(bboxes_df.head())\n\n# -------------------------------\n# Step 2: Inspect Segmentation Files and Match Train UIDs\n# -------------------------------\nprint(\"\\nInspecting segmentation files (train only)...\")\nsegmentation_files = os.listdir(SEGMENTATIONS_DIR)\nprint(f\"Total segmentation files found: {len(segmentation_files)}\")\nprint(\"First 5 segmentation file names:\")\ndisplay(segmentation_files[:5])\n\n# Each segmentation file is named StudyInstanceUID.nii (without extension).\nsegmentation_uids = {os.path.splitext(f)[0] for f in segmentation_files}\n\n# Identify which train UIDs actually have a segmentation file\ntrain_uids_with_seg = set(train_df['StudyInstanceUID'].unique()) & segmentation_uids\nprint(f\"Train UIDs with segmentation files: {len(train_uids_with_seg)}\")\n\n# -------------------------------\n# Step 3: Determine Train/Test Splits\n# -------------------------------\nprint(f\"\\nSelecting {NUM_SAMPLES} total samples, with ~{int(TRAIN_RATIO*100)}% for train (with segmentations) and ~{int((1-TRAIN_RATIO)*100)}% for test.\")\n\nTRAIN_SAMPLES = int(NUM_SAMPLES * TRAIN_RATIO)\nTEST_SAMPLES = NUM_SAMPLES - TRAIN_SAMPLES\n\nrandom.seed(RANDOM_SEED)\n\n# 3.1: Train UIDs that have segmentations\navailable_train_uids = list(train_uids_with_seg)\nprint(f\"Total train UIDs (with seg) available: {len(available_train_uids)}\")\n\nif TRAIN_SAMPLES > len(available_train_uids):\n    print(f\"Requested TRAIN_SAMPLES={TRAIN_SAMPLES} exceeds available train UIDs with seg={len(available_train_uids)}. Adjusting to {len(available_train_uids)}.\")\n    TRAIN_SAMPLES = len(available_train_uids)\n    TEST_SAMPLES = NUM_SAMPLES - TRAIN_SAMPLES\n\n# Randomly pick train UIDs from those that have segmentations\nselected_train_uids = random.sample(available_train_uids, TRAIN_SAMPLES)\n\nprint(f\"\\nSelected {len(selected_train_uids)} Train UIDs (with segmentations):\")\nfor uid in selected_train_uids:\n    print(uid)\n\n# 3.2: Select Test UIDs Based on Actual Folders in test_images/\nprint(\"\\nInspecting test_images/ directory for test UIDs...\")\ntest_dirs_all = [\n    d for d in os.listdir(TEST_IMAGES_DIR)\n    if os.path.isdir(os.path.join(TEST_IMAGES_DIR, d))\n]\nprint(f\"Total test directories found: {len(test_dirs_all)}\")\nprint(\"First 5 test directory names:\", test_dirs_all[:5])\n\nif TEST_SAMPLES > len(test_dirs_all):\n    print(f\"Requested TEST_SAMPLES={TEST_SAMPLES} exceeds available test folders={len(test_dirs_all)}. Adjusting to {len(test_dirs_all)}.\")\n    TEST_SAMPLES = len(test_dirs_all)\n\nselected_test_uids = random.sample(test_dirs_all, TEST_SAMPLES)\nprint(f\"\\nSelected {len(selected_test_uids)} Test UIDs (from actual test dir):\")\nfor uid in selected_test_uids:\n    print(uid)\n\n# -------------------------------\n# Step 4: Copy Train Image Folders\n# -------------------------------\nprint(\"\\nCopying train image folders (with segmentation support)...\")\nfor uid in tqdm(selected_train_uids, desc=\"Copying Train Images\"):\n    src_path = os.path.join(TRAIN_IMAGES_DIR, uid)\n    dest_path = os.path.join(PARTIAL_TRAIN_IMAGES_DIR, uid)\n    \n    if os.path.exists(src_path):\n        shutil.copytree(src_path, dest_path)\n    else:\n        print(f\"❌ [Train] UID {uid} not found in train_images.\")\nprint(\"✅ Train image folders copied successfully.\")\n\n# -------------------------------\n# Step 5: Copy Test Image Folders (No Segmentations)\n# -------------------------------\nif selected_test_uids:\n    print(\"\\nCopying test image folders (no segmentation needed)...\")\n    for uid in tqdm(selected_test_uids, desc=\"Copying Test Images\"):\n        src_path = os.path.join(TEST_IMAGES_DIR, uid)\n        dest_path = os.path.join(PARTIAL_TEST_IMAGES_DIR, uid)\n        \n        if os.path.exists(src_path):\n            shutil.copytree(src_path, dest_path)\n        else:\n            print(f\"❌ [Test] UID {uid} not found in test_images.\")\n    print(\"✅ Test image folders copied successfully.\")\nelse:\n    print(\"\\nNo test samples selected for test set.\")\n\n# -------------------------------\n# Step 6: Filter and Save Train Bounding Boxes\n# -------------------------------\nprint(\"\\nFiltering bounding boxes for selected train UIDs...\")\nfiltered_bboxes = bboxes_df[bboxes_df['StudyInstanceUID'].isin(selected_train_uids)]\nfiltered_bboxes.to_csv(PARTIAL_BBOXES_PATH, index=False)\nprint(\"✅ Filtered train_bounding_boxes.csv saved.\")\n\n# -------------------------------\n# Step 7: Copy Corresponding Segmentation Files (Train)\n# -------------------------------\nprint(\"\\nCopying segmentation files for selected train UIDs...\")\nfor uid in tqdm(selected_train_uids, desc=\"Copying Segmentations\"):\n    seg_file = f\"{uid}.nii\"\n    src_seg_path = os.path.join(SEGMENTATIONS_DIR, seg_file)\n    dest_seg_path = os.path.join(PARTIAL_SEGMENTATIONS_DIR, seg_file)\n    \n    if os.path.exists(src_seg_path):\n        shutil.copy(src_seg_path, dest_seg_path)\n    else:\n        print(f\"❌ No segmentation file found for UID {uid} (skipped).\")\nprint(\"✅ Train segmentation files copied where available.\")\n\n# -------------------------------\n# Step 8: Organize and Compress the Partial Dataset\n# -------------------------------\nprint(\"\\nOrganizing partial dataset...\")\nshutil.move(PARTIAL_TRAIN_IMAGES_DIR, os.path.join(DATASET_DIR, 'train_images_partial'))\n\nif selected_test_uids:\n    shutil.move(PARTIAL_TEST_IMAGES_DIR, os.path.join(DATASET_DIR, 'test_images_partial'))\n\nshutil.move(PARTIAL_BBOXES_PATH, os.path.join(DATASET_DIR, 'train_bounding_boxes_partial.csv'))\nshutil.move(PARTIAL_SEGMENTATIONS_DIR, os.path.join(DATASET_DIR, 'segmentations_partial'))\n\nprint(\"Compressing partial dataset...\")\nshutil.make_archive('RSNA2022_Spine_Dataset', 'zip', DATASET_DIR)\nos.rename('RSNA2022_Spine_Dataset.zip', ZIP_FILENAME)\n\nprint(f\"✅ Partial dataset compressed into '{ZIP_FILENAME}'\")\n\n# -------------------------------\n# Download Link\n# -------------------------------\nprint(\"\\nClick the link below to download your partial dataset:\")\ndisplay(FileLink(ZIP_FILENAME))\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-01-17T17:10:29.02177Z","iopub.execute_input":"2025-01-17T17:10:29.022289Z","iopub.status.idle":"2025-01-17T17:15:13.613248Z","shell.execute_reply.started":"2025-01-17T17:10:29.022256Z","shell.execute_reply":"2025-01-17T17:15:13.611528Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Quick EDA","metadata":{}},{"cell_type":"code","source":"# -------------------------------\n# Step 9: Exploratory Data Analysis (EDA) on Train Subset\n# -------------------------------\nprint(\"\\nStarting Exploratory Data Analysis (EDA) on the extracted train data...\")\n\n# Load the filtered bounding boxes\npartial_bboxes_df = pd.read_csv(os.path.join(DATASET_DIR, 'train_bounding_boxes_partial.csv'))\nprint(\"\\nFirst 5 rows of filtered train_bounding_boxes_partial.csv:\")\ndisplay(partial_bboxes_df.head())\n\n# Merge bounding boxes with train_df to get patient_overall\npatient_outcomes = train_df[['StudyInstanceUID','patient_overall']].drop_duplicates()\nmerged_df = partial_bboxes_df.merge(patient_outcomes, on='StudyInstanceUID', how='left')\n\n# Quick summary\nprint(\"\\nSummary Statistics of Bounding Boxes:\")\ndisplay(merged_df.describe())\n\n# -- Count bounding boxes by StudyInstanceUID --\nbbox_count = merged_df.groupby('StudyInstanceUID').size().reset_index(name='bbox_count')\n\nplt.figure(figsize=(10,5))\nsns.barplot(x='StudyInstanceUID', y='bbox_count', data=bbox_count, palette='viridis')\nplt.xticks(rotation=90)\nplt.title('Number of Bounding Boxes per StudyInstanceUID')\nplt.xlabel('StudyInstanceUID')\nplt.ylabel('Count of Boxes')\nplt.tight_layout()\nplt.show()\n\n# Distribution of slice_number (optional)\nplt.figure(figsize=(10,5))\nsns.histplot(merged_df['slice_number'], kde=False, bins=20, color='blue')\nplt.title('Distribution of Bounding Box Slice Numbers')\nplt.xlabel('slice_number')\nplt.ylabel('Count')\nplt.tight_layout()\nplt.show()\n\n# Patient-level outcome distribution\nplt.figure(figsize=(6,4))\nsns.countplot(\n    x='patient_overall',\n    data=merged_df.drop_duplicates(subset='StudyInstanceUID'),\n    palette='magma'\n)\nplt.title('Patient-level Outcome Distribution')\nplt.xlabel('patient_overall (0: No Fracture, 1: Fracture)')\nplt.ylabel('Number of Patients')\nplt.tight_layout()\nplt.show()\n\nprint(\"✅ EDA completed on train subset.\")\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-17T17:15:13.615691Z","iopub.execute_input":"2025-01-17T17:15:13.616146Z","iopub.status.idle":"2025-01-17T17:15:14.793897Z","shell.execute_reply.started":"2025-01-17T17:15:13.61611Z","shell.execute_reply":"2025-01-17T17:15:14.792652Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import random\nimport nibabel as nib\nimport imageio\nimport numpy as np\nimport cv2\nimport os\nimport pydicom\nimport matplotlib.pyplot as plt\n\n# -------------------------------\n# Where the segmentation is and what it's doing\n# -------------------------------\n# The segmentation files (*.nii) are stored in 'segmentations_partial/' directory for each\n# train StudyInstanceUID. Each file is a 3D volume typically of shape (X, Y, Z) or (Z, Y, X).\n# Each voxel's integer value indicates whether it's part of a structure (e.g. vertebra).\n# We threshold seg_data > 0 to create a binary mask (1 = structure, 0 = background).\n#\n# Because NIfTI orientation may differ from how your DICOM slices are enumerated,\n# we rotate or flip the segmentation volume so the slice indices align with the numeric\n# DICOM filenames.\n\ndef rotate_segmentation_3d(seg_data):\n    \"\"\"\n    Optionally rotate/transpose/flip the 3D seg_data volume\n    so that seg_data[:,:,z] aligns with your DICOM slice z-order.\n    \n    Adjust these calls if your segmentation seems misaligned.\n    \"\"\"\n    # Example: a single 90-degree rotation along the first two axes\n    seg_data = np.rot90(seg_data, k=1, axes=(0, 1))\n    \n    # Or flipping up-down\n    # seg_data = np.flipud(seg_data)\n    \n    # Or no rotation at all if you're certain it's correct\n    # seg_data = seg_data  # no-op\n    \n    return seg_data\n\ndef create_2d_3d_gif(uid, dataset_dir, outdir, overlay_color=(255, 0, 0)):\n    \"\"\"\n    Creates a GIF showing each DICOM slice overlaid with the segmentation mask (if available).\n    \n    Parameters:\n      uid (str): StudyInstanceUID\n      dataset_dir (str): Path to the partial dataset directory (where train_images_partial & segs exist)\n      outdir (str): Directory to save the resulting GIF\n      overlay_color (tuple): (R, G, B) color for the segmentation overlay\n    \n    Returns:\n      The filepath of the created GIF (or None if not possible).\n    \"\"\"\n    train_images_path = os.path.join(dataset_dir, 'train_images_partial', uid)\n    seg_path = os.path.join(dataset_dir, 'segmentations_partial', f\"{uid}.nii\")\n    \n    if not os.path.exists(train_images_path):\n        print(f\"❌ No train images found for UID {uid}. Cannot create GIF.\")\n        return None\n    \n    # Sort DICOM files by integer slice number in filename if numeric\n    dicom_files = sorted(\n        [f for f in os.listdir(train_images_path) if f.endswith('.dcm')],\n        key=lambda x: int(os.path.splitext(x)[0])\n    )\n\n    # Load segmentation if it exists\n    seg_data = None\n    if os.path.exists(seg_path):\n        nii = nib.load(seg_path)\n        seg_data = nii.get_fdata()  # shape might be (X, Y, Z) or (rows, cols, slices)\n        \n        # Try rotating/flipping if mask is misaligned\n        seg_data = rotate_segmentation_3d(seg_data)\n        \n        print(f\"✅ Loaded segmentation for UID {uid}, shape={seg_data.shape}\")\n    else:\n        print(f\"⚠ No segmentation file found for UID {uid}. Creating GIF without overlay.\")\n    \n    frames = []\n    for dfile in dicom_files:\n        dcm_path = os.path.join(train_images_path, dfile)\n        dcm = pydicom.dcmread(dcm_path)\n        \n        # Convert DICOM pixel data to [0..255] grayscale\n        image = dcm.pixel_array.astype(float)\n        image = (np.maximum(image, 0) / image.max()) * 255.0\n        image = image.astype(np.uint8)\n        \n        # Convert grayscale to 3 channels for overlay\n        image_3ch = cv2.cvtColor(image, cv2.COLOR_GRAY2BGR)\n        \n        # If seg_data is loaded, try to overlay\n        if seg_data is not None:\n            # Suppose the numeric portion of filename indicates slice_idx\n            slice_idx = int(os.path.splitext(dfile)[0])\n            \n            # Check if slice_idx is within seg_data's range\n            if slice_idx < seg_data.shape[2]:\n                seg_slice = seg_data[:, :, slice_idx]\n                \n                # Create a binary mask from seg_slice (everything > 0)\n                mask = (seg_slice > 0).astype(np.uint8)\n                \n                # If mismatch in shape, do naive resize\n                if mask.shape != (image_3ch.shape[0], image_3ch.shape[1]):\n                    mask = cv2.resize(mask, (image_3ch.shape[1], image_3ch.shape[0]), interpolation=cv2.INTER_NEAREST)\n                \n                # Create a color layer in overlay_color where mask=1\n                color_layer = np.zeros_like(image_3ch, dtype=np.uint8)\n                color_layer[mask == 1] = overlay_color  # e.g. red\n                \n                # Blend the overlay with partial transparency\n                alpha = 0.3  # overlay transparency\n                image_3ch = cv2.addWeighted(color_layer, alpha, image_3ch, 1 - alpha, 0)\n            else:\n                # The slice index from the DICOM file is beyond seg_data range\n                pass\n        \n        frames.append(image_3ch)\n    \n    if not frames:\n        print(f\"❌ No frames created for UID {uid}. Check your DICOM files.\")\n        return None\n    \n    # Create a GIF from the frames\n    gif_path = os.path.join(outdir, f\"{uid}_3d.gif\")\n    imageio.mimsave(gif_path, frames, fps=5)\n    print(f\"✅ Created GIF for UID {uid}: {gif_path}\")\n    return gif_path\n\n# Example usage: \n# After building your partial dataset and having 'selected_train_uids', create a few GIFs:\ngif_sample_uids = random.sample(selected_train_uids, min(2, len(selected_train_uids)))\nfor uid in gif_sample_uids:\n    create_2d_3d_gif(uid, dataset_dir=DATASET_DIR, outdir=WORKING_DIR)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-17T17:19:08.992339Z","iopub.execute_input":"2025-01-17T17:19:08.992863Z","iopub.status.idle":"2025-01-17T17:19:35.860617Z","shell.execute_reply.started":"2025-01-17T17:19:08.992825Z","shell.execute_reply":"2025-01-17T17:19:35.85929Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from IPython.display import Image, display\n\n# Paths to the GIF files\ngif1_path = \"/kaggle/working/1.2.826.0.1.3680043.19333_3d.gif\"\ngif2_path = \"/kaggle/working/1.2.826.0.1.3680043.27292_3d.gif\"\n\n# Display the GIFs\n# display(Image(filename=gif1_path))\ndisplay(Image(filename=gif2_path))\n","metadata":{"trusted":true,"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!apt-get install tree\n!tree -a --filelimit=20 /kaggle/working/","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-17T17:20:59.74708Z","iopub.execute_input":"2025-01-17T17:20:59.747551Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}