{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":99552,"databundleVersionId":13441085}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# --- Core Libraries ---\nimport os\nimport glob\nimport numpy as np\nimport pandas as pd\n\n# --- DICOM Handling ---\nimport pydicom\n\n# --- Visualization ---\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom tqdm.notebook import tqdm\n\n# --- Settings for better display ---\npd.set_option('display.max_columns', 50)\nsns.set_style('whitegrid')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-08-30T09:04:15.974695Z","iopub.execute_input":"2025-08-30T09:04:15.975441Z","iopub.status.idle":"2025-08-30T09:04:17.539223Z","shell.execute_reply.started":"2025-08-30T09:04:15.975405Z","shell.execute_reply":"2025-08-30T09:04:17.538439Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- Define the base path for the competition data ---\nBASE_PATH = '/kaggle/input/rsna-intracranial-aneurysm-detection/'\n\n# --- Define paths to specific files and directories ---\nTRAIN_CSV_PATH = os.path.join(BASE_PATH, 'train.csv')\nLOCALIZERS_CSV_PATH = os.path.join(BASE_PATH, 'train_localizers.csv')\nSERIES_DIR = os.path.join(BASE_PATH, 'series/')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-30T09:04:21.898715Z","iopub.execute_input":"2025-08-30T09:04:21.899289Z","iopub.status.idle":"2025-08-30T09:04:21.903285Z","shell.execute_reply.started":"2025-08-30T09:04:21.899266Z","shell.execute_reply":"2025-08-30T09:04:21.90261Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- Load the training labels and localizer data ---\ndf_train = pd.read_csv(TRAIN_CSV_PATH)\ndf_localizers = pd.read_csv(LOCALIZERS_CSV_PATH)\n\n# --- Display the first few rows of the training dataframe ---\nprint(\"Training Data Head:\")\ndisplay(df_train.head())\n\n# --- Display basic info about the training dataframe ---\nprint(\"\\nTraining Data Info:\")\ndf_train.info()\n\n# --- Display the first few rows of the localizers dataframe ---\nprint(\"\\nLocalizers Data Head:\")\ndisplay(df_localizers.head())\n\nprint(\"\\nLocalizers Data Info:\")\ndf_localizers.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-30T09:05:44.744228Z","iopub.execute_input":"2025-08-30T09:05:44.744507Z","iopub.status.idle":"2025-08-30T09:05:44.79452Z","shell.execute_reply.started":"2025-08-30T09:05:44.744487Z","shell.execute_reply":"2025-08-30T09:05:44.793839Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_dicom_series(series_uid, base_path):\n    \"\"\"\n    Loads a DICOM series and returns it as a 3D NumPy array.\n    \n    Args:\n        series_uid (str): The SeriesInstanceUID of the scan to load.\n        base_path (str): The base directory for the series data.\n        \n    Returns:\n        np.ndarray: A 3D array representing the medical scan.\n    \"\"\"\n    # Find all DICOM files for the given series UID\n    series_path = os.path.join(base_path, series_uid, '*.dcm')\n    dicom_files = glob.glob(series_path)\n    \n    # Read all DICOM files and store them with their instance number\n    slices = []\n    for file_path in dicom_files:\n        dicom = pydicom.dcmread(file_path)\n        # The 'InstanceNumber' tag is crucial for ordering the slices correctly\n        slices.append((int(dicom.InstanceNumber), dicom))\n    \n    # Sort the slices based on their instance number\n    slices.sort(key=lambda x: x[0])\n    \n    # Stack the pixel data of the sorted slices to form a 3D volume\n    volume_3d = np.stack([s[1].pixel_array for s in slices])\n    \n    return volume_3d","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-30T09:15:48.916523Z","iopub.execute_input":"2025-08-30T09:15:48.91725Z","iopub.status.idle":"2025-08-30T09:15:48.92237Z","shell.execute_reply.started":"2025-08-30T09:15:48.917219Z","shell.execute_reply":"2025-08-30T09:15:48.921641Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Select the 13 location columns\nlocation_columns = df_train.columns[4:17]\n\n# For each row, check if the sum of location labels is greater than 0\n# This creates a boolean Series (True if aneurysm exists, False otherwise)\nderived_aneurysm_present = (df_train[location_columns].sum(axis=1) > 0).astype(int)\n\n# Compare our derived result with the actual 'Aneurysm Present' column\n# The .all() method will return True only if every single row matches.\nis_consistent = (derived_aneurysm_present == df_train['Aneurysm Present']).all()\n\nif is_consistent:\n    print(\"✅ Sanity Check Passed: 'Aneurysm Present' is consistent with the 13 location labels.\")\nelse:\n    print(\"❌ Sanity Check Failed: There is a discrepancy between the main target and location labels.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-30T09:37:23.511276Z","iopub.execute_input":"2025-08-30T09:37:23.511823Z","iopub.status.idle":"2025-08-30T09:37:23.519389Z","shell.execute_reply.started":"2025-08-30T09:37:23.511797Z","shell.execute_reply":"2025-08-30T09:37:23.518672Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(8, 6))\nsns.countplot(x='Aneurysm Present', data=df_train, palette='viridis')\nplt.title('Overall Aneurysm Presence Distribution', fontsize=16)\nplt.xlabel('Aneurysm Present (0 = No, 1 = Yes)', fontsize=12)\nplt.ylabel('Number of Scans', fontsize=12)\n\n# Add annotations\nax = plt.gca()\nfor p in ax.patches:\n    ax.annotate(f'{p.get_height()}', (p.get_x() + p.get_width() / 2., p.get_height()),\n                ha='center', va='center', xytext=(0, 10), textcoords='offset points', fontsize=12)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-30T09:40:12.929679Z","iopub.execute_input":"2025-08-30T09:40:12.930176Z","iopub.status.idle":"2025-08-30T09:40:13.07628Z","shell.execute_reply.started":"2025-08-30T09:40:12.93015Z","shell.execute_reply":"2025-08-30T09:40:13.075591Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Select the 13 location columns\nlocation_columns = df_train.columns[4:17]\n\n# Calculate the sum of positive cases for each location\nlocation_counts = df_train[location_columns].sum().sort_values(ascending=False)\n\nplt.figure(figsize=(12, 8))\nsns.barplot(x=location_counts.values, y=location_counts.index, orient='h', palette='crest')\nplt.title('Distribution of Aneurysms by Location', fontsize=16)\nplt.xlabel('Number of Positive Cases', fontsize=12)\nplt.ylabel('Aneurysm Location', fontsize=12)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-30T09:40:30.800555Z","iopub.execute_input":"2025-08-30T09:40:30.800815Z","iopub.status.idle":"2025-08-30T09:40:31.110625Z","shell.execute_reply.started":"2025-08-30T09:40:30.800796Z","shell.execute_reply":"2025-08-30T09:40:31.109826Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Select a sample series ID that has a known aneurysm\nsample_uid = df_localizers['SeriesInstanceUID'].iloc[0]\n\nprint(f\"Loading and visualizing scan for Series UID: {sample_uid}\")\n\n# Load the 3D volume\nbrain_scan_3d = load_dicom_series(sample_uid, SERIES_DIR)\n\nprint(f\"Scan loaded. Shape: {brain_scan_3d.shape}\")\n\n# --- Visualize three slices from the volume (e.g., start, middle, end) ---\nnum_slices = brain_scan_3d.shape[0]\nslice_indices = [0, num_slices // 2, num_slices - 1]\n\nfig, axes = plt.subplots(1, 3, figsize=(18, 6))\n\nfor i, slice_idx in enumerate(slice_indices):\n    ax = axes[i]\n    ax.imshow(brain_scan_3d[slice_idx], cmap='bone')\n    ax.set_title(f'Slice {slice_idx + 1}/{num_slices}')\n    ax.axis('off')\n\nplt.suptitle(f'Sample Slices for Series: {sample_uid}', fontsize=16)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-30T09:16:40.335151Z","iopub.execute_input":"2025-08-30T09:16:40.335746Z","iopub.status.idle":"2025-08-30T09:16:45.855354Z","shell.execute_reply.started":"2025-08-30T09:16:40.335721Z","shell.execute_reply":"2025-08-30T09:16:45.854483Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# We'll use the 'brain_scan_3d' volume we loaded earlier for the sample_uid\n# Its shape corresponds to (z, y, x)\n\nprint(f\"Original 3D volume shape: {brain_scan_3d.shape} -> (Z, Y, X)\")\n\n# --- Calculate the three MIPs ---\n\n# Axial view (Top-down): Project along the Z-axis (axis 0)\nmip_axial = np.max(brain_scan_3d, axis=0)\n\n# Coronal view (Front-back): Project along the Y-axis (axis 1)\nmip_coronal = np.max(brain_scan_3d, axis=1)\n\n# Sagittal view (Side): Project along the X-axis (axis 2)\nmip_sagittal = np.max(brain_scan_3d, axis=2)\n\n\n# --- Visualize the three projections ---\nfig, axes = plt.subplots(1, 3, figsize=(18, 8))\n\n# Axial Plot\naxes[0].imshow(mip_axial, cmap='bone')\naxes[0].set_title('Axial View (Top-down)', fontsize=14)\naxes[0].axis('off')\n\n# Coronal Plot\naxes[1].imshow(mip_coronal, cmap='bone')\naxes[1].set_title('Coronal View (Front-back)', fontsize=14)\naxes[1].axis('off')\n\n# Sagittal Plot\naxes[2].imshow(mip_sagittal, cmap='bone')\naxes[2].set_title('Sagittal View (Side)', fontsize=14)\naxes[2].axis('off')\n\nplt.suptitle(f'Maximum Intensity Projections for Scan: {sample_uid}', fontsize=18)\nplt.tight_layout(rect=[0, 0, 1, 0.95]) # Adjust layout to make room for suptitle\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-30T09:49:41.582502Z","iopub.execute_input":"2025-08-30T09:49:41.58279Z","iopub.status.idle":"2025-08-30T09:49:42.421765Z","shell.execute_reply.started":"2025-08-30T09:49:41.582771Z","shell.execute_reply":"2025-08-30T09:49:42.420988Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- Get the specific localization info for our sample scan ---\nsample_uid = '1.2.826.0.1.3680043.8.498.10076056930521523789588901704956188485'\naneurysm_details = df_localizers[df_localizers['SeriesInstanceUID'] == sample_uid].iloc[0]\n\n# --- Get the specific slice identifier and the coordinates string ---\nslice_sop_uid = aneurysm_details['SOPInstanceUID']\ncoords_str = aneurysm_details['coordinates']\n\n# --- Parse the coordinate string ---\nimport ast\nparsed_coords = ast.literal_eval(coords_str)\n\n# --- Robustly handle the coordinate format ---\ncoords_dict = None\nif isinstance(parsed_coords, list) and parsed_coords:\n    # If it's a list of dictionaries, take the first one\n    coords_dict = parsed_coords[0]\nelif isinstance(parsed_coords, dict):\n    # If it's already a dictionary, use it directly\n    coords_dict = parsed_coords\n\n# --- Proceed if we successfully extracted a coordinate dictionary ---\nif coords_dict and 'x' in coords_dict and 'y' in coords_dict:\n    x_coord = coords_dict['x']\n    y_coord = coords_dict['y']\n\n    # --- Construct the full path to the specific DICOM slice file ---\n    slice_path = os.path.join(SERIES_DIR, sample_uid, f'{slice_sop_uid}.dcm')\n\n    # --- Read the single DICOM file for that slice ---\n    dicom_slice = pydicom.dcmread(slice_path)\n    slice_pixels = dicom_slice.pixel_array\n\n    # --- Visualize the slice and highlight the aneurysm ---\n    fig, ax = plt.subplots(1, 1, figsize=(10, 10))\n    ax.imshow(slice_pixels, cmap='bone')\n    \n    # Draw a red circle to mark the aneurysm's location\n    highlight_circle = plt.Circle((x_coord, y_coord), radius=20, color='red', fill=False, linewidth=2)\n    ax.add_patch(highlight_circle)\n\n    ax.set_title(f'Aneurysm Location on Slice: {slice_sop_uid}')\n    ax.axis('off')\n    plt.show()\n\nelse:\n    print(f\"Could not parse valid x, y coordinates from: {coords_str}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-30T10:07:39.854478Z","iopub.execute_input":"2025-08-30T10:07:39.855099Z","iopub.status.idle":"2025-08-30T10:07:40.323614Z","shell.execute_reply.started":"2025-08-30T10:07:39.855074Z","shell.execute_reply":"2025-08-30T10:07:40.322869Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import SimpleITK as sitk\nimport nibabel as nib\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport pydicom\nimport ast\n\n# --- Define Paths (using the same UID from the example) ---\nuid = '1.2.826.0.1.3680043.8.498.10076056930521523789588901704956188485'\ndicom_series_path = f\"/kaggle/input/rsna-intracranial-aneurysm-detection/series/{uid}\"\nnii_seg_path = f\"/kaggle/input/rsna-intracranial-aneurysm-detection/segmentations/{uid}_cowseg.nii\"\n\n# --- Step 1: Load Both Volumes with Full Spatial Metadata ---\nreader = sitk.ImageSeriesReader()\ndicom_names = reader.GetGDCMSeriesFileNames(dicom_series_path)\nreader.SetFileNames(dicom_names)\ndicom_image_sitk = reader.Execute()\nseg_image_sitk = sitk.ReadImage(nii_seg_path)\n\n# --- Step 2: Resample the Segmentation to Match the DICOM Space ---\nresampler = sitk.ResampleImageFilter()\nresampler.SetReferenceImage(dicom_image_sitk)\nresampler.SetInterpolator(sitk.sitkNearestNeighbor)\nresampled_seg_sitk = resampler.Execute(seg_image_sitk)\n\n# --- Step 3: Convert Aligned Volumes to NumPy Arrays ---\ndicom_vol = sitk.GetArrayFromImage(dicom_image_sitk)\nresampled_seg_vol = sitk.GetArrayFromImage(resampled_seg_sitk)\n\n# --- Step 4: Create the MIP and Visualize the Overlay ---\nmip = np.max(dicom_vol, axis=0)\ny_coords, x_coords = np.where(resampled_seg_vol.sum(axis=0) > 0)\n\nplt.figure(figsize=(10, 10))\nplt.imshow(mip, cmap=\"gray\")\nplt.scatter(x_coords, y_coords, s=1, c='cyan', alpha=0.2, label=\"Aligned Segmentation\")\n\n# --- THIS IS THE CORRECTED COORDINATE PARSING LOGIC ---\ncoords_str = df_localizers[df_localizers[\"SeriesInstanceUID\"] == uid].coordinates.iloc[0]\nparsed_obj = ast.literal_eval(coords_str)\n\ncoords_dict = None\nif isinstance(parsed_obj, list) and parsed_obj:\n    coords_dict = parsed_obj[0]  # Handles format: [{'x': 1, 'y': 2}]\nelif isinstance(parsed_obj, dict):\n    coords_dict = parsed_obj      # Handles format: {'x': 1, 'y': 2}\n\nif coords_dict:\n    lx, ly = int(coords_dict['x']), int(coords_dict['y'])\n    plt.scatter([lx], [ly], c=\"red\", s=40, label=\"Ground Truth Aneurysm\", marker=\"X\")\n# ---------------------------------------------------------\n\nplt.title(\"Correctly Aligned Segmentation on DICOM MIP\", fontsize=16)\nplt.axis(\"off\")\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-30T09:57:23.79751Z","iopub.execute_input":"2025-08-30T09:57:23.797761Z","iopub.status.idle":"2025-08-30T09:57:49.380555Z","shell.execute_reply.started":"2025-08-30T09:57:23.797744Z","shell.execute_reply":"2025-08-30T09:57:49.37986Z"}},"outputs":[],"execution_count":null}]}