{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":99552,"databundleVersionId":13851420}],"dockerImageVersionId":31328,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\n\n# Confirm dataset path\ndataset_path = \"/kaggle/input/competitions/rsna-intracranial-aneurysm-detection\"\n\nfor item in sorted(os.listdir(dataset_path))[:20]:\n    print(item)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T13:33:32.176306Z","iopub.execute_input":"2026-05-06T13:33:32.176782Z","iopub.status.idle":"2026-05-06T13:33:32.188562Z","shell.execute_reply.started":"2026-05-06T13:33:32.176745Z","shell.execute_reply":"2026-05-06T13:33:32.187385Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Check what's inside series/\nseries_path = \"/kaggle/input/competitions/rsna-intracranial-aneurysm-detection/series\"\nprint(\"=== series/ folder ===\")\nfor item in sorted(os.listdir(series_path))[:10]:\n    print(item)\n\n# Check one patient's folder\nfirst_patient = sorted(os.listdir(series_path))[0]\npatient_path = os.path.join(series_path, first_patient)\nprint(f\"\\n=== Inside first patient: {first_patient} ===\")\nfor item in sorted(os.listdir(patient_path))[:10]:\n    print(item)\n\n# Load labels\ntrain_df = pd.read_csv(\"/kaggle/input/competitions/rsna-intracranial-aneurysm-detection/train.csv\")\nprint(f\"\\n=== train.csv ===\")\nprint(f\"Shape: {train_df.shape}\")\nprint(train_df.head())\nprint(f\"\\nColumns: {list(train_df.columns)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T13:33:32.190466Z","iopub.execute_input":"2026-05-06T13:33:32.190859Z","iopub.status.idle":"2026-05-06T13:33:33.394934Z","shell.execute_reply.started":"2026-05-06T13:33:32.190826Z","shell.execute_reply":"2026-05-06T13:33:33.393706Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check segmentations folder\nseg_path = \"/kaggle/input/competitions/rsna-intracranial-aneurysm-detection/segmentations\"\nseg_files = sorted(os.listdir(seg_path))\nprint(f\"=== segmentations/ ===\")\nprint(f\"Total files: {len(seg_files)}\")\nprint(\"First 5:\", seg_files[:5])\n\n# Filter CTA only (as per Phase 2 report)\ncta_df = train_df[train_df['Modality'] == 'CTA'].copy()\nprint(f\"\\n=== Modality breakdown ===\")\nprint(train_df['Modality'].value_counts())\nprint(f\"\\nCTA only: {len(cta_df)} series\")\nprint(f\"CTA with aneurysm: {cta_df['Aneurysm Present'].sum()}\")\nprint(f\"CTA without aneurysm: {(cta_df['Aneurysm Present'] == 0).sum()}\")\n\n# Check if segmentation files match SeriesInstanceUIDs\nprint(f\"\\n=== Checking one segmentation file ===\")\nprint(seg_files[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T13:33:33.396703Z","iopub.execute_input":"2026-05-06T13:33:33.397157Z","iopub.status.idle":"2026-05-06T13:33:33.570206Z","shell.execute_reply.started":"2026-05-06T13:33:33.397114Z","shell.execute_reply":"2026-05-06T13:33:33.569042Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import nibabel as nib\nimport numpy as np\n\n# Find which CTA series have segmentation masks\nseg_ids = set()\nfor f in seg_files:\n    uid = f.replace('_cowseg.nii', '').replace('.nii', '')\n    seg_ids.add(uid)\n\nprint(f\"Series with segmentation masks: {len(seg_ids)}\")\n\n# How many of these are CTA?\ncta_with_seg = cta_df[cta_df['SeriesInstanceUID'].isin(seg_ids)]\ncta_without_seg = cta_df[~cta_df['SeriesInstanceUID'].isin(seg_ids)]\nprint(f\"CTA series WITH pixel masks : {len(cta_with_seg)}\")\nprint(f\"CTA series WITHOUT pixel masks: {len(cta_without_seg)}\")\nprint(f\"  - of which positive (label only): {cta_without_seg['Aneurysm Present'].sum()}\")\nprint(f\"  - of which negative: {(cta_without_seg['Aneurysm Present']==0).sum()}\")\n\n# Load one segmentation to understand its structure\nsample_uid = cta_with_seg['SeriesInstanceUID'].values[0]\nseg_file = os.path.join(seg_path, f\"{sample_uid}.nii\")\nprint(f\"\\n=== Sample segmentation: {sample_uid[:30]}... ===\")\n\nnii = nib.load(seg_file)\nmask_data = nii.get_fdata()\nprint(f\"Shape       : {mask_data.shape}\")\nprint(f\"Unique vals : {np.unique(mask_data)}\")\nprint(f\"Positive px : {(mask_data > 0).sum()}\")\nprint(f\"Voxel spacing: {nii.header.get_zooms()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T13:33:33.571485Z","iopub.execute_input":"2026-05-06T13:33:33.571863Z","iopub.status.idle":"2026-05-06T13:33:41.018608Z","shell.execute_reply.started":"2026-05-06T13:33:33.571834Z","shell.execute_reply":"2026-05-06T13:33:41.017463Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load the cowseg file for the same patient\ncowseg_file = os.path.join(seg_path, f\"{sample_uid}_cowseg.nii\")\ncow = nib.load(cowseg_file)\ncow_data = cow.get_fdata()\n\nprint(f\"=== _cowseg.nii ===\")\nprint(f\"Shape       : {cow_data.shape}\")\nprint(f\"Unique vals : {np.unique(cow_data)}\")\nprint(f\"Positive px : {(cow_data > 0).sum()}\")\nprint(f\"Voxel spacing: {cow.header.get_zooms()}\")\n\n# Also check the plain .nii again more carefully\nprint(f\"\\n=== plain .nii (revisit) ===\")\nnii2 = nib.load(seg_file)\nvol = nii2.get_fdata()\nprint(f\"Shape       : {vol.shape}\")\nprint(f\"Min / Max   : {vol.min():.1f} / {vol.max():.1f}\")\nprint(f\"Is it HU?   : {vol.min() < -500}\")  # HU air is around -1000\n\n# Check what the label says for this patient\nprint(f\"\\n=== Label for this patient ===\")\nrow = cta_df[cta_df['SeriesInstanceUID'] == sample_uid]\nprint(row[['SeriesInstanceUID', 'Aneurysm Present'] + \n          [c for c in cta_df.columns if c not in \n           ['SeriesInstanceUID','PatientAge','PatientSex','Modality']]].to_string())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T13:33:41.02117Z","iopub.execute_input":"2026-05-06T13:33:41.021594Z","iopub.status.idle":"2026-05-06T13:33:43.933321Z","shell.execute_reply.started":"2026-05-06T13:33:41.021563Z","shell.execute_reply":"2026-05-06T13:33:43.932185Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# The cowseg has values 0-13, matching the 13 anatomical location columns\n# Let's map them\n\nlocation_columns = [\n    'Left Infraclinoid Internal Carotid Artery',    # likely label 1\n    'Right Infraclinoid Internal Carotid Artery',   # likely label 2\n    'Left Supraclinoid Internal Carotid Artery',    # likely label 3\n    'Right Supraclinoid Internal Carotid Artery',   # likely label 4\n    'Left Middle Cerebral Artery',                  # likely label 5\n    'Right Middle Cerebral Artery',                 # likely label 6\n    'Anterior Communicating Artery',                # likely label 7\n    'Left Anterior Cerebral Artery',                # likely label 8\n    'Right Anterior Cerebral Artery',               # likely label 9\n    'Left Posterior Communicating Artery',          # likely label 10\n    'Right Posterior Communicating Artery',         # likely label 11\n    'Basilar Tip',                                  # likely label 12\n    'Other Posterior Circulation',                  # likely label 13\n]\n\nprint(\"=== cowseg unique values and pixel counts ===\")\nfor val in np.unique(cow_data):\n    count = (cow_data == val).sum()\n    if val == 0:\n        label = \"background\"\n    elif int(val) <= len(location_columns):\n        label = location_columns[int(val)-1]\n    else:\n        label = \"unknown\"\n    print(f\"  Value {int(val):2d} → {count:8,} px → {label}\")\n\nprint(f\"\\n=== This patient's positive locations ===\")\nfor col in location_columns:\n    val = row[col].values[0]\n    if val == 1:\n        idx = location_columns.index(col) + 1\n        pixel_count = (cow_data == idx).sum()\n        print(f\"  {col} (label={idx}) → {pixel_count} pixels in mask\")\n\n# Create binary aneurysm mask from positive locations only\naneurysm_labels = []\nfor col in location_columns:\n    if row[col].values[0] == 1:\n        aneurysm_labels.append(location_columns.index(col) + 1)\n\nbinary_mask = np.isin(cow_data, aneurysm_labels).astype(np.uint8)\nprint(f\"\\n=== Binary aneurysm mask ===\")\nprint(f\"Aneurysm labels used : {aneurysm_labels}\")\nprint(f\"Positive voxels      : {binary_mask.sum()}\")\nprint(f\"Slices with aneurysm : {binary_mask.any(axis=(0,1)).sum()} out of {binary_mask.shape[2]}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T13:33:43.934343Z","iopub.execute_input":"2026-05-06T13:33:43.934607Z","iopub.status.idle":"2026-05-06T13:33:47.783343Z","shell.execute_reply.started":"2026-05-06T13:33:43.934581Z","shell.execute_reply":"2026-05-06T13:33:47.782422Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import subprocess\nsubprocess.run(['pip', 'install', '-q', 'SimpleITK', 'scikit-image', 'tqdm'], \n               capture_output=True)\n\nimport os, json, random, gc, warnings\nfrom pathlib import Path\nfrom dataclasses import dataclass, asdict, field\nfrom typing import List, Tuple, Dict, Optional\n\nimport numpy as np\nimport pandas as pd\nimport SimpleITK as sitk\nimport nibabel as nib\nfrom scipy.ndimage import zoom\nfrom sklearn.model_selection import KFold\nfrom tqdm.notebook import tqdm\nimport matplotlib.pyplot as plt\n\nwarnings.filterwarnings('ignore')\nsitk.ProcessObject_SetGlobalWarningDisplay(False)\n\n# Fixed seeds everywhere\nSEED = 42\nrandom.seed(SEED)\nnp.random.seed(SEED)\nos.environ['PYTHONHASHSEED'] = str(SEED)\n\n# ── Paths ────────────────────────────────────────────────────────────────────\nDATASET_ROOT = \"/kaggle/input/competitions/rsna-intracranial-aneurysm-detection\"\nSERIES_DIR   = f\"{DATASET_ROOT}/series\"\nSEG_DIR      = f\"{DATASET_ROOT}/segmentations\"\nTRAIN_CSV    = f\"{DATASET_ROOT}/train.csv\"\nOUTPUT_DIR   = \"/kaggle/working/processed\"\n\n# ── Anatomical location label mapping (train.csv column → cowseg value) ─────\nLOCATION_COLUMNS = [\n    'Left Infraclinoid Internal Carotid Artery',     # cowseg = 1\n    'Right Infraclinoid Internal Carotid Artery',    # cowseg = 2\n    'Left Supraclinoid Internal Carotid Artery',     # cowseg = 3\n    'Right Supraclinoid Internal Carotid Artery',    # cowseg = 4\n    'Left Middle Cerebral Artery',                   # cowseg = 5\n    'Right Middle Cerebral Artery',                  # cowseg = 6\n    'Anterior Communicating Artery',                 # cowseg = 7\n    'Left Anterior Cerebral Artery',                 # cowseg = 8\n    'Right Anterior Cerebral Artery',                # cowseg = 9\n    'Left Posterior Communicating Artery',           # cowseg = 10\n    'Right Posterior Communicating Artery',          # cowseg = 11\n    'Basilar Tip',                                   # cowseg = 12\n    'Other Posterior Circulation',                   # cowseg = 13\n]\n\n@dataclass\nclass Config:\n    # Paths\n    series_dir  : str = SERIES_DIR\n    seg_dir     : str = SEG_DIR\n    output_dir  : str = OUTPUT_DIR\n\n    # Resampling\n    spacing_xy  : float = 0.5   # mm/px in-plane\n    spacing_z   : float = 1.0   # mm/slice through-plane\n\n    # Windowing\n    hu_min      : int = 0\n    hu_max      : int = 700\n\n    # Crop\n    crop_size   : int = 256\n\n    # 2.5-D\n    n_channels  : int = 3       # [Z-1, Z, Z+1]\n\n    # Sampling\n    neg_pos_ratio : int = 3\n\n    # CV\n    n_folds     : int = 5\n    seed        : int = 42\n\n    def __post_init__(self):\n        for d in ['tensors', 'masks', 'metadata', 'viz']:\n            Path(self.output_dir, d).mkdir(parents=True, exist_ok=True)\n\nCFG = Config()\n\n# ── Load & filter CTA only ───────────────────────────────────────────────────\ntrain_df = pd.read_csv(TRAIN_CSV)\ncta_df   = train_df[train_df['Modality'] == 'CTA'].copy().reset_index(drop=True)\n\n# Find which series have pixel-level segmentation masks\nseg_ids = set()\nfor f in os.listdir(SEG_DIR):\n    uid = f.replace('_cowseg.nii', '').replace('.nii', '')\n    seg_ids.add(uid)\n\ncta_df['has_seg_mask'] = cta_df['SeriesInstanceUID'].isin(seg_ids)\n\nprint(\"✅ Config ready\")\nprint(f\"   Total CTA series         : {len(cta_df)}\")\nprint(f\"   CTA with pixel masks     : {cta_df['has_seg_mask'].sum()}\")\nprint(f\"   CTA positive (label only): {((cta_df['Aneurysm Present']==1) & ~cta_df['has_seg_mask']).sum()}\")\nprint(f\"   CTA negative             : {(cta_df['Aneurysm Present']==0).sum()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T13:33:47.784426Z","iopub.execute_input":"2026-05-06T13:33:47.785368Z","iopub.status.idle":"2026-05-06T13:33:54.483166Z","shell.execute_reply.started":"2026-05-06T13:33:47.785315Z","shell.execute_reply":"2026-05-06T13:33:54.482272Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ══════════════════════════════════════════════════════════════════\n# FUNCTION 1: Load DICOM volume → HU array\n# ══════════════════════════════════════════════════════════════════\ndef load_dicom_volume(series_dir: str) -> Tuple[np.ndarray, Tuple]:\n    reader = sitk.ImageSeriesReader()\n    dicom_names = reader.GetGDCMSeriesFileNames(series_dir)\n    reader.SetFileNames(dicom_names)\n    sitk_img = reader.Execute()\n    \n    spacing_xyz = sitk_img.GetSpacing()          # (x, y, z)\n    spacing_zyx = (spacing_xyz[2], spacing_xyz[1], spacing_xyz[0])\n    volume = sitk.GetArrayFromImage(sitk_img).astype(np.float32)  # (Z, H, W)\n    return volume, spacing_zyx\n\n\n# ══════════════════════════════════════════════════════════════════\n# FUNCTION 2: Load NIfTI segmentation mask → binary aneurysm mask\n# ══════════════════════════════════════════════════════════════════\ndef load_aneurysm_mask(series_uid: str, row: pd.Series) -> Optional[np.ndarray]:\n    \"\"\"\n    Load _cowseg.nii and extract binary aneurysm mask\n    using the positive location labels from train.csv.\n    Returns None if no segmentation file exists.\n    \"\"\"\n    cowseg_path = os.path.join(SEG_DIR, f\"{series_uid}_cowseg.nii\")\n    if not os.path.exists(cowseg_path):\n        return None\n    \n    cow_data = nib.load(cowseg_path).get_fdata()\n    \n    # NIfTI is stored (H, W, Z) → transpose to (Z, H, W) to match DICOM\n    cow_data = np.transpose(cow_data, (2, 1, 0))\n    \n    # Find which labels are positive for this patient\n    aneurysm_labels = []\n    for i, col in enumerate(LOCATION_COLUMNS):\n        if row[col] == 1:\n            aneurysm_labels.append(i + 1)  # cowseg label = column index + 1\n    \n    if not aneurysm_labels:\n        return np.zeros(cow_data.shape, dtype=np.uint8)\n    \n    binary_mask = np.isin(cow_data, aneurysm_labels).astype(np.uint8)\n    return binary_mask\n\n\n# ══════════════════════════════════════════════════════════════════\n# FUNCTION 3: Resample volume to target spacing\n# ══════════════════════════════════════════════════════════════════\ndef resample_volume(volume: np.ndarray,\n                    current_spacing: Tuple,\n                    target_spacing: Tuple,\n                    order: int = 1) -> Tuple[np.ndarray, Tuple]:\n    zoom_factors = [current_spacing[i] / target_spacing[i] for i in range(3)]\n    resampled = zoom(volume, zoom_factors, order=order)\n    return resampled.astype(np.float32), target_spacing\n\n\ndef resample_mask(mask: np.ndarray,\n                  current_spacing: Tuple,\n                  target_spacing: Tuple) -> np.ndarray:\n    zoom_factors = [current_spacing[i] / target_spacing[i] for i in range(3)]\n    resampled = zoom(mask.astype(np.float32), zoom_factors, order=0)  # nearest neighbour\n    return (resampled > 0.5).astype(np.uint8)\n\n\n# ══════════════════════════════════════════════════════════════════\n# FUNCTION 4: HU windowing + normalisation\n# ══════════════════════════════════════════════════════════════════\ndef window_normalise(volume: np.ndarray,\n                     hu_min: int = 0,\n                     hu_max: int = 700) -> np.ndarray:\n    clipped = np.clip(volume, hu_min, hu_max)\n    return ((clipped - hu_min) / (hu_max - hu_min)).astype(np.float32)\n\n\n# ══════════════════════════════════════════════════════════════════\n# FUNCTION 5: Centre-crop single slice\n# ══════════════════════════════════════════════════════════════════\ndef centre_crop(arr: np.ndarray, size: int = 256) -> np.ndarray:\n    h, w = arr.shape[-2], arr.shape[-1]\n    \n    # Pad if smaller than crop size\n    if h < size or w < size:\n        ph = max(0, size - h)\n        pw = max(0, size - w)\n        if arr.ndim == 2:\n            arr = np.pad(arr, ((ph//2, ph-ph//2), (pw//2, pw-pw//2)), \n                        mode='constant')\n        else:\n            arr = np.pad(arr, ((0,0),(ph//2, ph-ph//2),(pw//2, pw-pw//2)), \n                        mode='constant')\n        h, w = arr.shape[-2], arr.shape[-1]\n    \n    sh = (h - size) // 2\n    sw = (w - size) // 2\n    if arr.ndim == 2:\n        return arr[sh:sh+size, sw:sw+size]\n    return arr[:, sh:sh+size, sw:sw+size]\n\n\n# ══════════════════════════════════════════════════════════════════\n# FUNCTION 6: Extract 2.5-D tensor [Z-1, Z, Z+1]\n# ══════════════════════════════════════════════════════════════════\ndef extract_25d(volume: np.ndarray, z: int, n_ch: int = 3) -> np.ndarray:\n    Z = volume.shape[0]\n    half = n_ch // 2\n    channels = []\n    for dz in range(-half, half + 1):\n        zi = z + dz\n        if 0 <= zi < Z:\n            channels.append(volume[zi])\n        else:\n            channels.append(np.zeros(volume.shape[1:], dtype=np.float32))  # zero-pad boundary\n    return np.stack(channels, axis=0)  # (n_ch, H, W)\n\n\n# ══════════════════════════════════════════════════════════════════\n# FUNCTION 7: Slice selection with negative sampling\n# ══════════════════════════════════════════════════════════════════\ndef select_slices(mask_volume: np.ndarray,\n                  ratio: int = 3,\n                  seed: int = 42) -> Tuple[List[int], List[int]]:\n    Z = mask_volume.shape[0]\n    pos_idx = [z for z in range(Z) if mask_volume[z].any()]\n    neg_idx = [z for z in range(Z) if not mask_volume[z].any()]\n    \n    n_neg = min(len(pos_idx) * ratio, len(neg_idx))\n    rng = np.random.RandomState(seed)\n    sampled_neg = rng.choice(neg_idx, size=n_neg, replace=False).tolist()\n    \n    return pos_idx, sorted(sampled_neg)\n\n\nprint(\"✅ All preprocessing functions defined\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T13:33:54.484304Z","iopub.execute_input":"2026-05-06T13:33:54.484897Z","iopub.status.idle":"2026-05-06T13:33:54.507915Z","shell.execute_reply.started":"2026-05-06T13:33:54.484868Z","shell.execute_reply":"2026-05-06T13:33:54.507066Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def process_patient(row: pd.Series, cfg: Config, epoch_seed: int = 42) -> Dict:\n    \"\"\"\n    Full Stage 1 pipeline for a single CTA series.\n    Returns a stats dict and saves tensors + masks to disk.\n    \"\"\"\n    uid = row['SeriesInstanceUID']\n    series_path = os.path.join(cfg.series_dir, uid)\n    \n    stats = {\n        'uid'              : uid,\n        'status'           : 'ok',\n        'has_seg_mask'     : bool(row['has_seg_mask']),\n        'aneurysm_present' : int(row['Aneurysm Present']),\n        'original_shape'   : None,\n        'resampled_shape'  : None,\n        'n_pos_slices'     : 0,\n        'n_neg_slices'     : 0,\n        'error'            : None,\n    }\n    \n    try:\n        # ── Step 1: Load DICOM ───────────────────────────────────────────\n        volume, spacing = load_dicom_volume(series_path)\n        stats['original_shape'] = list(volume.shape)\n        stats['original_spacing'] = list(spacing)\n        \n        target_spacing = (cfg.spacing_z, cfg.spacing_xy, cfg.spacing_xy)\n        \n        # ── Step 2: Load segmentation mask ──────────────────────────────\n        if row['has_seg_mask']:\n            mask_volume = load_aneurysm_mask(uid, row)\n            \n            # Mask may have different shape than volume — resample mask\n            # to match volume shape first, then resample together\n            if mask_volume is not None and mask_volume.shape != volume.shape:\n                # Compute zoom to match volume shape exactly\n                zf = [volume.shape[i] / mask_volume.shape[i] for i in range(3)]\n                mask_volume = zoom(mask_volume.astype(np.float32), zf, order=0)\n                mask_volume = (mask_volume > 0.5).astype(np.uint8)\n        else:\n            # No pixel mask — create empty mask\n            # Positive slices will be treated as weakly supervised\n            mask_volume = np.zeros(volume.shape, dtype=np.uint8)\n        \n        # ── Step 3: Resample volume + mask together ──────────────────────\n        volume, _ = resample_volume(volume, spacing, target_spacing, order=1)\n        if row['has_seg_mask'] and mask_volume is not None:\n            mask_volume, _ = resample_volume(mask_volume.astype(np.float32),\n                                             spacing, target_spacing, order=0)\n            mask_volume = (mask_volume > 0.5).astype(np.uint8)\n        else:\n            mask_volume = np.zeros(volume.shape, dtype=np.uint8)\n        \n        stats['resampled_shape'] = list(volume.shape)\n        \n        # ── Step 4: Window + normalise ───────────────────────────────────\n        volume = window_normalise(volume, cfg.hu_min, cfg.hu_max)\n        \n        # ── Step 5: Centre-crop volume + mask ───────────────────────────\n        Z = volume.shape[0]\n        vol_cropped  = np.stack([centre_crop(volume[z], cfg.crop_size) for z in range(Z)])\n        mask_cropped = np.stack([centre_crop(mask_volume[z], cfg.crop_size) for z in range(Z)])\n        \n        # ── Step 6: Select slices ────────────────────────────────────────\n        if row['has_seg_mask'] and mask_cropped.any():\n            pos_idx, neg_idx = select_slices(mask_cropped, cfg.neg_pos_ratio, epoch_seed)\n        else:\n            # No pixel mask: for positives treat all slices as negative\n            # (weak supervision — presence label only)\n            pos_idx = []\n            all_idx = list(range(Z))\n            n_neg = min(cfg.neg_pos_ratio * 5, len(all_idx))  # grab a fixed sample\n            rng = np.random.RandomState(epoch_seed)\n            neg_idx = sorted(rng.choice(all_idx, size=n_neg, replace=False).tolist())\n        \n        stats['n_pos_slices'] = len(pos_idx)\n        stats['n_neg_slices'] = len(neg_idx)\n        \n        # ── Step 7: Extract 2.5-D tensors + save ────────────────────────\n        tensor_dir = Path(cfg.output_dir) / 'tensors'\n        mask_dir   = Path(cfg.output_dir) / 'masks'\n        \n        for z in sorted(pos_idx + neg_idx):\n            tensor = extract_25d(vol_cropped, z, cfg.n_channels)  # (3, 256, 256)\n            mask   = mask_cropped[z]                               # (256, 256)\n            \n            np.save(str(tensor_dir / f\"{uid}_z{z:04d}.npy\"), tensor)\n            np.save(str(mask_dir   / f\"{uid}_z{z:04d}.npy\"), mask)\n    \n    except Exception as e:\n        stats['status'] = 'error'\n        stats['error']  = str(e)\n    \n    finally:\n        try:\n            del volume, mask_volume, vol_cropped, mask_cropped\n        except:\n            pass\n        gc.collect()\n    \n    return stats\n\n\nprint(\"✅ process_patient() defined\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T13:33:54.509615Z","iopub.execute_input":"2026-05-06T13:33:54.510013Z","iopub.status.idle":"2026-05-06T13:33:54.538035Z","shell.execute_reply.started":"2026-05-06T13:33:54.509984Z","shell.execute_reply":"2026-05-06T13:33:54.537219Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Patient-level 5-fold cross-validation splits\n# Using SeriesInstanceUID as the patient identifier\n\nall_uids = cta_df['SeriesInstanceUID'].values\nkf = KFold(n_splits=CFG.n_folds, shuffle=True, random_state=CFG.seed)\n\nfold_assignments = np.zeros(len(cta_df), dtype=int)\nfor fold_idx, (_, val_idx) in enumerate(kf.split(all_uids)):\n    fold_assignments[val_idx] = fold_idx\n\ncta_df['fold'] = fold_assignments\n\n# Save fold assignments\nfold_path = Path(CFG.output_dir) / 'metadata' / 'fold_assignments.csv'\ncta_df[['SeriesInstanceUID', 'Aneurysm Present', 'has_seg_mask', 'fold']].to_csv(\n    fold_path, index=False\n)\n\nprint(\"✅ 5-fold CV assignments created\")\nprint(f\"   Saved → {fold_path}\")\nprint()\nprint(\"Fold breakdown:\")\nfor f in range(CFG.n_folds):\n    fold_data = cta_df[cta_df['fold'] == f]\n    pos = fold_data['Aneurysm Present'].sum()\n    neg = (fold_data['Aneurysm Present'] == 0).sum()\n    print(f\"  Fold {f}: {len(fold_data):4d} series | {pos} positive | {neg} negative\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T13:33:54.539151Z","iopub.execute_input":"2026-05-06T13:33:54.53954Z","iopub.status.idle":"2026-05-06T13:33:54.580105Z","shell.execute_reply.started":"2026-05-06T13:33:54.539511Z","shell.execute_reply":"2026-05-06T13:33:54.579232Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Pick one CTA series that HAS a pixel mask to test the full pipeline\ntest_row = cta_df[cta_df['has_seg_mask'] == True].iloc[0]\n\nprint(f\"Testing on: {test_row['SeriesInstanceUID'][:40]}...\")\nprint(f\"Aneurysm Present : {test_row['Aneurysm Present']}\")\nprint(f\"Has seg mask     : {test_row['has_seg_mask']}\")\nprint()\n\nstats = process_patient(test_row, CFG, epoch_seed=SEED)\n\nprint(f\"Status           : {stats['status']}\")\nprint(f\"Original shape   : {stats['original_shape']}\")\nprint(f\"Resampled shape  : {stats['resampled_shape']}\")\nprint(f\"Positive slices  : {stats['n_pos_slices']}\")\nprint(f\"Negative slices  : {stats['n_neg_slices']}\")\nif stats['error']:\n    print(f\"Error: {stats['error']}\")\n\n# Verify saved files\ntensor_dir = Path(CFG.output_dir) / 'tensors'\nmask_dir   = Path(CFG.output_dir) / 'masks'\nsaved_tensors = list(tensor_dir.glob(f\"{test_row['SeriesInstanceUID']}*.npy\"))\nprint(f\"\\nSaved tensors    : {len(saved_tensors)}\")\n\n# Load one tensor and mask to verify shape\nif saved_tensors:\n    sample_tensor = np.load(str(saved_tensors[0]))\n    sample_mask   = np.load(str(mask_dir / saved_tensors[0].name))\n    print(f\"Tensor shape     : {sample_tensor.shape}  ← should be (3, 256, 256)\")\n    print(f\"Mask shape       : {sample_mask.shape}    ← should be (256, 256)\")\n    print(f\"Tensor range     : [{sample_tensor.min():.3f}, {sample_tensor.max():.3f}]  ← should be [0, 1]\")\n    print(f\"Mask unique vals : {np.unique(sample_mask)}  ← should be [0] or [0 1]\")\n    print(f\"Mask has aneurysm: {sample_mask.any()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T13:33:54.581313Z","iopub.execute_input":"2026-05-06T13:33:54.582104Z","iopub.status.idle":"2026-05-06T13:34:15.93067Z","shell.execute_reply.started":"2026-05-06T13:33:54.582073Z","shell.execute_reply":"2026-05-06T13:34:15.929403Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Debug mask vs volume alignment\nuid = test_row['SeriesInstanceUID']\n\n# Load raw volume\nvolume_raw, spacing = load_dicom_volume(os.path.join(CFG.series_dir, uid))\nprint(f\"Volume shape (Z,H,W): {volume_raw.shape}\")\nprint(f\"Spacing (z,y,x)     : {spacing}\")\n\n# Load raw NIfTI files and check all axis orientations\nnii_vol  = nib.load(os.path.join(SEG_DIR, f\"{uid}.nii\"))\nnii_cow  = nib.load(os.path.join(SEG_DIR, f\"{uid}_cowseg.nii\"))\n\nvol_nii  = nii_vol.get_fdata()\ncow_nii  = nii_cow.get_fdata()\n\nprint(f\"\\nNIfTI volume shape  : {vol_nii.shape}  (raw, before transpose)\")\nprint(f\"NIfTI cowseg shape  : {cow_nii.shape}  (raw, before transpose)\")\nprint(f\"NIfTI affine diagonal: {np.diag(nii_cow.affine)}\")\n\n# Check positive voxels in cowseg without any transpose\nprint(f\"\\nCowseg unique vals (raw): {np.unique(cow_nii)}\")\nprint(f\"Cowseg positive voxels  : {(cow_nii > 0).sum()}\")\n\n# Find which axis is Z by checking which dimension matches volume Z\nprint(f\"\\nVolume Z slices : {volume_raw.shape[0]}\")\nprint(f\"NIfTI dims      : {vol_nii.shape}  ← which one is {volume_raw.shape[0]}?\")\n\n# Try all 3 transpose options and check which gives correct Z\nfor axes in [(2,1,0), (2,0,1), (0,1,2), (1,0,2), (0,2,1), (1,2,0)]:\n    transposed = np.transpose(cow_nii, axes)\n    positive_slices = sum(1 for z in range(transposed.shape[0]) if transposed[z].any())\n    print(f\"  Transpose {axes} → shape {transposed.shape} → {positive_slices} positive slices\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T13:34:15.931947Z","iopub.execute_input":"2026-05-06T13:34:15.93267Z","iopub.status.idle":"2026-05-06T13:34:32.034321Z","shell.execute_reply.started":"2026-05-06T13:34:15.932636Z","shell.execute_reply":"2026-05-06T13:34:32.033444Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_aneurysm_mask(series_uid: str, row: pd.Series) -> Optional[np.ndarray]:\n    \"\"\"\n    Fixed version — uses (2,1,0) transpose confirmed by debug cell.\n    \"\"\"\n    cowseg_path = os.path.join(SEG_DIR, f\"{series_uid}_cowseg.nii\")\n    if not os.path.exists(cowseg_path):\n        return None\n    \n    cow_data = nib.load(cowseg_path).get_fdata()\n    \n    # NIfTI stored as (H, W, Z) → transpose to (Z, H, W) to match DICOM\n    cow_data = np.transpose(cow_data, (2, 1, 0))  # now (228, 512, 512)\n    \n    # Find positive location labels for this patient\n    aneurysm_labels = []\n    for i, col in enumerate(LOCATION_COLUMNS):\n        if row[col] == 1:\n            aneurysm_labels.append(i + 1)\n    \n    if not aneurysm_labels:\n        return np.zeros(cow_data.shape, dtype=np.uint8)\n    \n    binary_mask = np.isin(cow_data, aneurysm_labels).astype(np.uint8)\n    return binary_mask\n\n\ndef process_patient(row: pd.Series, cfg: Config, epoch_seed: int = 42) -> Dict:\n    uid = row['SeriesInstanceUID']\n    series_path = os.path.join(cfg.series_dir, uid)\n    \n    stats = {\n        'uid'              : uid,\n        'status'           : 'ok',\n        'has_seg_mask'     : bool(row['has_seg_mask']),\n        'aneurysm_present' : int(row['Aneurysm Present']),\n        'original_shape'   : None,\n        'resampled_shape'  : None,\n        'n_pos_slices'     : 0,\n        'n_neg_slices'     : 0,\n        'error'            : None,\n    }\n    \n    try:\n        # ── Step 1: Load DICOM ───────────────────────────────────────────\n        volume, spacing = load_dicom_volume(series_path)\n        stats['original_shape']   = list(volume.shape)\n        stats['original_spacing'] = list(spacing)\n        target_spacing = (cfg.spacing_z, cfg.spacing_xy, cfg.spacing_xy)\n        \n        # ── Step 2: Load mask (same shape as volume, no correction needed)\n        if row['has_seg_mask']:\n            mask_volume = load_aneurysm_mask(uid, row)\n            # Shapes must match — assert to catch any future issues\n            if mask_volume is not None and mask_volume.shape != volume.shape:\n                raise ValueError(\n                    f\"Mask shape {mask_volume.shape} != volume shape {volume.shape}\"\n                )\n        else:\n            mask_volume = np.zeros(volume.shape, dtype=np.uint8)\n        \n        # ── Step 3: Resample volume + mask together ──────────────────────\n        volume, _ = resample_volume(volume, spacing, target_spacing, order=1)\n        mask_volume, _ = resample_volume(\n            mask_volume.astype(np.float32), spacing, target_spacing, order=0\n        )\n        mask_volume = (mask_volume > 0.5).astype(np.uint8)\n        stats['resampled_shape'] = list(volume.shape)\n        \n        # ── Step 4: Window + normalise ───────────────────────────────────\n        volume = window_normalise(volume, cfg.hu_min, cfg.hu_max)\n        \n        # ── Step 5: Centre-crop ──────────────────────────────────────────\n        Z = volume.shape[0]\n        vol_cropped  = np.stack([centre_crop(volume[z],      cfg.crop_size) for z in range(Z)])\n        mask_cropped = np.stack([centre_crop(mask_volume[z], cfg.crop_size) for z in range(Z)])\n        \n        # ── Step 6: Select slices ────────────────────────────────────────\n        if row['has_seg_mask'] and mask_cropped.any():\n            pos_idx, neg_idx = select_slices(mask_cropped, cfg.neg_pos_ratio, epoch_seed)\n        else:\n            pos_idx = []\n            rng = np.random.RandomState(epoch_seed)\n            n_neg = min(cfg.neg_pos_ratio * 5, Z)\n            neg_idx = sorted(rng.choice(Z, size=n_neg, replace=False).tolist())\n        \n        stats['n_pos_slices'] = len(pos_idx)\n        stats['n_neg_slices'] = len(neg_idx)\n        \n        # ── Step 7: Save tensors + masks ─────────────────────────────────\n        tensor_dir = Path(cfg.output_dir) / 'tensors'\n        mask_dir   = Path(cfg.output_dir) / 'masks'\n        \n        for z in sorted(pos_idx + neg_idx):\n            tensor = extract_25d(vol_cropped, z, cfg.n_channels)\n            mask   = mask_cropped[z]\n            np.save(str(tensor_dir / f\"{uid}_z{z:04d}.npy\"), tensor)\n            np.save(str(mask_dir   / f\"{uid}_z{z:04d}.npy\"), mask)\n    \n    except Exception as e:\n        stats['status'] = 'error'\n        stats['error']  = str(e)\n    \n    finally:\n        try: del volume, mask_volume, vol_cropped, mask_cropped\n        except: pass\n        gc.collect()\n    \n    return stats\n\n\n# ── Clean previous dry run output + re-test ──────────────────────────────────\nimport shutil\nshutil.rmtree(Path(CFG.output_dir) / 'tensors')\nshutil.rmtree(Path(CFG.output_dir) / 'masks')\n(Path(CFG.output_dir) / 'tensors').mkdir()\n(Path(CFG.output_dir) / 'masks').mkdir()\n\n# Re-run dry run\nstats = process_patient(test_row, CFG, epoch_seed=SEED)\nprint(f\"Status          : {stats['status']}\")\nprint(f\"Original shape  : {stats['original_shape']}\")\nprint(f\"Resampled shape : {stats['resampled_shape']}\")\nprint(f\"Positive slices : {stats['n_pos_slices']}\")\nprint(f\"Negative slices : {stats['n_neg_slices']}\")\nif stats['error']:\n    print(f\"❌ Error: {stats['error']}\")\n\n# Verify a positive slice actually has mask pixels\ntensor_dir = Path(CFG.output_dir) / 'tensors'\nmask_dir   = Path(CFG.output_dir) / 'masks'\nsaved = sorted(tensor_dir.glob(f\"{test_row['SeriesInstanceUID']}*.npy\"))\n\npos_masks_found = 0\nfor p in saved:\n    m = np.load(str(mask_dir / p.name))\n    if m.any():\n        pos_masks_found += 1\n\nprint(f\"\\nSaved files     : {len(saved)}\")\nprint(f\"Files with mask : {pos_masks_found}  ← should be ~{stats['n_pos_slices']}\")\nprint(f\"✅ Mask fix verified!\" if pos_masks_found > 0 else \"❌ Still broken\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T13:34:32.036707Z","iopub.execute_input":"2026-05-06T13:34:32.03708Z","iopub.status.idle":"2026-05-06T13:34:48.687084Z","shell.execute_reply.started":"2026-05-06T13:34:32.037051Z","shell.execute_reply":"2026-05-06T13:34:48.68595Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Visualise a positive slice to confirm mask overlays correctly\nmask_dir   = Path(CFG.output_dir) / 'masks'\ntensor_dir = Path(CFG.output_dir) / 'tensors'\n\n# Find a slice with aneurysm pixels\npos_files = []\nfor p in sorted(tensor_dir.glob(f\"{test_row['SeriesInstanceUID']}*.npy\")):\n    m = np.load(str(mask_dir / p.name))\n    if m.any():\n        pos_files.append(p)\n\n# Plot 3 positive slices\nfig, axes = plt.subplots(3, 4, figsize=(16, 12), facecolor='#0d1117')\nfig.suptitle('QC — Positive Slices with Aneurysm Mask Overlay', \n             color='white', fontsize=14, fontweight='bold')\n\nfor row_idx, p in enumerate(pos_files[:3]):\n    tensor = np.load(str(p))          # (3, 256, 256)\n    mask   = np.load(str(mask_dir / p.name))  # (256, 256)\n    \n    channel_labels = ['Z-1 (prev)', 'Z (central)', 'Z+1 (next)', 'Central + Mask']\n    \n    for ch in range(3):\n        ax = axes[row_idx, ch]\n        ax.imshow(tensor[ch], cmap='gray', vmin=0, vmax=1)\n        ax.set_title(channel_labels[ch], color='#8b949e', fontsize=9)\n        ax.axis('off')\n        ax.set_facecolor('#161b22')\n    \n    # 4th column: central slice + mask overlay\n    ax = axes[row_idx, 3]\n    ax.imshow(tensor[1], cmap='gray', vmin=0, vmax=1)\n    if mask.any():\n        ax.contour(mask, levels=[0.5], colors=['#ff4444'], linewidths=2)\n    ax.set_title(channel_labels[3], color='#ff7b72', fontsize=9)\n    ax.axis('off')\n    \n    # Slice info\n    z_idx = int(p.stem.split('_z')[1])\n    axes[row_idx, 0].set_ylabel(f'Slice Z={z_idx}', color='#58a6ff', fontsize=9)\n\nplt.tight_layout()\nplt.savefig(\n    Path(CFG.output_dir) / 'viz' / 'qc_positive_slices.png',\n    dpi=150, bbox_inches='tight', facecolor='#0d1117'\n)\nplt.show()\nprint(\"✅ QC plot saved\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T13:34:48.688399Z","iopub.execute_input":"2026-05-06T13:34:48.688772Z","iopub.status.idle":"2026-05-06T13:34:52.018437Z","shell.execute_reply.started":"2026-05-06T13:34:48.688733Z","shell.execute_reply":"2026-05-06T13:34:52.01713Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"all_stats = []\nerrors    = []\n\nprint(f\"Processing {len(cta_df)} CTA series...\")\nprint(f\"  - {cta_df['has_seg_mask'].sum()} with pixel masks\")\nprint(f\"  - {(~cta_df['has_seg_mask']).sum()} label-only\")\nprint()\n\nfor _, row in tqdm(cta_df.iterrows(), total=len(cta_df), desc=\"Preprocessing\"):\n    stats = process_patient(row, CFG, epoch_seed=SEED)\n    all_stats.append(stats)\n    if stats['status'] == 'error':\n        errors.append(stats)\n\n# Summary\nstats_df = pd.DataFrame(all_stats)\nprint(f\"\\n{'='*45}\")\nprint(f\"  PREPROCESSING COMPLETE\")\nprint(f\"{'='*45}\")\nprint(f\"  Total processed   : {len(stats_df)}\")\nprint(f\"  Successful        : {(stats_df.status == 'ok').sum()}\")\nprint(f\"  Errors            : {(stats_df.status == 'error').sum()}\")\nprint(f\"  Total pos slices  : {stats_df.n_pos_slices.sum():,}\")\nprint(f\"  Total neg slices  : {stats_df.n_neg_slices.sum():,}\")\nprint(f\"  Total slices saved: {stats_df.n_pos_slices.sum() + stats_df.n_neg_slices.sum():,}\")\nprint(f\"{'='*45}\")\n\nif errors:\n    print(f\"\\n⚠️  Errors:\")\n    for e in errors[:5]:\n        print(f\"  {e['uid'][:40]}... → {e['error']}\")\n\n# Save stats\nstats_df.to_csv(\n    Path(CFG.output_dir) / 'metadata' / 'processing_stats.csv',\n    index=False\n)\nprint(\"\\n✅ Stats saved\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T13:34:52.020001Z","iopub.execute_input":"2026-05-06T13:34:52.020464Z","execution_failed":"2026-05-07T01:33:26.476Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}}]}