{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":99552,"databundleVersionId":13441085}],"dockerImageVersionId":31089,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":" **Step 0 : Meta loader and Exploratory Data Analysis (EDA)**\n  \n     Show Metadata Distributions\n  \n-  **Step 1: Data Loader**\n\n-  \n    • Read DICOM series using pydicom or SimpleITK (or MONAI).\n    • Stack slices → 3D numpy array.\n    • Normalize per modality:\n        ◦ CTA: clip HU (-1000 to 1000) → rescale.\n        ◦ MRI: z-score normalization.\n\n- **Step 2: Preprocessing**\n\n- \n    • Resample to isotropic voxel spacing (1 mm or 1.5 mm).\n    • Resize to fixed shape (like (128, 128, 128)).\n\n- **Step 3: Baseline Model Options**\n\n- \n    - Option A (Preferred): 3D DenseNet121 (MONAI)\n        • Input shape: (1, 128, 128, 128).\n        • Output: 14 (multi-label).\n        • Loss: BCEWithLogitsLoss with weight (13 for aneurysm present).\n    - Option B (Backup): 2.5D EfficientNet\n        • Convert scan to 2D slices or 3-slice stacks.\n        • Use pretrained ImageNet model.\n        • Easier but loses full 3D context.\n\n- **Step 4: Offline Inference**\n\n- \n    • Save model weights (.pth) into Kaggle input/.\n    • Create inference function:\n      python\n      CopyEdit\n      def predict(series_uid):\n          # load DICOM → preprocess → model → sigmoid outputs\n    • Kaggle evaluation API wrapper calls predict().","metadata":{}},{"cell_type":"markdown","source":"References more from to make sure submit score\n\nhttps://www.kaggle.com/code/ahsuna123/voxel-by-voxel-3d-cnn-intracranial-aneurysms","metadata":{}},{"cell_type":"markdown","source":"# Imports","metadata":{}},{"cell_type":"code","source":"%%time\nimport os\nimport gc\nimport shutil\nfrom collections import OrderedDict\nfrom typing import Tuple, List\nimport seaborn as sns\nimport pandas as pd\nimport cv2\nfrom glob import glob\nimport numpy as np\nimport polars as pl\nimport pydicom\nfrom scipy import ndimage\nimport matplotlib.pyplot as plt\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch import amp\nimport warnings\nwarnings.filterwarnings('ignore')\n\nimport kaggle_evaluation.rsna_inference_server\n\n# Device / AMP\nDEVICE = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nUSE_AMP = torch.cuda.is_available()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T08:26:45.364838Z","iopub.execute_input":"2025-08-22T08:26:45.365165Z","iopub.status.idle":"2025-08-22T08:26:49.452877Z","shell.execute_reply.started":"2025-08-22T08:26:45.36511Z","shell.execute_reply":"2025-08-22T08:26:49.451532Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#  Load Metadata","metadata":{}},{"cell_type":"code","source":"%%time\ntrain_df = pd.read_csv(\"/kaggle/input/rsna-intracranial-aneurysm-detection/train.csv\")\nlocalizer_df = pd.read_csv(\"/kaggle/input/rsna-intracranial-aneurysm-detection/train_localizers.csv\")\n\nprint(\"Training samples:\", len(train_df))\nprint(\"Aneurysm prevalence: {:.2f}%\".format(train_df[\"Aneurysm Present\"].mean() * 100))   ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T08:26:49.454031Z","iopub.execute_input":"2025-08-22T08:26:49.454649Z","iopub.status.idle":"2025-08-22T08:26:49.502724Z","shell.execute_reply.started":"2025-08-22T08:26:49.45461Z","shell.execute_reply":"2025-08-22T08:26:49.501354Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Exploratory Data Analysis (EDA)\n Show Metadata Distributions","metadata":{}},{"cell_type":"code","source":"%%time\nfig, axes = plt.subplots(1, 3, figsize=(15, 5))\n\n# Age distribution\nsns.histplot(train_df, x=\"PatientAge\", hue=\"Aneurysm Present\", ax=axes[0], bins=30)\naxes[0].set_title(\"Age Distribution\")\n\n# Sex distribution\nsns.countplot(data=train_df, x=\"PatientSex\", hue=\"Aneurysm Present\", ax=axes[1])\naxes[1].set_title(\"Sex vs Aneurysm\")\n\n# Modality distribution\nsns.countplot(data=train_df, x=\"Modality\", ax=axes[2])\naxes[2].set_title(\"Imaging Modality\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T08:26:49.504613Z","iopub.execute_input":"2025-08-22T08:26:49.504935Z","iopub.status.idle":"2025-08-22T08:26:50.264665Z","shell.execute_reply.started":"2025-08-22T08:26:49.504908Z","shell.execute_reply":"2025-08-22T08:26:50.26322Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Load a Single DICOM Series","metadata":{}},{"cell_type":"code","source":"%%time\ndef load_dicom_slice(dicom_path):\n    ds = pydicom.dcmread(dicom_path)\n    img = ds.pixel_array\n    img = (img - img.min()) / (img.max() - img.min() + 1e-6)  # Normalize\n    return (img * 255).astype(np.uint8)\n\ndef get_first_aneurysm_series():\n    return train_df[train_df[\"Aneurysm Present\"] == 1].iloc[0][\"SeriesInstanceUID\"]\n\ndef get_first_normal_series():\n    return train_df[train_df[\"Aneurysm Present\"] == 0].iloc[0][\"SeriesInstanceUID\"]   ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T08:26:50.267109Z","iopub.execute_input":"2025-08-22T08:26:50.267448Z","iopub.status.idle":"2025-08-22T08:26:50.274787Z","shell.execute_reply.started":"2025-08-22T08:26:50.267421Z","shell.execute_reply":"2025-08-22T08:26:50.273715Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":" Visualize Aneurysm Case with Localizer","metadata":{}},{"cell_type":"code","source":"%%time\ndef plot_aneurysm_with_localizer(series_uid):\n    # Get all DICOMs in series\n    series_path = f\"/kaggle/input/rsna-intracranial-aneurysm-detection/series/{series_uid}\"\n    dicom_files = sorted(glob(os.path.join(series_path, \"*.dcm\")))\n    \n    if len(dicom_files) == 0:\n        print(\"No DICOMs found.\")\n        return\n\n    # Find matching localizer\n    loc_row = localizer_df[localizer_df[\"SeriesInstanceUID\"] == series_uid]\n    if len(loc_row) == 0:\n        print(\"No localizer for this series.\")\n        return\n\n    sop_uid = loc_row.iloc[0][\"SOPInstanceUID\"]\n    x, y = eval(loc_row.iloc[0][\"coordinates\"])  # e.g., \"[120, 150]\"\n    location = loc_row.iloc[0][\"location\"]\n\n    # Find and show that slice\n    target_file = None\n    for f in dicom_files:\n        if sop_uid in f:\n            target_file = f\n            break\n\n    if not target_file:\n        print(\"Localizer SOP not found in series.\")\n        return\n\n    img = load_dicom_slice(target_file)\n\n    plt.figure(figsize=(8, 8))\n    plt.imshow(img, cmap=\"gray\")\n    plt.scatter(x, y, color=\"red\", s=200, marker='x')\n    plt.title(f\"Aneurysm: {location} (x={x}, y={y})\\nSeries: {series_uid}\")\n    plt.axis(\"off\")\n    plt.show()\n\n# Show one aneurysm case\naneurysm_series = get_first_aneurysm_series()\n#print(aneurysm_series)\nplot_aneurysm_with_localizer(aneurysm_series) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T08:26:50.27578Z","iopub.execute_input":"2025-08-22T08:26:50.276107Z","iopub.status.idle":"2025-08-22T08:26:50.589535Z","shell.execute_reply.started":"2025-08-22T08:26:50.276083Z","shell.execute_reply":"2025-08-22T08:26:50.588218Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"This will display a single DICOM slice with a red 'X' marking the exact location of the aneurysm, using the coordinates from train_localizers.csv.","metadata":{}},{"cell_type":"code","source":"%%time\ndef load_dicom_volume(series_path):\n    dcm_paths = sorted(glob(os.path.join(series_path, \"*.dcm\")))\n    datasets = [pydicom.dcmread(p) for p in dcm_paths]\n    \n    # Sort by Z position\n    z_pos = [float(d.ImagePositionPatient[2]) for d in datasets]\n    sorted_dcms = [d for _, d in sorted(zip(z_pos, datasets))]\n    \n    volume = np.stack([d.pixel_array for d in sorted_dcms])\n    return volume\n\ndef create_mip(volume, axis=0):\n    \"\"\"Create MIP along given axis\"\"\"\n    return np.max(volume, axis=axis)\n\ndef plot_mip_example(series_uid):\n    series_path = f\"/kaggle/input/rsna-intracranial-aneurysm-detection/series/{series_uid}\"\n    volume = load_dicom_volume(series_path)\n    \n    # Create MIPs\n    mip_axial = create_mip(volume, axis=0)  # (H, W)\n    mip_coronal = create_mip(volume, axis=1)  # (D, W)\n    mip_sagittal = create_mip(volume, axis=2)  # (D, H)\n\n    # Normalize\n    def norm(img):\n        return (img - img.min()) / (img.max() - img.min() + 1e-6)\n\n    # Plot\n    fig, axes = plt.subplots(1, 3, figsize=(15, 5))\n    axes[0].imshow(norm(mip_axial), cmap=\"gray\")\n    axes[0].set_title(\"Axial MIP\")\n    axes[0].axis(\"off\")\n\n    axes[1].imshow(norm(mip_coronal), cmap=\"gray\")\n    axes[1].set_title(\"Coronal MIP\")\n    axes[1].axis(\"off\")\n\n    axes[2].imshow(norm(mip_sagittal), cmap=\"gray\")\n    axes[2].set_title(\"Sagittal MIP\")\n    axes[2].axis(\"off\")\n\n    plt.suptitle(f\"MIPs - Series: {series_uid}\")\n    plt.tight_layout()\n    plt.show()\n\n# Show MIPs for aneurysm case\nplot_mip_example(aneurysm_series)   ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T08:26:50.590967Z","iopub.execute_input":"2025-08-22T08:26:50.591758Z","iopub.status.idle":"2025-08-22T08:26:55.261Z","shell.execute_reply.started":"2025-08-22T08:26:50.591717Z","shell.execute_reply":"2025-08-22T08:26:55.260184Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Compare Aneurysm vs Normal MIPs","metadata":{}},{"cell_type":"code","source":"%%time\nnormal_series = get_first_normal_series()\n\nprint(\"✅ Aneurysm Case\")\nplot_mip_example(aneurysm_series)\n\nprint(\"✅ Normal Case\")\nplot_mip_example(normal_series)   ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T08:26:55.261814Z","iopub.execute_input":"2025-08-22T08:26:55.262156Z","iopub.status.idle":"2025-08-22T08:27:03.218648Z","shell.execute_reply.started":"2025-08-22T08:26:55.262109Z","shell.execute_reply":"2025-08-22T08:27:03.217564Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Possible reasons for mismatches:","metadata":{}},{"cell_type":"code","source":"%%time\n# Aneurysm presence by modality\nplt.figure(figsize=(8, 5))\nsns.countplot(data=train_df, x='Modality', hue='Aneurysm Present')\nplt.title('Aneurysm Distribution by Imaging Modality')\nplt.xlabel('Modality')\nplt.ylabel('Count')\nplt.legend(title='Aneurysm Present', labels=['No', 'Yes'])\nplt.show()\n\n# Age distribution\nplt.figure(figsize=(8, 5))\nsns.histplot(data=train_df, x='PatientAge', hue='Aneurysm Present', bins=30, kde=True)\nplt.title('Patient Age Distribution by Aneurysm Status')\nplt.xlabel('Age')\nplt.ylabel('Count')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T08:27:03.219982Z","iopub.execute_input":"2025-08-22T08:27:03.220363Z","iopub.status.idle":"2025-08-22T08:27:03.817163Z","shell.execute_reply.started":"2025-08-22T08:27:03.220336Z","shell.execute_reply":"2025-08-22T08:27:03.815624Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Processing, prediction and submition","metadata":{}},{"cell_type":"code","source":"%%time\nfrom typing import Tuple, List\n\n# =====================\n# Competition Constants\n# =====================\nID_COL: str = 'SeriesInstanceUID'\n\nLABEL_COLS: List[str] = [\n    'Left Infraclinoid Internal Carotid Artery',\n    'Right Infraclinoid Internal Carotid Artery',\n    'Left Supraclinoid Internal Carotid Artery',\n    'Right Supraclinoid Internal Carotid Artery',\n    'Left Middle Cerebral Artery',\n    'Right Middle Cerebral Artery',\n    'Anterior Communicating Artery',\n    'Left Anterior Cerebral Artery',\n    'Right Anterior Cerebral Artery',\n    'Left Posterior Communicating Artery',\n    'Right Posterior Communicating Artery',\n    'Basilar Tip',\n    'Other Posterior Circulation',\n    'Aneurysm Present',\n]\n\n# =========\n# File Paths\n# =========\nMODEL_WEIGHTS_PATH: str = \"/kaggle/input/voxel-by-voxel-3d-cnn-train/model_weights.pth\"\n\n# ==========================\n# Processing & Model Configs\n# ==========================\nTARGET_SIZE: Tuple[int, int, int] = (64, 64, 64)      # (Depth, Height, Width)\nTARGET_SPACING_MM: float = 1.0                        # Isotropic voxel spacing (mm)\nCTA_WINDOW: Tuple[float, float] = (300.0, 700.0)      # (center, width) for CT windowing\nMRI_Z_CLIP: float = 3.0                               # Z-score clipping threshold for MRI\nLRU_CAPACITY: int = 8                                 # Cache size for LRU memory   ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T08:27:03.818187Z","iopub.execute_input":"2025-08-22T08:27:03.81846Z","iopub.status.idle":"2025-08-22T08:27:03.826004Z","shell.execute_reply.started":"2025-08-22T08:27:03.818438Z","shell.execute_reply":"2025-08-22T08:27:03.82476Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n# ==========================\n# Utility Resizing Functions\n# ==========================\nfrom typing import Tuple, Sequence\nimport numpy as np\nfrom scipy import ndimage\n\ndef _safe_zoom(\n    volume: np.ndarray,\n    zoom_factors: Sequence[float],\n    order: int = 1\n) -> np.ndarray:\n    \"\"\"\n    Safely resample a volume using zoom factors.\n    \n    Ensures zoom factors are valid and compatible with volume dimensions.\n    Uses `ndimage.zoom` with bounds protection.\n    \n    Args:\n        volume: Input array of any dimension.\n        zoom_factors: Scaling factors per dimension (applied in order).\n        order: Interpolation order (0=nearest, 1=linear, 3=cubic).\n    \n    Returns:\n        Resampled array with shape determined by zoom_factors.\n    \"\"\"\n    # Ensure finite values\n    volume = np.nan_to_num(volume, copy=False)\n\n    # Normalize zoom_factors to match volume dimensions\n    ndim = volume.ndim\n    if len(zoom_factors) != ndim:\n        # Extend or truncate zoom_factors to match ndim\n        if len(zoom_factors) > ndim:\n            zoom_factors = zoom_factors[:ndim]\n        else:\n            # Prepend 1.0s for leading dimensions (e.g., time, batch)\n            zoom_factors = (1.0,) * (ndim - len(zoom_factors)) + tuple(zoom_factors)\n    \n    # Clamp zoom factors to avoid numerical issues\n    zoom_factors = tuple(max(1e-6, float(f)) for f in zoom_factors)\n    \n    return ndimage.zoom(volume, zoom_factors, order=order, mode='nearest')\n\n\ndef _resize_slice(\n    arr: np.ndarray,\n    out_h: int,\n    out_w: int\n) -> np.ndarray:\n    \"\"\"\n    Resize a 2D array to target height and width using linear interpolation.\n    \n    Args:\n        arr: 2D input array (H, W).\n        out_h: Target height.\n        out_w: Target width.\n    \n    Returns:\n        Resized 2D array of shape (out_h, out_w), float32 dtype.\n    \"\"\"\n    h, w = arr.shape\n    if h == out_h and w == out_w:\n        return arr.astype(np.float32, copy=False)\n    \n    # Avoid division by zero\n    scale_h = out_h / h\n    scale_w = out_w / w\n    \n    return _safe_zoom(arr, (scale_h, scale_w), order=1).astype(np.float32, copy=False)   ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T08:27:03.827113Z","iopub.execute_input":"2025-08-22T08:27:03.827512Z","iopub.status.idle":"2025-08-22T08:27:03.851815Z","shell.execute_reply.started":"2025-08-22T08:27:03.82746Z","shell.execute_reply":"2025-08-22T08:27:03.850766Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n# === IMPORTS ===\nfrom typing import List, Tuple, Optional\nimport os\nimport numpy as np\nfrom collections import OrderedDict\nimport pydicom\nfrom scipy import ndimage\n\n# === CONSTANTS (Ensure these are defined earlier in the notebook) ===\nID_COL = 'SeriesInstanceUID'\nLABEL_COLS = [\n    'Left Infraclinoid Internal Carotid Artery',\n    'Right Infraclinoid Internal Carotid Artery',\n    'Left Supraclinoid Internal Carotid Artery',\n    'Right Supraclinoid Internal Carotid Artery',\n    'Left Middle Cerebral Artery',\n    'Right Middle Cerebral Artery',\n    'Anterior Communicating Artery',\n    'Left Anterior Cerebral Artery',\n    'Right Anterior Cerebral Artery',\n    'Left Posterior Communicating Artery',\n    'Right Posterior Communicating Artery',\n    'Basilar Tip',\n    'Other Posterior Circulation',\n    'Aneurysm Present',\n]\n\nTARGET_SIZE = (64, 64, 64)\nTARGET_SPACING_MM = 1.0\nCTA_WINDOW = (300.0, 700.0)\nMRI_Z_CLIP = 3.0\nLRU_CAPACITY = 8   ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T08:27:03.85284Z","iopub.execute_input":"2025-08-22T08:27:03.853209Z","iopub.status.idle":"2025-08-22T08:27:03.880165Z","shell.execute_reply.started":"2025-08-22T08:27:03.853181Z","shell.execute_reply":"2025-08-22T08:27:03.878947Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n# ==========================\n# Utility Resizing Functions\n# ==========================\ndef _safe_zoom(\n    volume: np.ndarray,\n    zoom_factors: Tuple[float, ...],\n    order: int = 1\n) -> np.ndarray:\n    \"\"\"Safely resample volume using zoom factors with dimension alignment and bounds protection.\"\"\"\n    volume = np.nan_to_num(volume, nan=0.0, posinf=None, neginf=None)\n    \n    ndim = volume.ndim\n    if len(zoom_factors) != ndim:\n        if len(zoom_factors) > ndim:\n            zoom_factors = zoom_factors[:ndim]\n        else:\n            zoom_factors = (1.0,) * (ndim - len(zoom_factors)) + tuple(zoom_factors)\n    \n    zoom_factors = tuple(max(1e-6, float(f)) for f in zoom_factors)\n    return ndimage.zoom(volume, zoom_factors, order=order, mode='nearest')\n\n\ndef _resize_slice(\n    arr: np.ndarray,\n    out_h: int,\n    out_w: int\n) -> np.ndarray:\n    \"\"\"Resize 2D array to (out_h, out_w) using linear interpolation.\"\"\"\n    h, w = arr.shape\n    if h == out_h and w == out_w:\n        return arr.astype(np.float32, copy=False)\n    \n    scale_h, scale_w = out_h / h, out_w / w\n    return _safe_zoom(arr, (scale_h, scale_w), order=1).astype(np.float32, copy=False)   ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T08:27:03.881295Z","iopub.execute_input":"2025-08-22T08:27:03.881612Z","iopub.status.idle":"2025-08-22T08:27:03.90571Z","shell.execute_reply.started":"2025-08-22T08:27:03.881578Z","shell.execute_reply":"2025-08-22T08:27:03.904092Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n# ==========================\n# DICOM Series Processor\n# ==========================\nfrom collections import OrderedDict\nfrom typing import Tuple, Optional\nimport numpy as np\n\nclass DICOMProcessor:\n    \"\"\"\n    Convert a DICOM series folder into a normalized 3D volume (D, H, W) in [0, 1].\n    Applies spacing correction, resizing, and modality-specific intensity normalization.\n    \"\"\"\n    \n    def __init__(\n        self,\n        target_size: Tuple[int, int, int] = TARGET_SIZE,\n        target_spacing_mm: float = TARGET_SPACING_MM,\n        cta_window: Tuple[float, float] = CTA_WINDOW,\n        mri_z_clip: float = MRI_Z_CLIP,\n        lru_capacity: int = LRU_CAPACITY,\n    ):\n        self.target_size = target_size\n        self.target_spacing_mm = target_spacing_mm\n        self.cta_window = cta_window\n        self.mri_z_clip = mri_z_clip\n        self.lru_capacity = lru_capacity\n        self.memory_cache: OrderedDict[str, np.ndarray] = OrderedDict()\n\n    # -------------------\n    # LRU Cache Management\n    # -------------------\n    def _cache_get(self, key: str) -> Optional[np.ndarray]:\n        \"\"\"Retrieve volume from cache and update access order.\"\"\"\n        if key in self.memory_cache:\n            # Move to end to mark as recently used\n            self.memory_cache.move_to_end(key)\n            return self.memory_cache[key]\n        return None\n\n    def _cache_put(self, key: str, vol: np.ndarray) -> None:\n        \"\"\"Insert volume into cache, enforcing LRU eviction policy.\"\"\"\n        self.memory_cache[key] = vol\n        self.memory_cache.move_to_end(key)\n        if len(self.memory_cache) > self.lru_capacity:\n            self.memory_cache.popitem(last=False)  # Remove oldest   ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T08:27:03.910001Z","iopub.execute_input":"2025-08-22T08:27:03.910353Z","iopub.status.idle":"2025-08-22T08:27:03.975965Z","shell.execute_reply.started":"2025-08-22T08:27:03.910328Z","shell.execute_reply":"2025-08-22T08:27:03.974511Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n# ==========================\n# Utility Resizing Functions\n# ==========================\ndef _safe_zoom(\n    volume: np.ndarray,\n    zoom_factors: Tuple[float, ...],\n    order: int = 1\n) -> np.ndarray:\n    \"\"\"Safely resample volume using zoom factors with dimension alignment and bounds protection.\"\"\"\n    volume = np.nan_to_num(volume, nan=0.0, posinf=None, neginf=None)\n    \n    ndim = volume.ndim\n    if len(zoom_factors) != ndim:\n        if len(zoom_factors) > ndim:\n            zoom_factors = zoom_factors[:ndim]\n        else:\n            zoom_factors = (1.0,) * (ndim - len(zoom_factors)) + tuple(zoom_factors)\n    \n    zoom_factors = tuple(max(1e-6, float(f)) for f in zoom_factors)\n    return ndimage.zoom(volume, zoom_factors, order=order, mode='nearest')\n\n\ndef _resize_slice(\n    arr: np.ndarray,\n    out_h: int,\n    out_w: int\n) -> np.ndarray:\n    \"\"\"Resize 2D array to (out_h, out_w) using linear interpolation.\"\"\"\n    h, w = arr.shape\n    if h == out_h and w == out_w:\n        return arr.astype(np.float32, copy=False)\n    \n    scale_h, scale_w = out_h / h, out_w / w\n    return _safe_zoom(arr, (scale_h, scale_w), order=1).astype(np.float32, copy=False)   ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T08:27:03.977258Z","iopub.execute_input":"2025-08-22T08:27:03.977569Z","iopub.status.idle":"2025-08-22T08:27:04.004572Z","shell.execute_reply.started":"2025-08-22T08:27:03.977544Z","shell.execute_reply":"2025-08-22T08:27:04.003355Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"    # -------------------\n    # LRU Cache Management\n    # -------------------\n    def _cache_put(self, key: str, vol: np.ndarray) -> None:\n        \"\"\"Insert volume into cache, enforcing LRU eviction policy.\"\"\"\n        self.memory_cache[key] = vol\n        self.memory_cache.move_to_end(key)\n        if len(self.memory_cache) > self.lru_capacity:\n            self.memory_cache.popitem(last=False)  # Remove least recently used   ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T08:27:04.005358Z","iopub.execute_input":"2025-08-22T08:27:04.005689Z","iopub.status.idle":"2025-08-22T08:27:04.030257Z","shell.execute_reply.started":"2025-08-22T08:27:04.005661Z","shell.execute_reply":"2025-08-22T08:27:04.029197Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":" def _sort_slices(self, ds_list: List[pydicom.dataset.FileDataset]) -> List[pydicom.dataset.FileDataset]:\n        \"\"\"\n        Sort slices by spatial position along the slice normal vector.\n        Falls back to InstanceNumber if geometric sorting fails.\n        \"\"\"\n        try:\n            orient = np.array(ds_list[0].ImageOrientationPatient, dtype=np.float32)\n            row = orient[:3]\n            col = orient[3:]\n            normal = np.cross(row, col)\n            ipp_list = []\n            for ds in ds_list:\n                ipp = np.array(getattr(ds, \"ImagePositionPatient\", [0.0, 0.0, 0.0]), dtype=np.float32)\n                ipp_list.append(np.dot(ipp, normal))\n            sorted_idx = np.argsort(ipp_list)\n            return [ds_list[i] for i in sorted_idx]\n        except Exception:\n            return sorted(ds_list, key=lambda ds: getattr(ds, \"InstanceNumber\", 0))   ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T08:27:04.03127Z","iopub.execute_input":"2025-08-22T08:27:04.031586Z","iopub.status.idle":"2025-08-22T08:27:04.051172Z","shell.execute_reply.started":"2025-08-22T08:27:04.031556Z","shell.execute_reply":"2025-08-22T08:27:04.050402Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":" def _choose_base_shape(self, ds_list: List[pydicom.dataset.FileDataset]) -> Tuple[int, int]:\n        \"\"\"\n        Choose the most frequent (Rows, Columns) as base 2D shape.\n        Falls back to pixel_array shape if metadata missing.\n        \"\"\"\n        shapes = []\n        for ds in ds_list:\n            try:\n                h, w = int(ds.Rows), int(ds.Columns)\n                if h > 0 and w > 0:\n                    shapes.append((h, w))\n                    continue\n            except Exception:\n                pass\n\n            # Fallback: use pixel_array\n            arr = ds.pixel_array\n            if arr.ndim >= 2:\n                h, w = arr.shape[-2], arr.shape[-1]\n                if h > 0 and w > 0:\n                    shapes.append((h, w))\n\n        if not shapes:\n            return self.target_size[1], self.target_size[2]  # Fallback to target (H, W)\n\n        # Find most frequent shape\n        shapes_arr = np.array(shapes)\n        unique_shapes, counts = np.unique(shapes_arr, axis=0, return_counts=True)\n        best_shape = unique_shapes[np.argmax(counts)]\n        return int(best_shape[0]), int(best_shape[1])   ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T08:27:04.052705Z","iopub.execute_input":"2025-08-22T08:27:04.053037Z","iopub.status.idle":"2025-08-22T08:27:04.078249Z","shell.execute_reply.started":"2025-08-22T08:27:04.053006Z","shell.execute_reply":"2025-08-22T08:27:04.077124Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":" def _normalize_by_modality(self, volume: np.ndarray, modality_tag: str) -> np.ndarray:\n        \"\"\"\n        Normalize intensity based on modality:\n          - CT: Apply CTA window (center, width) and scale to [0,1].\n          - MR (or other): Z-score, clip outliers, then scale to [0,1].\n        \n        Returns:\n            Normalized volume with values in [0,1], same shape and dtype float32.\n        \"\"\"\n        # Ensure no NaNs or infinities\n        volume = np.nan_to_num(volume, nan=0.0, posinf=None, neginf=None)\n\n        if modality_tag == \"CT\":\n            # Apply CT windowing: (volume → [0,1])\n            center, width = self.cta_window\n            lower = center - width / 2.0\n            upper = center + width / 2.0\n            volume = np.clip(volume, lower, upper)\n            volume = (volume - lower) / (upper - lower + 1e-6)  # Avoid division by zero\n        else:\n            # MRI or unknown: z-score + clip + scale\n            mean = float(np.mean(volume))\n            std = float(np.std(volume))\n            if std < 1e-6:\n                std = 1e-6  # Prevent division by zero\n            volume = (volume - mean) / std\n            # Clip outliers\n            z_clip = self.mri_z_clip\n            volume = np.clip(volume, -z_clip, z_clip)\n            # Scale to [0,1]\n            volume = (volume + z_clip) / (2.0 * z_clip)\n\n        return volume.astype(np.float32, copy=False)   ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T08:27:04.079179Z","iopub.execute_input":"2025-08-22T08:27:04.079437Z","iopub.status.idle":"2025-08-22T08:27:04.093542Z","shell.execute_reply.started":"2025-08-22T08:27:04.079418Z","shell.execute_reply":"2025-08-22T08:27:04.09245Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"    def load_dicom_series(self, series_path: str) -> np.ndarray:\n        \"\"\"\n        Load and preprocess a DICOM series into a normalized 3D volume (D, H, W) in [0,1].\n        \n        Returns:\n            np.ndarray: Shape self.target_size, dtype float32, values in [0,1].\n        \"\"\"\n        series_id = os.path.basename(series_path.strip(\"/\"))\n\n        # Check memory cache first\n        cached = self._cache_get(series_id)\n        if cached is not None and cached.shape == self.target_size:\n            return cached\n\n        try:\n            # Step 1: Load valid DICOMs with PixelData\n            dicoms = []\n            for root, _, files in os.walk(series_path):\n                for f in files:\n                    if f.lower().endswith(\".dcm\"):\n                        dcm_path = os.path.join(root, f)\n                        try:\n                            ds = pydicom.dcmread(dcm_path, force=True, stop_before_pixels=False)\n                            if hasattr(ds, \"PixelData\"):\n                                dicoms.append(ds)\n                        except Exception as e:\n                            print(f\"[DICOM Read Error] {series_id}: {e}\")\n            if not dicoms:\n                raise ValueError(\"No valid DICOM files found\")\n\n            # Step 2: Sort slices spatially\n            dicoms = self._sort_slices(dicoms)\n\n            # Step 3: Extract metadata\n            has_multiframe = any(getattr(ds, \"NumberOfFrames\", 1) > 1 for ds in dicoms)\n            spacing = self._get_spacing(dicoms, has_multiframe=has_multiframe)\n            base_h, base_w = self._choose_base_shape(dicoms)\n            modality_tag = getattr(dicoms[0], \"Modality\", \"\").upper()\n\n            # Step 4: Decode and preprocess slices\n            vol_slices = []\n            for ds in dicoms:\n                arr = ds.pixel_array.astype(np.float32)\n\n                # Handle multi-frame (e.g., 3D volumes in single file)\n                                # Handle multi-frame (e.g., 3D volumes in single file)\n                if arr.ndim >= 3:\n                    n_frames = int(np.prod(arr.shape[:-2]))\n                    h, w = arr.shape[-2:]\n                    frames = arr.reshape(n_frames, h, w)\n                else:\n                    frames = arr[np.newaxis, ...]\n\n                for frame in frames:\n                    # Handle photometric interpretation\n                    if getattr(ds, \"PhotometricInterpretation\", \"\") == \"MONOCHROME1\":\n                        frame = frame.max() - frame\n\n                    # Apply rescale (Hounsfield units for CT)\n                    slope = float(getattr(ds, \"RescaleSlope\", 1.0))\n                    intercept = float(getattr(ds, \"RescaleIntercept\", 0.0))\n                    frame = frame * slope + intercept\n\n                    # Resize to common base resolution\n                    frame = _resize_slice(frame, base_h, base_w)\n                    vol_slices.append(frame)\n\n            if not vol_slices:\n                raise ValueError(\"No frames extracted\")\n\n            # Stack into (D, H, W)\n            volume = np.stack(vol_slices, axis=0).astype(np.float32)\n\n            # Step 5: Modality-specific normalization\n            volume = self._normalize_by_modality(volume, modality_tag)\n\n            # Step 6: Isotropic resampling to target spacing\n            if self.target_spacing_mm is not None:\n                dz, dy, dx = spacing\n                z, y, x = volume.shape\n                new_shape = (\n                    max(1, int(round(z * dz / self.target_spacing_mm))),\n                    max(1, int(round(y * dy / self.target_spacing_mm))),\n                    max(1, int(round(x * dx / self.target_spacing_mm))),\n                )\n                scale_factors = (new_shape[0] / z, new_shape[1] / y, new_shape[2] / x)\n                volume = _safe_zoom(volume, scale_factors, order=1)\n\n            # Step 7: Final resize to target grid size\n            tz, ty, tx = self.target_size\n            z, y, x = volume.shape\n            if (z, y, x) != (tz, ty, tx):\n                scale_factors = (tz / z, ty / y, tx / x)\n                volume = _safe_zoom(volume, scale_factors, order=1)\n\n            volume = volume.astype(np.float32, copy=False)\n\n            # Cache and return\n            self._cache_put(series_id, volume)\n            return volume\n\n        except Exception as e:\n            print(f\"[Processor] Error processing {series_id}: {e}\")\n            # Return zero-filled volume as fallback\n            fallback = np.zeros(self.target_size, dtype=np.float32)\n            self._cache_put(series_id, fallback)\n            return fallback   ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T08:27:04.094652Z","iopub.execute_input":"2025-08-22T08:27:04.094932Z","iopub.status.idle":"2025-08-22T08:27:04.122099Z","shell.execute_reply.started":"2025-08-22T08:27:04.09491Z","shell.execute_reply":"2025-08-22T08:27:04.121076Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n# =======================\n# 3D CNN (Logits Output)\n# =======================\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom typing import List\n\nclass Simple3DCNN(nn.Module):\n    \"\"\"\n    A compact 3D Convolutional Neural Network for medical volume classification.\n    \n    Architecture:\n        - 4x (Conv3D → BatchNorm3D → ReLU → MaxPool3D)\n        - Adaptive average pooling to (2,2,2)\n        - 3-layer MLP with dropout\n    Output: raw logits (no sigmoid/softmax applied)\n    \n    Input:  (B, 1, D, H, W)  - single-channel 3D volumes\n    Output: (B, num_classes)  - logits\n    \"\"\"\n    \n    def __init__(self, num_classes: int = len(LABEL_COLS)):\n        super(Simple3DCNN, self).__init__()\n        self.num_classes = num_classes\n\n        # Encoder: 4 downsampling blocks\n        self.features = nn.Sequential(\n            # Block 1\n            nn.Conv3d(1, 16, kernel_size=3, padding=1),\n            nn.BatchNorm3d(16),\n            nn.ReLU(inplace=True),\n            nn.MaxPool3d(2),\n            \n            # Block 2\n            nn.Conv3d(16, 32, kernel_size=3, padding=1),\n            nn.BatchNorm3d(32),\n            nn.ReLU(inplace=True),\n            nn.MaxPool3d(2),\n            \n            # Block 3\n            nn.Conv3d(32, 64, kernel_size=3, padding=1),\n            nn.BatchNorm3d(64),\n            nn.ReLU(inplace=True),\n            nn.MaxPool3d(2),\n            \n            # Block 4\n            nn.Conv3d(64, 128, kernel_size=3, padding=1),\n            nn.BatchNorm3d(128),\n            nn.ReLU(inplace=True),\n            nn.MaxPool3d(2),\n        )\n\n        # Global pooling + classifier\n        self.global_pool = nn.AdaptiveAvgPool3d((2, 2, 2))\n        self.classifier = nn.Sequential(\n            nn.Linear(128 * 2 * 2 * 2, 256),\n            nn.ReLU(inplace=True),\n            nn.Dropout(0.5),\n            \n            nn.Linear(256, 128),\n            nn.ReLU(inplace=True),\n            nn.Dropout(0.3),\n            \n            nn.Linear(128, num_classes)  # Logits (no activation)\n        )\n\n    def forward(self, x: torch.Tensor) -> torch.Tensor:\n        \"\"\"\n        Forward pass.\n        \n        Args:\n            x: Input tensor of shape (B, 1, D, H, W)\n        \n        Returns:\n            Logits of shape (B, num_classes)\n        \"\"\"\n        # Ensure input is float32\n        if x.dtype != torch.float32:\n            x = x.float()\n\n        # Feature extraction\n        x = self.features(x)  # (B, 128, d, h, w)\n\n        # Global pooling\n        x = self.global_pool(x)  # (B, 128, 2, 2, 2)\n\n        # Flatten\n        x = torch.flatten(x, 1)  # (B, 128*8)\n\n        # Classification\n        x = self.classifier(x)  # (B, num_classes)\n\n        return x   ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T08:27:04.123058Z","iopub.execute_input":"2025-08-22T08:27:04.123339Z","iopub.status.idle":"2025-08-22T08:27:04.152741Z","shell.execute_reply.started":"2025-08-22T08:27:04.123319Z","shell.execute_reply":"2025-08-22T08:27:04.15157Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n# ==============\n# Inference API\n# ==============\n@torch.no_grad()\ndef predict(series_path: str) -> pl.DataFrame:\n    \"\"\"Server calls this function. Assumes global `model` and `processor` are ready.\"\"\"\n    try:\n        # CPU preprocessing\n        volume = processor.load_dicom_series(series_path)  # (D,H,W) in [0,1]\n        volume_tensor = torch.from_numpy(volume).unsqueeze(0).unsqueeze(0).to(DEVICE)  # (1,1,D,H,W)\n\n        # Forward pass\n        with amp.autocast(device_type='cuda', enabled=USE_AMP):\n            logits = model(volume_tensor)               # (1,14)\n            probs = torch.sigmoid(logits).cpu().numpy().flatten().tolist()\n\n        result_df = pl.DataFrame(data=[probs], schema=LABEL_COLS, orient='row')\n\n        # Cleanup\n        del volume_tensor\n        if torch.cuda.is_available():\n            torch.cuda.empty_cache()\n        gc.collect()\n\n    except Exception as e:\n        print(f\"[Predict] Error: {e}\")\n        result_df = pl.DataFrame(data=[[0.5] * len(LABEL_COLS)], schema=LABEL_COLS, orient='row')\n\n    # Remove shared temp if present\n    shutil.rmtree('/kaggle/shared', ignore_errors=True)\n    return result_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T08:27:04.153684Z","iopub.execute_input":"2025-08-22T08:27:04.153958Z","iopub.status.idle":"2025-08-22T08:27:04.180516Z","shell.execute_reply.started":"2025-08-22T08:27:04.153937Z","shell.execute_reply":"2025-08-22T08:27:04.179335Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n# ==========================\n# Run the Evaluation Server\n# ==========================\ninference_server = kaggle_evaluation.rsna_inference_server.RSNAInferenceServer(predict)\n\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway()\n    try:\n        sub = pl.read_parquet('/kaggle/working/submission.parquet')\n        print(sub.head())\n    except Exception as e:\n        print(f\"Submission parquet not found yet: {e}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T08:27:04.181991Z","iopub.execute_input":"2025-08-22T08:27:04.182312Z","iopub.status.idle":"2025-08-22T08:27:09.201771Z","shell.execute_reply.started":"2025-08-22T08:27:04.182279Z","shell.execute_reply":"2025-08-22T08:27:09.200664Z"}},"outputs":[],"execution_count":null}]}