{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":99552,"databundleVersionId":13851420},{"sourceType":"datasetVersion","sourceId":13382569,"datasetId":8491061,"databundleVersionId":14092251},{"sourceType":"modelInstanceVersion","sourceId":828450,"databundleVersionId":16625315,"modelInstanceId":629989,"modelId":641898}],"dockerImageVersionId":31328,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# ============================================================\n# CELL 1 — Setup\n# ============================================================\nimport os, ast, gc\nos.environ[\"PYTORCH_CUDA_ALLOC_CONF\"] = \"expandable_segments:True\"\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom tqdm.notebook import tqdm\n\n# ── PATHS ────────────────────────────────────────────────────\nINPUT_DIR = \"/kaggle/input/competitions/rsna-intracranial-aneurysm-detection/\"\nNPY_DIR   = \"/kaggle/input/datasets/sachchidanandadaki/rsna-3d-preprocessed/processed_3D_isotropic/\"\nMODEL_DIR = \"/kaggle/input/models/diyasilawat/aneurysm-detection-model/pytorch/best_model/1/\"\nMASK_DIR  = \"/kaggle/working/processed_masks/\"   # ← was missing\n\nos.makedirs(MASK_DIR, exist_ok=True)\n\n# Quick verification\nprint(\"INPUT_DIR exists:\", os.path.exists(INPUT_DIR))\nprint(\"NPY_DIR   exists:\", os.path.exists(NPY_DIR))\nprint(\"MASK_DIR  exists:\", os.path.exists(MASK_DIR))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T20:00:49.71702Z","iopub.execute_input":"2026-04-13T20:00:49.717298Z","iopub.status.idle":"2026-04-13T20:00:49.725394Z","shell.execute_reply.started":"2026-04-13T20:00:49.717274Z","shell.execute_reply":"2026-04-13T20:00:49.724358Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CELL 2 — Load data and check what you have\n# ============================================================\ntrain_df = pd.read_csv(f\"{INPUT_DIR}train.csv\")       \nloc_df   = pd.read_csv(f\"{INPUT_DIR}train_localizers.csv\")  \n\n# These are the 13 artery location columns\nlocation_cols = [\n    'Left Infraclinoid Internal Carotid Artery',\n    'Right Infraclinoid Internal Carotid Artery',\n    'Left Supraclinoid Internal Carotid Artery',\n    'Right Supraclinoid Internal Carotid Artery',\n    'Left Middle Cerebral Artery',\n    'Right Middle Cerebral Artery',\n    'Anterior Communicating Artery',\n    'Left Anterior Cerebral Artery',\n    'Right Anterior Cerebral Artery',\n    'Left Posterior Communicating Artery',\n    'Right Posterior Communicating Artery',\n    'Basilar Tip',\n    'Other Posterior Circulation',\n]\n\n# All 14 labels (presence first, then 13 locations)\nall_label_cols = ['Aneurysm Present'] + location_cols\n\nprint(f\"Total series in train.csv: {len(train_df)}\")\nprint(f\"Aneurysm Present rate: {train_df['Aneurysm Present'].mean():.1%}\")\nprint()\n\n# Check how many .npy files actually exist\nexisting = [\n    sid for sid in train_df['SeriesInstanceUID']\n    if os.path.exists(os.path.join(NPY_DIR, f\"{sid}.npy\"))\n]\nprint(f\"NPY files found: {len(existing)} / {len(train_df)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T20:01:17.964514Z","iopub.execute_input":"2026-04-13T20:01:17.964747Z","iopub.status.idle":"2026-04-13T20:01:18.085204Z","shell.execute_reply.started":"2026-04-13T20:01:17.964723Z","shell.execute_reply":"2026-04-13T20:01:18.084345Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CELL 3 — Per-Class Positive Rate Analysis & Weight Computation\n# ============================================================\n# Restrict analysis to series with available preprocessed volumes.\navailable_df = train_df[train_df['SeriesInstanceUID'].isin(existing)].copy()\n\nprint(f\"{'Label':<50} {'Positives':>9} {'Rate':>8} {'pos_weight':>11}\")\nprint(\"-\" * 82)\n\npos_weights = {}\nfor col in all_label_cols:\n    if col not in available_df.columns:\n        continue\n    pos_rate = available_df[col].mean()\n    n_pos    = int(available_df[col].sum())\n    # Standard BCE pos_weight: negatives / positives.\n    # Capped at 50.0 to prevent gradient instability on near-zero-frequency classes.\n    pw = min((1 - pos_rate) / pos_rate if pos_rate > 0 else 50.0, 50.0)\n    pos_weights[col] = pw\n    print(f\"{col:<50} {n_pos:>9} {pos_rate:>7.1%} {pw:>11.2f}x\")\n\n# Convert to plain float — np.float64 wrappers cause issues in PyTorch tensor construction.\npos_weights_list = [float(round(pos_weights[c], 2)) for c in all_label_cols]\nprint(f\"\\npos_weights_list = {pos_weights_list}\")\n\n# ── Coordinate structure audit (critical for Cell 4 mask generation) ──\nsample_coords = loc_df['coordinates'].iloc[0]\nprint(f\"\\nSample coordinate: {sample_coords}\")\nhas_z = all(\n    'z' in ast.literal_eval(r) if isinstance(r, str) else False\n    for r in loc_df['coordinates'].head(20)\n)\nprint(f\"Z coordinate present in loc_df: {has_z}\")\nprint(f\"loc_df shape: {loc_df.shape} | columns: {loc_df.columns.tolist()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T20:11:06.243723Z","iopub.execute_input":"2026-04-13T20:11:06.244009Z","iopub.status.idle":"2026-04-13T20:11:06.256894Z","shell.execute_reply.started":"2026-04-13T20:11:06.243986Z","shell.execute_reply":"2026-04-13T20:11:06.255634Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CELL 4 — Manifest Generation with Anisotropic Gaussian Masks\n# ============================================================\n# Coordinates in loc_df contain only x and y (no z).\n# Strategy: anchor the heatmap at the correct x,y position using a\n# tight sigma, and span the full z-axis with a wide sigma to encode\n# depth uncertainty without discarding the spatial information we have.\n\nTARGET_SHAPE = (96, 96, 96)   # Locked — must match training and inference.\nORIG_XY      = 512\nSCALE_XY     = TARGET_SHAPE[1] / ORIG_XY   # 0.1875\n\ndef create_anisotropic_mask(shape, xy_coords, sigma_xy=3.0, sigma_z=16.0):\n    \"\"\"\n    Generates a 3-D Gaussian heatmap for cases where only (x, y) are known.\n\n    Parameters\n    ----------\n    shape     : (D, H, W) output volume shape\n    xy_coords : list of (y, x) tuples in target-space pixels\n    sigma_xy  : spatial spread in the x and y axes (tight — coordinates known)\n    sigma_z   : spatial spread in the z axis (wide — coordinate unknown)\n    \"\"\"\n    mask = np.zeros(shape, dtype=np.float32)\n    if not xy_coords:\n        return mask\n\n    D, H, W = shape\n    z_center = D // 2   # Place heatmap at volume midpoint along depth axis.\n\n    for (y, x) in xy_coords:\n        z_min, z_max = 0, D\n        y_min = max(0, int(y - 4 * sigma_xy))\n        y_max = min(H, int(y + 4 * sigma_xy + 1))\n        x_min = max(0, int(x - 4 * sigma_xy))\n        x_max = min(W, int(x + 4 * sigma_xy + 1))\n\n        gz = np.arange(z_min, z_max)\n        gy = np.arange(y_min, y_max)\n        gx = np.arange(x_min, x_max)\n        gz, gy, gx = np.meshgrid(gz, gy, gx, indexing='ij')\n\n        heatmap = np.exp(\n            -((gz - z_center) ** 2) / (2 * sigma_z  ** 2)\n            -((gy - y)        ** 2) / (2 * sigma_xy ** 2)\n            -((gx - x)        ** 2) / (2 * sigma_xy ** 2)\n        )\n        mask[z_min:z_max, y_min:y_max, x_min:x_max] = np.maximum(\n            mask[z_min:z_max, y_min:y_max, x_min:x_max], heatmap\n        )\n    return mask\n\n\nloc_grouped = loc_df.groupby('SeriesInstanceUID')\nmanifest    = []\n\nprint(f\"Building manifest for {len(existing)} series...\")\n\nfor sid in tqdm(existing):\n    row      = train_df[train_df['SeriesInstanceUID'] == sid].iloc[0]\n    img_path = os.path.join(NPY_DIR, f\"{sid}.npy\")\n\n    # Parse x,y coordinates; scale from original 512 space to TARGET_SHAPE space.\n    xy_coords = []\n    if sid in loc_grouped.groups:\n        for _, loc_row in loc_grouped.get_group(sid).iterrows():\n            try:\n                raw = ast.literal_eval(loc_row['coordinates'])\n                x   = float(raw['x']) * SCALE_XY\n                y   = float(raw['y']) * SCALE_XY\n                xy_coords.append((\n                    float(np.clip(y, 0, TARGET_SHAPE[1] - 1)),\n                    float(np.clip(x, 0, TARGET_SHAPE[2] - 1)),\n                ))\n            except Exception:\n                continue\n\n    # Generate mask: tight xy (sigma=3), wide z (sigma=16 ≈ ±1/3 of depth axis).\n    mask      = create_anisotropic_mask(TARGET_SHAPE, xy_coords,\n                                        sigma_xy=3.0, sigma_z=16.0)\n    mask_path = os.path.join(MASK_DIR, f\"{sid}_mask.npz\")\n    np.savez_compressed(mask_path, mask=mask.astype(np.float16))\n\n    # Class label vector: [Aneurysm Present] + [13 artery locations].\n    cls_label = [int(row['Aneurysm Present'])] + [\n        int(row[col]) if col in row.index else 0\n        for col in location_cols\n    ]\n\n    manifest.append({\n        \"image\": img_path,\n        \"label\": mask_path,\n        \"cls\":   str(cls_label),\n    })\n\nmanifest_df = pd.DataFrame(manifest)\nmanifest_df.to_csv(\"/kaggle/working/manifest.csv\", index=False)\nprint(f\"\\nManifest saved — {len(manifest_df)} entries\")\nprint(manifest_df.head(3))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T20:14:34.919329Z","iopub.execute_input":"2026-04-13T20:14:34.91956Z","iopub.status.idle":"2026-04-13T20:15:47.788398Z","shell.execute_reply.started":"2026-04-13T20:14:34.91954Z","shell.execute_reply":"2026-04-13T20:15:47.787865Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CELL 5 — Visual Sanity Check: Mask Alignment Verification\n# ============================================================\ndf = pd.read_csv(\"/kaggle/working/manifest.csv\")\ndf['is_pos'] = df['cls'].apply(lambda x: ast.literal_eval(x)[0] == 1)\npos_df = df[df['is_pos']]\n\nprint(f\"Positive cases: {len(pos_df)} / {len(df)}\")\n\nif len(pos_df) > 0:\n    sample    = pos_df.iloc[0]\n    image     = np.load(sample['image'])\n    if image.ndim == 4: image = image[0]\n    mask_data = np.load(sample['label'])\n    mask      = mask_data['mask'].astype(np.float32)\n\n    if image.shape != mask.shape:\n        from scipy.ndimage import zoom\n        zf    = [m/i for m, i in zip(mask.shape, image.shape)]\n        image = zoom(image, zf, order=1)\n\n    z, y, x = np.unravel_index(mask.argmax(), mask.shape)\n    print(f\"Peak mask value : {mask.max():.4f}\")\n    print(f\"Peak location   : Z={z}, Y={y}, X={x}\")\n\n    fig, axes = plt.subplots(1, 3, figsize=(15, 5))\n    axes[0].imshow(image[z], cmap='bone');  axes[0].set_title(f'CT Slice  Z={z}')\n    axes[1].imshow(mask[z],  cmap='hot');   axes[1].set_title('Heatmap Mask')\n    axes[2].imshow(image[z], cmap='bone');  axes[2].imshow(mask[z], cmap='hot', alpha=0.5)\n    axes[2].set_title('Overlay')\n    plt.tight_layout(); plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T20:17:16.849249Z","iopub.execute_input":"2026-04-13T20:17:16.849494Z","iopub.status.idle":"2026-04-13T20:17:17.578202Z","shell.execute_reply.started":"2026-04-13T20:17:16.849474Z","shell.execute_reply":"2026-04-13T20:17:17.577502Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}