{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":13451,"datasetId":654585,"databundleVersionId":1188070}],"dockerImageVersionId":31286,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"e09c0ee5","cell_type":"markdown","source":"# Notebook 1 — Exploratory Data Analysis\n**RSNA Intracranial Hemorrhage Detection**\n\nGoals:\n- Understand the label structure and class imbalance\n- Visualise raw DICOM images and the effect of CT windowing\n- Inspect DICOM metadata\n- Understand the co-occurrence of hemorrhage sub-types\n\n> No GPU required for this notebook.","metadata":{}},{"id":"f1f99aeb","cell_type":"code","source":"# ── 0. Install / confirm dependencies ──────────────────────────────────────\n# pydicom is pre-installed on Kaggle; uncomment locally if needed\n# !pip install pydicom -q\n\nimport os, glob, random\nimport numpy as np\nimport pandas as pd\nimport pydicom\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as mpatches\nimport seaborn as sns\nfrom pathlib import Path\n\nsns.set_theme(style='darkgrid', palette='muted')\nplt.rcParams['figure.dpi'] = 110\n\n# ── Kaggle paths (adjust if running locally) ───────────────────────────────\nBASE     = Path('/kaggle/input/competitions/rsna-intracranial-hemorrhage-detection/rsna-intracranial-hemorrhage-detection')\nTRAIN_CSV = BASE / 'stage_2_train.csv'\nTRAIN_DIR = BASE / 'stage_2_train/'   # folder that contains .dcm files\n\nprint('CSV  :', TRAIN_CSV.exists())\nprint('DCMS :', TRAIN_DIR.exists())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T08:04:18.659851Z","iopub.execute_input":"2026-02-19T08:04:18.661019Z","iopub.status.idle":"2026-02-19T08:04:18.676638Z","shell.execute_reply.started":"2026-02-19T08:04:18.660943Z","shell.execute_reply":"2026-02-19T08:04:18.675596Z"}},"outputs":[],"execution_count":null},{"id":"10f27cee","cell_type":"code","source":"# ── 1. Load CSV ──────────────────────────────────────────────────────────────\nraw = pd.read_csv(TRAIN_CSV)\nprint(f'Shape: {raw.shape}')\nraw.head(12)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T08:04:21.998377Z","iopub.execute_input":"2026-02-19T08:04:21.998703Z","iopub.status.idle":"2026-02-19T08:04:26.421186Z","shell.execute_reply.started":"2026-02-19T08:04:21.998677Z","shell.execute_reply":"2026-02-19T08:04:26.419759Z"}},"outputs":[],"execution_count":null},{"id":"d8801fd4","cell_type":"code","source":"# ── 2. Parse ID column into image_id + subtype ────────────────────────────\nraw[['image_id', 'subtype']] = raw['ID'].str.rsplit('_', n=1, expand=True)\n# Note: the column uses the last underscore as separator\n# IDs look like: ID_XXXXXXXX_epidural  → image_id=ID_XXXXXXXX, subtype=epidural\n\nprint('Unique subtypes:', raw['subtype'].unique())\nprint('Unique images  :', raw['image_id'].nunique())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T08:08:12.996158Z","iopub.execute_input":"2026-02-19T08:08:12.996513Z","iopub.status.idle":"2026-02-19T08:08:23.486468Z","shell.execute_reply.started":"2026-02-19T08:08:12.996485Z","shell.execute_reply":"2026-02-19T08:08:23.485215Z"}},"outputs":[],"execution_count":null},{"id":"3883b1b8","cell_type":"code","source":"# ── 3. Pivot to per-image format ─────────────────────────────────────────────\n# The RSNA CSV contains duplicate (image_id, subtype) rows — drop them first.\nraw = raw.drop_duplicates(subset=['image_id', 'subtype'], keep='first')\nprint(f'After dedup: {raw.shape[0]:,} rows')\n\ndf = raw.pivot(index='image_id', columns='subtype', values='Label').reset_index()\ndf.columns.name = None\n\nSUBTYPES = ['any', 'epidural', 'intraparenchymal',\n            'intraventricular', 'subarachnoid', 'subdural']\n\n# binarise (labels are already 0/1 but confirm)\nfor col in SUBTYPES:\n    df[col] = df[col].astype(int)\n\nprint(f'Per-image dataset shape: {df.shape}')\ndf.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T08:13:32.982288Z","iopub.execute_input":"2026-02-19T08:13:32.983517Z","iopub.status.idle":"2026-02-19T08:13:41.464783Z","shell.execute_reply.started":"2026-02-19T08:13:32.983479Z","shell.execute_reply":"2026-02-19T08:13:41.463843Z"}},"outputs":[],"execution_count":null},{"id":"86b69b3c","cell_type":"code","source":"# ── 4. Label distribution ─────────────────────────────────────────────────\ncounts = df[SUBTYPES].sum().sort_values(ascending=False)\ntotals = len(df)\n\nfig, axes = plt.subplots(1, 2, figsize=(14, 5))\n\n# Bar chart\naxes[0].bar(counts.index, counts.values, color=sns.color_palette('muted', 6))\naxes[0].set_title('Positive cases per sub-type')\naxes[0].set_ylabel('Count')\nfor i, (label, val) in enumerate(zip(counts.index, counts.values)):\n    axes[0].text(i, val + 200, f'{val/totals*100:.2f}%',\n                 ha='center', va='bottom', fontsize=9)\n\n# Pie chart for 'any' vs 'no hemorrhage'\nany_pos = int(df['any'].sum())\nany_neg = totals - any_pos\naxes[1].pie([any_pos, any_neg],\n            labels=[f'Hemorrhage ({any_pos})', f'No hemorrhage ({any_neg})'],\n            autopct='%1.1f%%',\n            colors=['#e07b54', '#74b9ff'],\n            startangle=90)\naxes[1].set_title('Overall class balance')\n\nplt.tight_layout()\nplt.savefig('/kaggle/working/label_distribution.png', bbox_inches='tight')\nplt.show()\n\nprint(counts.to_frame('count').assign(pct=lambda x: (x['count']/totals*100).round(2)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T08:14:14.555162Z","iopub.execute_input":"2026-02-19T08:14:14.555529Z","iopub.status.idle":"2026-02-19T08:14:15.209553Z","shell.execute_reply.started":"2026-02-19T08:14:14.555501Z","shell.execute_reply":"2026-02-19T08:14:15.208559Z"}},"outputs":[],"execution_count":null},{"id":"82895db2","cell_type":"code","source":"# ── 5. Co-occurrence matrix ───────────────────────────────────────────────\npos_df = df[df['any'] == 1][SUBTYPES[1:]]   # exclude 'any'\nco = pos_df.T.dot(pos_df)                   # sub-type × sub-type counts\n\nmask = np.zeros_like(co, dtype=bool)\nmask[np.triu_indices_from(mask, k=1)] = True   # upper triangle mask (not needed for full)\n\nplt.figure(figsize=(8, 6))\nsns.heatmap(co, annot=True, fmt='d', cmap='YlOrRd',\n            linewidths=0.5, cbar_kws={'label': 'Co-occurrence count'})\nplt.title('Sub-type Co-occurrence (hemorrhage-positive slices only)')\nplt.tight_layout()\n\nplt.savefig('/kaggle/working/cooccurrence.png', bbox_inches='tight')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T08:27:24.385347Z","iopub.execute_input":"2026-02-19T08:27:24.386101Z","iopub.status.idle":"2026-02-19T08:27:24.907517Z","shell.execute_reply.started":"2026-02-19T08:27:24.386068Z","shell.execute_reply":"2026-02-19T08:27:24.906583Z"}},"outputs":[],"execution_count":null},{"id":"7e46f597","cell_type":"code","source":"# ── 6. DICOM loading utilities ────────────────────────────────────────────\ndef get_dcm_path(image_id: str, dcm_dir: Path) -> Path:\n    \"\"\"Return path to the .dcm file for an image_id.\n    Kaggle stores them as <image_id>.dcm directly inside the folder.\n    The image_id already includes the 'ID_' prefix.\n    \"\"\"\n    return dcm_dir / f'{image_id}.dcm'\n\n\ndef load_dcm_pixel(path: Path) -> np.ndarray:\n    \"\"\"Load raw pixel array from DICOM after applying rescale slope/intercept.\"\"\"\n    dcm = pydicom.dcmread(str(path))\n    img = dcm.pixel_array.astype(np.float32)\n    # Apply Rescale Slope / Intercept to get Hounsfield Units (HU)\n    slope     = float(getattr(dcm, 'RescaleSlope',     1))\n    intercept = float(getattr(dcm, 'RescaleIntercept', 0))\n    img = img * slope + intercept\n    return img, dcm\n\n\ndef apply_window(img_hu: np.ndarray, window_center: int, window_width: int) -> np.ndarray:\n    \"\"\"Clip HU values to the specified window and normalise to [0, 1].\"\"\"\n    lo = window_center - window_width / 2\n    hi = window_center + window_width / 2\n    img = np.clip(img_hu, lo, hi)\n    img = (img - lo) / (hi - lo)   # → [0, 1]\n    return img\n\n\n# Standard CT windows used in neuro-radiology\nWINDOWS = {\n    'brain'    : (40,  80),    # WC=40, WW=80\n    'subdural' : (75, 215),    # WC=75, WW=215\n    'soft'     : (40, 380),    # WC=40, WW=380  (soft tissue)\n}\n\nprint('Windowing utilities defined.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T08:27:43.023298Z","iopub.execute_input":"2026-02-19T08:27:43.024196Z","iopub.status.idle":"2026-02-19T08:27:43.034983Z","shell.execute_reply.started":"2026-02-19T08:27:43.024147Z","shell.execute_reply":"2026-02-19T08:27:43.034029Z"}},"outputs":[],"execution_count":null},{"id":"daf43df0","cell_type":"code","source":"# ── 7. Sample a few images (positive hemorrhage) ──────────────────────────\nSEED = 42\nrandom.seed(SEED)\n\npos_ids = df[df['any'] == 1]['image_id'].tolist()\nneg_ids = df[df['any'] == 0]['image_id'].tolist()\n\nsample_pos = random.sample(pos_ids, 4)\nsample_neg = random.sample(neg_ids, 2)\nsample_ids = sample_pos + sample_neg\n\nprint('Positive samples:', sample_pos)\nprint('Negative samples:', sample_neg)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T08:27:53.963737Z","iopub.execute_input":"2026-02-19T08:27:53.964504Z","iopub.status.idle":"2026-02-19T08:27:54.133819Z","shell.execute_reply.started":"2026-02-19T08:27:53.964472Z","shell.execute_reply":"2026-02-19T08:27:54.132659Z"}},"outputs":[],"execution_count":null},{"id":"f83edcb6","cell_type":"code","source":"# ── 8. Visualise windowing effect side-by-side ────────────────────────────\nfig, axes = plt.subplots(len(sample_ids), 4, figsize=(16, len(sample_ids) * 3.5))\nfig.suptitle('CT Windows: Raw HU | Brain | Subdural | Soft-tissue', fontsize=13, y=1.01)\n\nwindow_names  = list(WINDOWS.keys())\ncol_titles    = ['Raw (HU clipped -100..300)', 'Brain (WC40,WW80)',\n                 'Subdural (WC75,WW215)', 'Soft-tissue (WC40,WW380)']\n\nfor row_idx, img_id in enumerate(sample_ids):\n    path = get_dcm_path(img_id, TRAIN_DIR)\n    if not path.exists():\n        print(f'WARNING: {path} not found – skipping')\n        continue\n    img_hu, _ = load_dcm_pixel(path)\n\n    # label string for y-axis\n    label_info = df[df['image_id'] == img_id][SUBTYPES].values[0]\n    label_str  = ', '.join([s for s, v in zip(SUBTYPES, label_info) if v])\n    if not label_str:\n        label_str = 'No hemorrhage'\n\n    # Column 0: clipped raw HU\n    ax = axes[row_idx, 0]\n    ax.imshow(np.clip(img_hu, -100, 300), cmap='gray')\n    ax.set_ylabel(f'{img_id}\\n{label_str}', fontsize=7)\n    ax.set_title(col_titles[0] if row_idx == 0 else '')\n    ax.axis('off')\n\n    # Columns 1-3: windowed\n    for col_idx, (wname, (wc, ww)) in enumerate(WINDOWS.items(), start=1):\n        windowed = apply_window(img_hu, wc, ww)\n        ax = axes[row_idx, col_idx]\n        ax.imshow(windowed, cmap='gray', vmin=0, vmax=1)\n        ax.set_title(col_titles[col_idx] if row_idx == 0 else '')\n        ax.axis('off')\n\nplt.tight_layout()\nplt.savefig('/kaggle/working/windowing_comparison.png', bbox_inches='tight')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T08:29:05.081326Z","iopub.execute_input":"2026-02-19T08:29:05.081677Z","iopub.status.idle":"2026-02-19T08:29:11.268025Z","shell.execute_reply.started":"2026-02-19T08:29:05.081652Z","shell.execute_reply":"2026-02-19T08:29:11.266637Z"}},"outputs":[],"execution_count":null},{"id":"65a66059","cell_type":"code","source":"# ── 9. 3-channel stacked window image (what we will feed the model) ───────\ndef dicom_to_3ch(img_hu: np.ndarray, size: int = 256) -> np.ndarray:\n    \"\"\"Stack 3 windows as RGB channels, resize, return uint8 HWC image.\"\"\"\n    import cv2\n    channels = []\n    for wc, ww in WINDOWS.values():\n        ch = apply_window(img_hu, wc, ww)\n        ch_resized = cv2.resize(ch, (size, size), interpolation=cv2.INTER_AREA)\n        channels.append(ch_resized)\n    img_3ch = np.stack(channels, axis=-1)  # (H, W, 3) in [0,1]\n    return (img_3ch * 255).astype(np.uint8)\n\n\nfig, axes = plt.subplots(2, 3, figsize=(12, 8))\nfig.suptitle('3-channel stacked window images (model input preview)', fontsize=12)\n\nfor idx, img_id in enumerate(sample_ids):\n    path = get_dcm_path(img_id, TRAIN_DIR)\n    if not path.exists():\n        continue\n    img_hu, _ = load_dcm_pixel(path)\n    img_3ch   = dicom_to_3ch(img_hu, size=256)\n\n    r, c = divmod(idx, 3)\n    axes[r, c].imshow(img_3ch)\n    label_info = df[df['image_id'] == img_id][SUBTYPES].values[0]\n    label_str  = ', '.join([s for s, v in zip(SUBTYPES, label_info) if v]) or 'Neg'\n    axes[r, c].set_title(f'{img_id[-8:]}  ({label_str})', fontsize=8)\n    axes[r, c].axis('off')\n\nplt.tight_layout()\nplt.savefig('/kaggle/working/3ch_input_preview.png', bbox_inches='tight')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T08:29:42.302289Z","iopub.execute_input":"2026-02-19T08:29:42.302766Z","iopub.status.idle":"2026-02-19T08:29:45.247143Z","shell.execute_reply.started":"2026-02-19T08:29:42.302734Z","shell.execute_reply":"2026-02-19T08:29:45.246047Z"}},"outputs":[],"execution_count":null},{"id":"322d0f94","cell_type":"code","source":"# ── 10. DICOM metadata exploration ───────────────────────────────────────\nmeta_records = []\nfor img_id in random.sample(pos_ids, 50):\n    path = get_dcm_path(img_id, TRAIN_DIR)\n    if not path.exists():\n        continue\n    dcm = pydicom.dcmread(str(path), stop_before_pixels=True)\n    meta_records.append({\n        'image_id'         : img_id,\n        'Rows'             : getattr(dcm, 'Rows',             None),\n        'Columns'          : getattr(dcm, 'Columns',          None),\n        'BitsAllocated'    : getattr(dcm, 'BitsAllocated',    None),\n        'RescaleSlope'     : getattr(dcm, 'RescaleSlope',     None),\n        'RescaleIntercept' : getattr(dcm, 'RescaleIntercept', None),\n        'WindowCenter'     : getattr(dcm, 'WindowCenter',     None),\n        'WindowWidth'      : getattr(dcm, 'WindowWidth',      None),\n        'PixelSpacing'     : str(getattr(dcm, 'PixelSpacing', None)),\n    })\n\nmeta_df = pd.DataFrame(meta_records)\nprint('Sample metadata:')\ndisplay(meta_df.head())\nprint('\\nUnique resolutions:', meta_df[['Rows','Columns']].drop_duplicates().values)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T08:29:55.470687Z","iopub.execute_input":"2026-02-19T08:29:55.471582Z","iopub.status.idle":"2026-02-19T08:29:56.253092Z","shell.execute_reply.started":"2026-02-19T08:29:55.471548Z","shell.execute_reply":"2026-02-19T08:29:56.252299Z"}},"outputs":[],"execution_count":null},{"id":"260f281d","cell_type":"code","source":"# ── 10b. PatientID structure exploration ──────────────────────────────────\n# Understand how many patients and slices/patient exist in the dataset.\n# This motivates the patient-level split in NB02.\npid_records = []\nfor img_id in random.sample(pos_ids + neg_ids, min(5000, len(pos_ids + neg_ids))):\n    path = get_dcm_path(img_id, TRAIN_DIR)\n    if not path.exists():\n        continue\n    try:\n        dcm = pydicom.dcmread(str(path), stop_before_pixels=True)\n        pid = str(getattr(dcm, 'PatientID', 'UNKNOWN'))\n        pid_records.append({'image_id': img_id, 'patient_id': pid})\n    except Exception:\n        continue\n\npid_df = pd.DataFrame(pid_records)\nslices_per_patient = pid_df.groupby('patient_id').size()\n\nprint(f'Sample of {len(pid_df)} images → {pid_df[\"patient_id\"].nunique()} unique patients')\nprint(f'Slices/patient: min={slices_per_patient.min()}, '\n      f'max={slices_per_patient.max()}, '\n      f'mean={slices_per_patient.mean():.1f}, '\n      f'median={slices_per_patient.median():.0f}')\n\nfig, ax = plt.subplots(figsize=(8, 3))\nax.hist(slices_per_patient.values, bins=30, color='steelblue', edgecolor='white')\nax.set(title='Slices per patient (sample of 5000)', xlabel='Number of slices', ylabel='Patients')\nplt.tight_layout()\nplt.savefig('/kaggle/working/slices_per_patient.png', bbox_inches='tight')\nplt.show()\n\n# ── Leakage risk assessment ──────────────────────────────────────────────\nn_shared = int((slices_per_patient > 1).sum())\nn_total_patients = len(slices_per_patient)\npct_shared = n_shared / n_total_patients * 100\n\nprint(f'\\n📊 Leakage risk assessment:')\nprint(f'   Patients with >1 slice : {n_shared} / {n_total_patients} ({pct_shared:.1f}%)')\nprint(f'   Mean slices/patient    : {slices_per_patient.mean():.2f}')\nprint(f'\\n   This is a slice-level dataset (not volumetric CT studies).')\nprint(f'   Most patients have only 1 slice → leakage risk is MODEST.')\nprint(f'   Estimated AUC inflation from naive split: ~0.01–0.02 (not 0.07–0.10).')\nprint(f'\\n   ✅ We still use GroupShuffleSplit in NB02 for methodological rigor,')\nprint(f'      but the practical impact is small for this dataset.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T08:43:33.197709Z","iopub.execute_input":"2026-02-19T08:43:33.198156Z","iopub.status.idle":"2026-02-19T08:44:31.332479Z","shell.execute_reply.started":"2026-02-19T08:43:33.198124Z","shell.execute_reply":"2026-02-19T08:44:31.331343Z"}},"outputs":[],"execution_count":null},{"id":"3dd73476","cell_type":"code","source":"# ── 11. Pixel HU statistics across sample ────────────────────────────────\nstats = []\nfor img_id in random.sample(pos_ids + neg_ids, 100):\n    path = get_dcm_path(img_id, TRAIN_DIR)\n    if not path.exists():\n        continue\n    img_hu, _ = load_dcm_pixel(path)\n    stats.append({\n        'image_id' : img_id,\n        'label'    : int(df[df['image_id'] == img_id]['any'].values[0]),\n        'hu_mean'  : float(img_hu.mean()),\n        'hu_std'   : float(img_hu.std()),\n        'hu_min'   : float(img_hu.min()),\n        'hu_max'   : float(img_hu.max()),\n    })\n\nstats_df = pd.DataFrame(stats)\n\nfig, axes = plt.subplots(1, 2, figsize=(12, 4))\nfor label, grp in stats_df.groupby('label'):\n    lname = 'Hemorrhage' if label else 'Normal'\n    axes[0].hist(grp['hu_mean'], bins=30, alpha=0.6, label=lname)\n    axes[1].hist(grp['hu_std'],  bins=30, alpha=0.6, label=lname)\n\naxes[0].set_title('Mean HU per slice'); axes[0].set_xlabel('HU'); axes[0].legend()\naxes[1].set_title('Std HU per slice');  axes[1].set_xlabel('HU'); axes[1].legend()\nplt.tight_layout()\nplt.savefig('/kaggle/working/hu_statistics.png', bbox_inches='tight')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T08:45:28.071124Z","iopub.execute_input":"2026-02-19T08:45:28.071529Z","iopub.status.idle":"2026-02-19T08:45:51.252984Z","shell.execute_reply.started":"2026-02-19T08:45:28.0715Z","shell.execute_reply":"2026-02-19T08:45:51.251889Z"}},"outputs":[],"execution_count":null},{"id":"bc724755","cell_type":"code","source":"# ── 12. Augmentation preview ──────────────────────────────────────────────\nimport torchvision.transforms as T\nfrom PIL import Image\n\naug = T.Compose([\n    T.RandomHorizontalFlip(p=1.0),\n    T.RandomRotation(degrees=15),\n    T.ColorJitter(brightness=0.1, contrast=0.1),\n])\n\n# Pick one positive sample\nimg_id  = sample_pos[0]\npath    = get_dcm_path(img_id, TRAIN_DIR)\nimg_hu, _ = load_dcm_pixel(path)\nimg_3ch   = dicom_to_3ch(img_hu)\npil_img   = Image.fromarray(img_3ch)\n\nfig, axes = plt.subplots(1, 5, figsize=(16, 3.5))\nfig.suptitle(f'Augmentation preview — {img_id[-8:]} (Hemorrhage positive)', fontsize=11)\n\naxes[0].imshow(pil_img); axes[0].set_title('Original'); axes[0].axis('off')\nfor k in range(1, 5):\n    axes[k].imshow(aug(pil_img)); axes[k].set_title(f'Aug {k}'); axes[k].axis('off')\n\nplt.tight_layout()\nplt.savefig('/kaggle/working/augmentation_preview.png', bbox_inches='tight')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T08:49:25.845454Z","iopub.execute_input":"2026-02-19T08:49:25.8463Z","iopub.status.idle":"2026-02-19T08:49:34.777298Z","shell.execute_reply.started":"2026-02-19T08:49:25.846267Z","shell.execute_reply":"2026-02-19T08:49:34.776232Z"}},"outputs":[],"execution_count":null},{"id":"32ab9348","cell_type":"code","source":"# ── 13. Summary printout ──────────────────────────────────────────────────\nprint('=' * 55)\nprint('EDA SUMMARY')\nprint('=' * 55)\nprint(f'Total slices          : {totals:,}')\nprint(f'Hemorrhage positive   : {int(df[\"any\"].sum()):,}  ({df[\"any\"].mean()*100:.2f}%)')\nprint(f'Hemorrhage negative   : {int((df[\"any\"]==0).sum()):,}')\nprint()\nprint('Sub-type prevalence (positive slices only):')\nfor s in SUBTYPES[1:]:\n    c = int(df[s].sum())\n    print(f'  {s:<22}: {c:>6,}  ({c/totals*100:.2f}%)')\nprint()\nprint('Key observations:')\nprint(' - Strong class imbalance (~14% positive overall)')\nprint(' - Epidural is rarest sub-type (~0.4%)')\nprint(' - Sub-types can co-occur (mixed hemorrhage cases)')\nprint(' - All CT images are 512x512 HU arrays')\nprint(' - Brain window (WC40,WW80) best isolates hemorrhage signal')\nprint('=' * 55)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T08:50:40.814658Z","iopub.execute_input":"2026-02-19T08:50:40.815282Z","iopub.status.idle":"2026-02-19T08:50:40.868271Z","shell.execute_reply.started":"2026-02-19T08:50:40.815251Z","shell.execute_reply":"2026-02-19T08:50:40.867288Z"}},"outputs":[],"execution_count":null},{"id":"8b2e561a","cell_type":"code","source":"# ── HEALTH CHECK — automated output validation ────────────────────────────\nimport json as _json_hc\n\nerrors = []\nexpected_plots = [\n    'label_distribution.png', 'cooccurrence.png', 'windowing_comparison.png',\n    '3ch_input_preview.png', 'hu_statistics.png', 'augmentation_preview.png',\n    'slices_per_patient.png',\n]\nfor plot in expected_plots:\n    if not os.path.exists(f'/kaggle/working/{plot}'):\n        errors.append(f'Missing plot: {plot}')\n\nif totals < 1000:\n    errors.append(f'Dataset too small: only {totals} images loaded')\n\nhealth = {\n    'notebook': '01_eda',\n    'status'  : 'PASS' if not errors else 'FAIL',\n    'errors'  : errors,\n    'n_images': totals,\n    'n_positive': int(df['any'].sum()),\n    'n_subtypes': len(SUBTYPES),\n    'plots_saved': [p for p in expected_plots\n                    if os.path.exists(f'/kaggle/working/{p}')],\n}\n\nwith open('/kaggle/working/health_check_nb01.json', 'w') as f:\n    _json_hc.dump(health, f, indent=2)\n\nif errors:\n    print('❌ HEALTH CHECK FAILED:')\n    for e in errors:\n        print(f'   • {e}')\nelse:\n    print('✅ HEALTH CHECK PASSED')\n    print(f'   {len(health[\"plots_saved\"])} plots saved')\n    print(f'   {totals:,} images explored')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T08:51:03.862423Z","iopub.execute_input":"2026-02-19T08:51:03.862878Z","iopub.status.idle":"2026-02-19T08:51:03.874515Z","shell.execute_reply.started":"2026-02-19T08:51:03.862839Z","shell.execute_reply":"2026-02-19T08:51:03.873475Z"}},"outputs":[],"execution_count":null},{"id":"99fad9df-e119-4a86-80a9-7e6efa131ac4","cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}