{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":13451,"datasetId":654585,"databundleVersionId":1188070}],"dockerImageVersionId":31286,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"3e1107a9","cell_type":"markdown","source":"# Notebook 2 — DICOM Preprocessing & Cache\n**RSNA Intracranial Hemorrhage Detection**\n\nThis notebook converts raw DICOM files into 3-channel windowed **NPY arrays**\n(uint8, 0–255 range, shape 256×256×3) and saves them as a Kaggle output artifact.\n\n**Subsequent notebooks add this notebook's output as an input dataset.**\n\n### Critical workflow\n1. **Pilot first**: Set `SUBSET_FRAC = 0.1` to validate pipeline on ~75K images\n2. Once pilot passes health check, set `SUBSET_FRAC = 1.0` and re-run\n3. Run via **Save Version → Save & Run All** (background commit)\n4. In the next notebook, click **Add Data → Notebook Output Files** and add this run\n\n### Key design decisions\n- **NPY uint8 format**: Compact arrays (~0.19 MB each), fast I/O, no BGR/RGB\n  confusion. Windowing maps HU to [0,1] then scales to [0,255] — 256 levels is\n  more than sufficient for CNN input (standard practice).\n- **Patient-level split**: `GroupShuffleSplit` on PatientID from DICOM headers.\n  Most patients have only 1–2 slices (mean ≈ 1.2), so leakage risk is modest\n  (~0.01–0.02 AUC), but group-split is used for methodological rigor.\n- **Normalization stats**: Dataset-specific mean/std computed on cached arrays\n  (divided by 255 to match `ToTensor()` output scale).\n- **Checkpointing**: Progress saved every 10K images — session crash resumes\n  from last checkpoint, not from zero.\n- **Robust metadata**: WindowCenter/Width handled as scalar or list (DICOM quirk).\n\n### Disk budget\n- 75K images × ~0.19 MB ≈ **14 GB** (pilot) — fits Kaggle's 19.5 GB limit\n- Full 750K images would need ~143 GB — requires chunked commits or on-the-fly loading","metadata":{}},{"id":"634ed992","cell_type":"code","source":"# ── 0. Config ──────────────────────────────────────────────────────────────\nimport os, gc, glob, random\nfrom pathlib import Path\n\n# ─── Kaggle paths ────────────────────────────────────────────────────────\nBASE      = Path('/kaggle/input/competitions/rsna-intracranial-hemorrhage-detection/rsna-intracranial-hemorrhage-detection')\nTRAIN_CSV = BASE / 'stage_2_train.csv'\nTRAIN_DIR = BASE / 'stage_2_train/'\n\n# ─── Output paths ────────────────────────────────────────────────────────\nOUT_DIR   = Path('/kaggle/working/cache')\nOUT_DIR.mkdir(parents=True, exist_ok=True)\n\n# ─── Processing config ───────────────────────────────────────────────────\nIMG_SIZE     = 256        # resize target (pixels)\nSUBSET_FRAC  = 0.1        # ⚠️ START WITH 0.1 (pilot ~75K). Set 1.0 for full run.\nVAL_FRAC     = 0.15       # 15% validation\nSEED         = 42\nNUM_WORKERS  = 4          # parallel DICOM workers\nCHECKPOINT_EVERY = 10000  # save progress index every N images\n\n# CT windows: (window_center, window_width)\nWINDOWS = [\n    (40,   80),    # brain\n    (75,  215),    # subdural\n    (40,  380),    # soft tissue\n]\n\nprint(f'Config OK. Output dir: {OUT_DIR}')\nprint(f'Mode: {\"PILOT (10%)\" if SUBSET_FRAC < 1.0 else \"FULL DATASET\"}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T09:58:45.318414Z","iopub.execute_input":"2026-02-19T09:58:45.319259Z","iopub.status.idle":"2026-02-19T09:58:45.332637Z","shell.execute_reply.started":"2026-02-19T09:58:45.319224Z","shell.execute_reply":"2026-02-19T09:58:45.331573Z"}},"outputs":[],"execution_count":null},{"id":"d9070461","cell_type":"code","source":"# ── 1. Load and prepare label dataframe ──────────────────────────────────\nimport numpy as np\nimport pandas as pd\n\nraw = pd.read_csv(TRAIN_CSV)\nraw[['image_id', 'subtype']] = raw['ID'].str.rsplit('_', n=1, expand=True)\n\n# Drop duplicate (image_id, subtype) rows present in the RSNA CSV\nraw = raw.drop_duplicates(subset=['image_id', 'subtype'], keep='first')\n\nSUBTYPES = ['any', 'epidural', 'intraparenchymal',\n            'intraventricular', 'subarachnoid', 'subdural']\n\ndf = raw.pivot(index='image_id', columns='subtype', values='Label').reset_index()\ndf.columns.name = None\nfor col in SUBTYPES:\n    df[col] = df[col].astype(int)\n\n# Subsample if requested — stratified on 'any' to preserve class balance\nif SUBSET_FRAC < 1.0:\n    from sklearn.model_selection import train_test_split\n    df, _ = train_test_split(df, train_size=SUBSET_FRAC,\n                             stratify=df['any'], random_state=SEED)\n    df = df.reset_index(drop=True)\n    print(f'Using {len(df):,} images ({SUBSET_FRAC*100:.0f}% stratified subset)')\nelse:\n    print(f'Using full dataset: {len(df):,} images')\n\nprint(f'Positive rate: {df[\"any\"].mean()*100:.2f}%')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T09:58:49.289611Z","iopub.execute_input":"2026-02-19T09:58:49.289949Z","iopub.status.idle":"2026-02-19T09:59:13.015172Z","shell.execute_reply.started":"2026-02-19T09:58:49.289921Z","shell.execute_reply":"2026-02-19T09:59:13.01417Z"}},"outputs":[],"execution_count":null},{"id":"059d2aee","cell_type":"code","source":"# ── 2. Extract PatientID from DICOM headers ──────────────────────────────\n# We need PatientID for patient-level train/val split (GroupShuffleSplit).\n# Note: Most patients in this dataset have only 1-2 slices (mean ≈ 1.2),\n# so leakage risk is modest, but we still group-split for rigor.\nimport pydicom\nfrom concurrent.futures import ProcessPoolExecutor, as_completed\nfrom tqdm.auto import tqdm\n\n\ndef _extract_patient_id(image_id):\n    \"\"\"Read DICOM header (no pixels) to get PatientID. Fast (~1ms each).\"\"\"\n    dcm_path = str(TRAIN_DIR / f'{image_id}.dcm')\n    try:\n        dcm = pydicom.dcmread(dcm_path, stop_before_pixels=True)\n        pid = str(getattr(dcm, 'PatientID', 'UNKNOWN'))\n        return image_id, pid\n    except Exception:\n        return image_id, 'UNKNOWN'\n\n\nall_ids = df['image_id'].tolist()\npatient_map = {}\n\nprint(f'Extracting PatientID from {len(all_ids):,} DICOM headers...')\nprint(f'(metadata-only read — no pixel data loaded, ~10-20 min for full dataset)')\n\nwith ProcessPoolExecutor(max_workers=NUM_WORKERS) as pool:\n    futures = {pool.submit(_extract_patient_id, img_id): img_id\n               for img_id in all_ids}\n    with tqdm(total=len(all_ids), unit='dcm', desc='PatientID scan') as pbar:\n        for future in as_completed(futures):\n            img_id, pid = future.result()\n            patient_map[img_id] = pid\n            pbar.update(1)\n\ndf['patient_id'] = df['image_id'].map(patient_map)\nn_patients = df['patient_id'].nunique()\nn_unknown  = (df['patient_id'] == 'UNKNOWN').sum()\n\nprint(f'\\nUnique patients  : {n_patients:,}')\nprint(f'Slices / patient : {len(df) / n_patients:.1f} (mean)')\nprint(f'Unknown PIDs     : {n_unknown} (will be assigned to training set)')\n\nif n_unknown > 0:\n    print(f'  WARNING: {n_unknown} images had no PatientID in DICOM header.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T09:59:13.016917Z","iopub.execute_input":"2026-02-19T09:59:13.017418Z","iopub.status.idle":"2026-02-19T10:03:55.332394Z","shell.execute_reply.started":"2026-02-19T09:59:13.01739Z","shell.execute_reply":"2026-02-19T10:03:55.330681Z"}},"outputs":[],"execution_count":null},{"id":"99c42f42","cell_type":"code","source":"# ── 3. Patient-level train/val split (GroupShuffleSplit) ──────────────────\n# GroupShuffleSplit ensures ALL slices from a given patient stay in the same fold.\n# Note: In this dataset most patients have only 1-2 slices (mean ≈ 1.2),\n# so practical leakage from a naive split would be modest (~0.01-0.02 AUC).\n# We still use patient-level splitting for methodological correctness.\nfrom sklearn.model_selection import GroupShuffleSplit\n\ngss = GroupShuffleSplit(n_splits=1, test_size=VAL_FRAC, random_state=SEED)\ngroups = df['patient_id'].values\n\ntrain_idx, val_idx = next(gss.split(df, y=df['any'], groups=groups))\n\ntrain_df = df.iloc[train_idx].copy(); train_df['split'] = 'train'\nval_df   = df.iloc[val_idx].copy();   val_df['split']   = 'val'\ndf_split = pd.concat([train_df, val_df], ignore_index=True)\n\n# Verify no patient leakage\ntrain_p = set(train_df['patient_id'])\nval_p   = set(val_df['patient_id'])\nleaked  = train_p & val_p\nassert len(leaked) == 0, f'PATIENT LEAKAGE: {len(leaked)} patients in both splits!'\n\nprint(f'Train: {len(train_df):,} images  |  Val: {len(val_df):,} images')\nprint(f'Train patients: {len(train_p):,}  |  Val patients: {len(val_p):,}')\nprint(f'Patient leakage : {len(leaked)} (MUST be 0) ✓')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T10:03:55.335168Z","iopub.execute_input":"2026-02-19T10:03:55.335637Z","iopub.status.idle":"2026-02-19T10:03:55.604665Z","shell.execute_reply.started":"2026-02-19T10:03:55.33558Z","shell.execute_reply":"2026-02-19T10:03:55.60354Z"}},"outputs":[],"execution_count":null},{"id":"12688ba4","cell_type":"code","source":"# ── 4. Core DICOM → NPY conversion function ──────────────────────────────\nimport cv2\nimport pydicom\n\n\ndef _to_scalar(val):\n    \"\"\"Safely extract scalar from DICOM field (may be list or MultiValue).\"\"\"\n    if isinstance(val, (list, pydicom.multival.MultiValue)):\n        return float(val[0])\n    return float(val)\n\n\ndef apply_window(img_hu: np.ndarray, wc: float, ww: float) -> np.ndarray:\n    lo = wc - ww / 2\n    hi = wc + ww / 2\n    return np.clip((img_hu - lo) / (hi - lo), 0.0, 1.0)\n\n\ndef dicom_to_npy(image_id: str,\n                 dcm_dir: Path,\n                 out_dir: Path,\n                 size: int = 256) -> bool:\n    \"\"\"\n    Read one DICOM, apply 3 CT windows, stack as (H, W, 3) uint8 [0,255],\n    resize, save as .npy. Returns True on success.\n    \"\"\"\n    out_path = out_dir / f'{image_id}.npy'\n    if out_path.exists():          # skip already-processed files (resume-safe)\n        return True\n\n    dcm_path = dcm_dir / f'{image_id}.dcm'\n    if not dcm_path.exists():\n        return False\n\n    try:\n        dcm = pydicom.dcmread(str(dcm_path))\n        img = dcm.pixel_array.astype(np.float32)\n\n        # Robust rescale — handle scalar or list DICOM fields\n        slope = _to_scalar(getattr(dcm, 'RescaleSlope', 1))\n        inter = _to_scalar(getattr(dcm, 'RescaleIntercept', 0))\n        img   = img * slope + inter          # → Hounsfield Units\n\n        channels = []\n        for wc, ww in WINDOWS:\n            ch = apply_window(img, wc, ww)\n            ch = cv2.resize(ch, (size, size), interpolation=cv2.INTER_AREA)\n            channels.append(ch)\n\n        img_3ch = (np.stack(channels, axis=-1) * 255).astype(np.uint8)  # (H,W,3) in [0,255]\n        np.save(str(out_path), img_3ch)\n        return True\n    except Exception as e:\n        print(f'  ERROR {image_id}: {e}')\n        return False\n\n\nprint('Conversion function defined (NPY uint8, compact).')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T10:03:55.607312Z","iopub.execute_input":"2026-02-19T10:03:55.607573Z","iopub.status.idle":"2026-02-19T10:03:56.115865Z","shell.execute_reply.started":"2026-02-19T10:03:55.607549Z","shell.execute_reply":"2026-02-19T10:03:56.114895Z"}},"outputs":[],"execution_count":null},{"id":"6201567f","cell_type":"code","source":"# ── 5. Parallel conversion with checkpointing ────────────────────────────\nfrom concurrent.futures import ProcessPoolExecutor, as_completed\nfrom tqdm.auto import tqdm\nimport json as _json_ckpt\n\nPROGRESS_FILE = Path('/kaggle/working/cache_progress.json')\n\ndef _worker(image_id):\n    \"\"\"Top-level function required for ProcessPoolExecutor pickling.\"\"\"\n    return image_id, dicom_to_npy(image_id, TRAIN_DIR, OUT_DIR, IMG_SIZE)\n\n\nall_ids = df_split['image_id'].tolist()\n\n# Resume: skip IDs that already have cached .npy files\nalready_done = {p.stem for p in OUT_DIR.glob('*.npy')}\nremaining_ids = [img_id for img_id in all_ids if img_id not in already_done]\n\nprint(f'Total: {len(all_ids):,}  |  Already cached: {len(already_done):,}  |  Remaining: {len(remaining_ids):,}')\nprint(f'Workers: {NUM_WORKERS}  |  Output size: {IMG_SIZE}x{IMG_SIZE}  |  Format: NPY uint8')\nprint(f'Checkpoint every: {CHECKPOINT_EVERY:,} images')\n\nfailed_ids = []\nprocessed  = len(already_done)\n\nwith ProcessPoolExecutor(max_workers=NUM_WORKERS) as pool:\n    futures = {pool.submit(_worker, img_id): img_id for img_id in remaining_ids}\n    with tqdm(total=len(remaining_ids), unit='img', initial=0) as pbar:\n        for future in as_completed(futures):\n            img_id, ok = future.result()\n            if not ok:\n                failed_ids.append(img_id)\n            processed += 1\n            pbar.update(1)\n\n            # Periodic checkpoint\n            if processed % CHECKPOINT_EVERY == 0:\n                ckpt = {'processed': processed, 'failed': len(failed_ids)}\n                with open(PROGRESS_FILE, 'w') as f:\n                    _json_ckpt.dump(ckpt, f)\n\n# Final checkpoint\nckpt = {'processed': processed, 'failed': len(failed_ids), 'done': True}\nwith open(PROGRESS_FILE, 'w') as f:\n    _json_ckpt.dump(ckpt, f)\n\nprint(f'\\nDone. Processed: {processed:,}  |  Failed: {len(failed_ids)}')\nif failed_ids:\n    print('  Failed IDs (first 10):', failed_ids[:10])\n\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T10:03:56.116967Z","iopub.execute_input":"2026-02-19T10:03:56.117385Z","iopub.status.idle":"2026-02-19T10:09:21.306741Z","shell.execute_reply.started":"2026-02-19T10:03:56.117356Z","shell.execute_reply":"2026-02-19T10:09:21.305542Z"}},"outputs":[],"execution_count":null},{"id":"6d433141","cell_type":"code","source":"# ── 6. Build final manifest CSV ───────────────────────────────────────────\n# The manifest maps image_id → cached NPY path + label columns + split + patient_id\n\ndf_split['npy_path'] = df_split['image_id'].apply(\n    lambda x: str(OUT_DIR / f'{x}.npy')\n)\n\n# Drop rows where NPY wasn't created (e.g. missing/corrupted DICOM)\ndf_split['npy_exists'] = df_split['npy_path'].apply(os.path.exists)\nmissing = (~df_split['npy_exists']).sum()\nif missing:\n    print(f'Dropping {missing} rows with missing NPY file')\ndf_split = df_split[df_split['npy_exists']].drop(columns='npy_exists').reset_index(drop=True)\n\n# Verify patient-level integrity after dropping\ntrain_pids_final = set(df_split[df_split['split'] == 'train']['patient_id'])\nval_pids_final   = set(df_split[df_split['split'] == 'val']['patient_id'])\nassert len(train_pids_final & val_pids_final) == 0, 'Patient leakage after dropping missing!'\n\n# Save manifest (now includes patient_id column)\nmanifest_path = Path('/kaggle/working/manifest.csv')\ndf_split.to_csv(manifest_path, index=False)\n\nprint(f'Manifest saved: {manifest_path}')\nprint(f'Rows: {len(df_split):,}')\nprint(f'Columns: {list(df_split.columns)}')\ndf_split.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T10:09:21.308167Z","iopub.execute_input":"2026-02-19T10:09:21.30893Z","iopub.status.idle":"2026-02-19T10:09:22.768414Z","shell.execute_reply.started":"2026-02-19T10:09:21.308897Z","shell.execute_reply":"2026-02-19T10:09:22.767172Z"}},"outputs":[],"execution_count":null},{"id":"e45d7b5c","cell_type":"code","source":"# ── 7. Post-cache verification ─────────────────────────────────────────────\n# Verify random cached files: shape, dtype, value range, channel integrity.\nimport matplotlib.pyplot as plt\n\nN_VERIFY = 20\nverify_sample = df_split.sample(min(N_VERIFY, len(df_split)), random_state=SEED)\n\nshapes_ok, dtype_ok, range_ok = 0, 0, 0\nbad_files = []\n\nfor _, row in verify_sample.iterrows():\n    arr = np.load(row['npy_path'])\n    s_ok = (arr.shape == (IMG_SIZE, IMG_SIZE, 3))\n    d_ok = (arr.dtype == np.uint8)\n    r_ok = (arr.min() >= 0 and arr.max() <= 255)\n    shapes_ok += s_ok\n    dtype_ok  += d_ok\n    range_ok  += r_ok\n    if not (s_ok and d_ok and r_ok):\n        bad_files.append((row['image_id'], arr.shape, arr.dtype, arr.min(), arr.max()))\n\nprint(f'Verified {len(verify_sample)} random cached files:')\nprint(f'  Shape  ({IMG_SIZE},{IMG_SIZE},3) : {shapes_ok}/{len(verify_sample)}')\nprint(f'  Dtype  uint8            : {dtype_ok}/{len(verify_sample)}')\nprint(f'  Range  [0, 255]         : {range_ok}/{len(verify_sample)}')\nif bad_files:\n    print(f'\\n  ⚠️ Bad files:')\n    for bf in bad_files:\n        print(f'    {bf}')\nelse:\n    print('  ✅ All files valid')\n\n# Visual sanity check — show 3 positive samples\nvis_samples = df_split[df_split['any'] == 1].sample(3, random_state=SEED)\nfig, axes = plt.subplots(1, 3, figsize=(10, 4))\nfor ax, (_, row) in zip(axes, vis_samples.iterrows()):\n    img = np.load(row['npy_path'])  # (H,W,3) uint8 [0,255]\n    ax.imshow(img)\n    subtypes_present = [s for s in SUBTYPES if row[s]]\n    ax.set_title('  '.join(subtypes_present), fontsize=8)\n    ax.axis('off')\nplt.suptitle('Cached NPY sanity check (positive samples)', fontsize=11)\nplt.tight_layout()\nplt.savefig('/kaggle/working/cache_sanity_check.png', bbox_inches='tight')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T10:09:22.769884Z","iopub.execute_input":"2026-02-19T10:09:22.770487Z","iopub.status.idle":"2026-02-19T10:09:23.857033Z","shell.execute_reply.started":"2026-02-19T10:09:22.770458Z","shell.execute_reply":"2026-02-19T10:09:23.855906Z"}},"outputs":[],"execution_count":null},{"id":"519b485f","cell_type":"code","source":"# ── 8. Compute dataset-specific normalization stats ───────────────────────\n# Compute actual mean/std on our 3-channel windowed NPY arrays.\n# Arrays are uint8 [0,255] — we divide by 255 to match ToTensor() output scale.\nimport json as _json\n\nNORM_SAMPLE_SIZE = 5000\nnorm_sample = df_split.sample(min(NORM_SAMPLE_SIZE, len(df_split)),\n                               random_state=SEED)\n\nchannel_sums    = np.zeros(3, dtype=np.float64)\nchannel_sq_sums = np.zeros(3, dtype=np.float64)\nn_pixels = 0\n\nfor _, row in tqdm(norm_sample.iterrows(), total=len(norm_sample),\n                   desc='Computing normalization stats'):\n    img = np.load(row['npy_path']).astype(np.float64) / 255.0  # (H,W,3) → [0,1]\n    channel_sums    += img.sum(axis=(0, 1))\n    channel_sq_sums += (img ** 2).sum(axis=(0, 1))\n    n_pixels += img.shape[0] * img.shape[1]\n\ndataset_mean = (channel_sums / n_pixels).tolist()\ndataset_std  = (np.sqrt(channel_sq_sums / n_pixels - (channel_sums / n_pixels) ** 2)).tolist()\n\nnorm_stats = {\n    'mean':        [round(m, 6) for m in dataset_mean],\n    'std':         [round(s, 6) for s in dataset_std],\n    'n_images':    len(norm_sample),\n    'n_pixels':    int(n_pixels),\n    'imagenet_mean': [0.485, 0.456, 0.406],\n    'imagenet_std':  [0.229, 0.224, 0.225],\n    'note': 'Use dataset-specific stats for best results. '\n            'ImageNet stats are provided as fallback for comparison.',\n}\n\nnorm_path = Path('/kaggle/working/normalization_stats.json')\nwith open(norm_path, 'w') as f:\n    _json.dump(norm_stats, f, indent=2)\n\nprint(f'\\nDataset mean: {norm_stats[\"mean\"]}')\nprint(f'Dataset std : {norm_stats[\"std\"]}')\nprint(f'ImageNet mean: {norm_stats[\"imagenet_mean\"]}')\nprint(f'ImageNet std : {norm_stats[\"imagenet_std\"]}')\nprint(f'Saved to: {norm_path}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T10:09:23.85848Z","iopub.execute_input":"2026-02-19T10:09:23.858859Z","iopub.status.idle":"2026-02-19T10:09:53.318523Z","shell.execute_reply.started":"2026-02-19T10:09:23.858832Z","shell.execute_reply":"2026-02-19T10:09:53.317453Z"}},"outputs":[],"execution_count":null},{"id":"31eaae0b","cell_type":"code","source":"# ── 9. Report disk usage ──────────────────────────────────────────────────\nimport shutil\n\ntotal_bytes, used_bytes, free_bytes = shutil.disk_usage('/kaggle/working')\nn_npys = len(list(OUT_DIR.glob('*.npy')))\ncache_bytes = sum(f.stat().st_size for f in OUT_DIR.glob('*.npy'))\n\nprint('─' * 40)\nprint(f'NPY files created : {n_npys:,}')\nprint(f'Cache size        : {cache_bytes/1e9:.2f} GB')\nprint(f'Working dir used  : {used_bytes/1e9:.2f} GB')\nprint(f'Working dir free  : {free_bytes/1e9:.2f} GB')\nprint('─' * 40)\nprint()\nif SUBSET_FRAC < 1.0:\n    print('⚠️  PILOT MODE — you processed a subset.')\n    print('    If health check passes, set SUBSET_FRAC = 1.0 and re-run.')\n    print('    Already-cached files will be skipped (resume-safe).')\nelse:\n    print('NEXT STEPS:')\n    print('1. Click \"Save Version\" → \"Save & Run All\" to commit this output')\n    print('2. In Notebook 03, click \"Add Data\" → \"Notebook Output Files\"')\n    print('   and search for this notebook to add the NPY cache as input')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T10:09:53.319886Z","iopub.execute_input":"2026-02-19T10:09:53.320641Z","iopub.status.idle":"2026-02-19T10:09:54.229049Z","shell.execute_reply.started":"2026-02-19T10:09:53.320573Z","shell.execute_reply":"2026-02-19T10:09:54.228148Z"}},"outputs":[],"execution_count":null},{"id":"55c2a08d","cell_type":"code","source":"# ── 10. HEALTH CHECK — automated output validation ───────────────────────\nimport json as _json_hc\n\nerrors = []\n\n# Check manifest\nmanifest_hc = Path('/kaggle/working/manifest.csv')\nif not manifest_hc.exists():\n    errors.append('manifest.csv is MISSING')\nelse:\n    mf = pd.read_csv(manifest_hc)\n    required_cols = ['image_id', 'patient_id', 'split', 'any', 'npy_path']\n    for col in required_cols:\n        if col not in mf.columns:\n            errors.append(f'manifest.csv missing column: {col}')\n    if 'patient_id' in mf.columns and 'split' in mf.columns:\n        train_p = set(mf[mf['split'] == 'train']['patient_id'])\n        val_p   = set(mf[mf['split'] == 'val']['patient_id'])\n        if len(train_p & val_p) > 0:\n            errors.append(f'PATIENT LEAKAGE: {len(train_p & val_p)} patients in both splits')\n    # Adjust threshold for pilot mode\n    min_rows = 1000 if SUBSET_FRAC < 1.0 else 100000\n    if len(mf) < min_rows:\n        errors.append(f'manifest.csv has only {len(mf)} rows — expected {min_rows}+')\n\n# Check normalization stats\nnorm_hc = Path('/kaggle/working/normalization_stats.json')\nif not norm_hc.exists():\n    errors.append('normalization_stats.json is MISSING')\nelse:\n    ns = _json_hc.load(open(norm_hc))\n    if 'mean' not in ns or 'std' not in ns:\n        errors.append('normalization_stats.json missing mean/std')\n\n# Check NPY cache\nnpy_count = len(list(OUT_DIR.glob('*.npy')))\nif npy_count == 0:\n    errors.append('No NPY files found in cache directory')\n\n# Spot-check one file for shape/dtype/range\nspot = list(OUT_DIR.glob('*.npy'))[:1]\nif spot:\n    arr = np.load(str(spot[0]))\n    if arr.shape != (IMG_SIZE, IMG_SIZE, 3):\n        errors.append(f'NPY shape mismatch: {arr.shape} != ({IMG_SIZE},{IMG_SIZE},3)')\n    if arr.dtype != np.uint8:\n        errors.append(f'NPY dtype mismatch: {arr.dtype} != uint8')\n    if arr.min() < 0 or arr.max() > 255:\n        errors.append(f'NPY value range invalid: [{arr.min()}, {arr.max()}]')\n\n# Save health check result\nhealth = {\n    'notebook'        : '02_preprocess_cache',\n    'status'          : 'PASS' if not errors else 'FAIL',\n    'errors'          : errors,\n    'mode'            : 'pilot' if SUBSET_FRAC < 1.0 else 'full',\n    'manifest_rows'   : len(mf) if manifest_hc.exists() else 0,\n    'npy_count'       : npy_count,\n    'patient_leakage' : False,\n}\n\nwith open('/kaggle/working/health_check_nb02.json', 'w') as f:\n    _json_hc.dump(health, f, indent=2)\n\nif errors:\n    print('❌ HEALTH CHECK FAILED:')\n    for e in errors:\n        print(f'   • {e}')\n    raise RuntimeError(f'NB02 health check failed with {len(errors)} error(s)')\nelse:\n    print('✅ HEALTH CHECK PASSED')\n    print(f'   Mode       : {\"PILOT\" if SUBSET_FRAC < 1.0 else \"FULL\"}')\n    print(f'   Manifest   : {health[\"manifest_rows\"]:,} rows')\n    print(f'   NPY cache  : {health[\"npy_count\"]:,} files')\n    print(f'   Patient leak: None')\n    print(f'   Norm stats  : saved')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T10:09:54.231531Z","iopub.execute_input":"2026-02-19T10:09:54.231907Z","iopub.status.idle":"2026-02-19T10:09:55.900779Z","shell.execute_reply.started":"2026-02-19T10:09:54.231879Z","shell.execute_reply":"2026-02-19T10:09:55.899913Z"}},"outputs":[],"execution_count":null},{"id":"7c0de326-4fe4-43c1-9808-cd88f35e67f5","cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}