{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":13451,"datasetId":654585,"databundleVersionId":1188070},{"sourceType":"kernelVersion","sourceId":298700524},{"sourceType":"kernelVersion","sourceId":298979112},{"sourceType":"kernelVersion","sourceId":298986995}],"dockerImageVersionId":31286,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"479c41e5","cell_type":"markdown","source":"# Notebook 7 — Clinical Report Generator & Final Analysis\n**RSNA Intracranial Hemorrhage Detection**\n\nThis notebook ties together all pipeline components to generate structured\nscreening reports for individual CT brain images, and provides the final\nanalysis sections required for a complete project deliverable.\n\n### Report schema (fixed fields, locked phrases)\nEach report contains:\n- **Screening outcome** — fixed vocabulary, no diagnostic claim\n- **Calibrated confidence** — numeric probability + triage band\n- **Recommended action** — one of three fixed phrases\n- **Grad-CAM overlay** — visual evidence with caveats\n- **Disclaimer** — regulatory/safety language\n\n### Final analysis sections\n- **Patient-level aggregation** — aggregate slice predictions to study/patient level\n- **Training stability summary** — convergence evidence across final epochs\n- **Capacity ceiling reflection** — architectural limits and future directions\n- **Deployment decision** — final model selection and configuration\n\n### Required inputs\n- NB02 output: `manifest.csv` + `cache/` NPY arrays\n- NB03 output: `best_model.pth`, `checkpoint.pth`, `training_metrics.csv`\n- NB06 output: `calibration_params.json`\n\n> The report generator is deterministic — same image always produces the same report.","metadata":{}},{"id":"4806e383","cell_type":"code","source":"# ── Config ────────────────────────────────────────────────────────────────\nimport os, json, base64, datetime, uuid, glob as _glob_mod\nfrom io import BytesIO\nfrom pathlib import Path\nimport numpy as np\nimport pandas as pd\nimport cv2                                  # still needed for cv2.resize in overlay + cv2.imwrite\nimport torch\nimport torch.nn as nn\nimport torchvision.models as models\nimport torchvision.transforms as T\nimport matplotlib.pyplot as plt\nimport matplotlib.cm as cm\nimport seaborn as sns\nfrom sklearn.metrics import (\n    roc_auc_score, roc_curve, confusion_matrix,\n    precision_recall_curve, brier_score_loss\n)\nfrom IPython.display import display, HTML\n\nDEVICE = 'cuda' if torch.cuda.is_available() else 'cpu'\n\n# ── Input paths ──────────────────────────────────────────────────────────\nCACHE_INPUT_DIR   = '/kaggle/input/notebooks/harshitghosh/nb02eda'\nNPY_CACHE_DIR     = f'{CACHE_INPUT_DIR}/cache'\nMANIFEST_PATH     = f'{CACHE_INPUT_DIR}/manifest.csv'\nMODEL_PATH        = '/kaggle/input/notebooks/harshitghosh/03nbeda/best_model.pth'\nCHECKPOINT_PATH   = '/kaggle/input/notebooks/harshitghosh/03nbeda/checkpoint.pth'\nMETRICS_LOG_PATH  = '/kaggle/input/notebooks/harshitghosh/03nbeda/training_metrics.csv'\nCALIB_PARAMS_PATH = '/kaggle/input/notebooks/harshitghosh/06-calibration/calibration_params.json'\n\nARCH        = 'efficientnet_b0'\nIMG_SIZE    = 256\nSEED        = 42\nN_REPORTS   = 10      # number of sample reports to generate and display\n\nREPORTS_DIR = Path('/kaggle/working/reports')\nREPORTS_DIR.mkdir(parents=True, exist_ok=True)\n\n# ─── Load normalization stats ────────────────────────────────────────────\n_norm_path = os.path.join(CACHE_INPUT_DIR, 'normalization_stats.json')\nif os.path.exists(_norm_path):\n    with open(_norm_path) as f:\n        _norm = json.load(f)\n    MEAN = _norm['mean']\n    STD  = _norm['std']\n    print(f'Dataset normalization: mean={MEAN}, std={STD}')\nelse:\n    MEAN = [0.485, 0.456, 0.406]\n    STD  = [0.229, 0.224, 0.225]\n    print(f'Using ImageNet defaults: mean={MEAN}, std={STD}')\n\nprint(f'Device: {DEVICE}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-21T04:26:35.092658Z","iopub.execute_input":"2026-02-21T04:26:35.0934Z","iopub.status.idle":"2026-02-21T04:26:44.893935Z","shell.execute_reply.started":"2026-02-21T04:26:35.093368Z","shell.execute_reply":"2026-02-21T04:26:44.893154Z"}},"outputs":[],"execution_count":null},{"id":"2121448a","cell_type":"code","source":"# ── 1. Load calibration parameters ───────────────────────────────────────\nwith open(CALIB_PARAMS_PATH) as f:\n    calib = json.load(f)\n\nTEMPERATURE     = calib['temperature']\n# Use calibrated threshold (re-optimised after temp scaling) if available,\n# otherwise fall back to the raw Youden threshold\nBASE_THRESHOLD  = calib.get('calibrated_threshold', calib['base_threshold'])\nHIGH_THRESHOLD  = calib['high_threshold']\nLOW_THRESHOLD   = calib['low_threshold']\nCALIB_METHOD    = calib.get('method', 'temperature')\n\nraw_ece = calib.get('raw_ece', None)\ncal_ece = calib.get('cal_ece', None)\nhc_ece_raw = calib.get('hc_ece_raw', None)\nhc_ece_cal = calib.get('hc_ece_cal', None)\n\nprint('Calibration parameters:')\nprint(json.dumps(calib, indent=2))\nprint(f'\\nDeployment calibrator : {CALIB_METHOD}')\nprint(f'Decision threshold    : {BASE_THRESHOLD:.4f} (calibrated)')\n\n# ── Calibration honest assessment ────────────────────────────────────────\nprint(f'\\n{\"─\" * 55}')\nprint('CALIBRATION EFFECT SUMMARY')\nprint(f'{\"─\" * 55}')\nif raw_ece is not None and cal_ece is not None:\n    ece_delta = cal_ece - raw_ece\n    direction = 'WORSENED' if ece_delta > 0 else 'IMPROVED'\n    print(f'  Global ECE     : {raw_ece:.5f} → {cal_ece:.5f} ({direction}, Δ={ece_delta:+.5f})')\nif hc_ece_raw is not None and hc_ece_cal is not None:\n    hc_delta = hc_ece_cal - hc_ece_raw\n    hc_dir = 'IMPROVED' if hc_delta < 0 else 'WORSENED'\n    print(f'  HC-band ECE    : {hc_ece_raw:.5f} → {hc_ece_cal:.5f} ({hc_dir}, Δ={hc_delta:+.5f})')\nraw_brier = calib.get('raw_brier', None)\ncal_brier = calib.get('cal_brier', None)\nif raw_brier is not None and cal_brier is not None:\n    brier_delta = cal_brier - raw_brier\n    brier_dir = 'IMPROVED' if brier_delta < 0 else 'WORSENED'\n    print(f'  Brier score    : {raw_brier:.5f} → {cal_brier:.5f} ({brier_dir}, Δ={brier_delta:+.5f})')\nprint(f'\\n  Assessment:')\nprint(f'    Temperature scaling (T={TEMPERATURE:.4f}) WORSENED global ECE slightly')\nprint(f'    (+0.026) but IMPROVED high-confidence band ECE by {abs(hc_ece_cal - hc_ece_raw):.3f}')\nprint(f'    and Brier score by {abs(cal_brier - raw_brier):.4f}.')\nprint(f'    For triage deployment, high-confidence calibration is the critical')\nprint(f'    metric — patients routed to URGENT review must have reliable')\nprint(f'    probability estimates. Global ECE regression is acceptable.')\nprint(f'{\"─\" * 55}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-21T04:26:44.895451Z","iopub.execute_input":"2026-02-21T04:26:44.895912Z","iopub.status.idle":"2026-02-21T04:26:44.912538Z","shell.execute_reply.started":"2026-02-21T04:26:44.895887Z","shell.execute_reply":"2026-02-21T04:26:44.91197Z"}},"outputs":[],"execution_count":null},{"id":"bfbedc30","cell_type":"code","source":"# ── 2. Load model ─────────────────────────────────────────────────────────\ndef build_model(arch):\n    if arch == 'efficientnet_b0':\n        m = models.efficientnet_b0(weights=None)\n        m.classifier = nn.Sequential(nn.Dropout(0.3), nn.Linear(m.classifier[1].in_features, 1))\n    elif arch == 'resnet50':\n        m = models.resnet50(weights=None)\n        m.fc = nn.Sequential(nn.Dropout(0.3), nn.Linear(m.fc.in_features, 1))\n    return m\n\n\nmodel = build_model(ARCH)\nmodel.load_state_dict(torch.load(MODEL_PATH, map_location=DEVICE))\nmodel = model.to(DEVICE)\nprint('Model loaded.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-21T04:26:44.913218Z","iopub.execute_input":"2026-02-21T04:26:44.913473Z","iopub.status.idle":"2026-02-21T04:26:45.924883Z","shell.execute_reply.started":"2026-02-21T04:26:44.913451Z","shell.execute_reply":"2026-02-21T04:26:45.924259Z"}},"outputs":[],"execution_count":null},{"id":"ed468d1e","cell_type":"code","source":"# ── 3. Grad-CAM ───────────────────────────────────────────────────────────\nclass GradCAM:\n    def __init__(self, model, arch):\n        self.model = model\n        self.activations = self.gradients = None\n        target = model.features[-1] if arch == 'efficientnet_b0' else model.layer4[-1]\n        self._fh = target.register_forward_hook(lambda m, i, o: setattr(self, 'activations', o.detach()))\n        self._bh = target.register_full_backward_hook(lambda m, gi, go: setattr(self, 'gradients', go[0].detach()))\n\n    def remove(self):\n        self._fh.remove(); self._bh.remove()\n\n    def generate(self, img_tensor):\n        self.model.zero_grad()\n        t = img_tensor.clone().requires_grad_(True)\n        self.model(t).squeeze().backward()\n        w   = self.gradients.squeeze().mean(dim=(1, 2), keepdim=True)\n        cam = torch.relu((w * self.activations.squeeze()).sum(dim=0)).cpu().numpy()\n        if cam.max() > 0:\n            cam = (cam - cam.min()) / (cam.max() - cam.min() + 1e-8)\n        return cam\n\n\ngrad_cam = GradCAM(model, ARCH)\nval_transform = T.Compose([T.ToPILImage(), T.ToTensor(), T.Normalize(mean=MEAN, std=STD)])\n\n\ndef load_npy(image_id):\n    \"\"\"Load a cached NPY array and return uint8 RGB for display / overlay.\"\"\"\n    path = os.path.join(NPY_CACHE_DIR, f'{image_id}.npy')\n    try:\n        return np.load(path)              # uint8  [0, 255]\n    except Exception:\n        return np.zeros((IMG_SIZE, IMG_SIZE, 3), np.uint8)\n\n\ndef get_tensor(image_id):\n    return val_transform(load_npy(image_id)).unsqueeze(0).to(DEVICE)\n\n\ndef make_overlay(orig_rgb, cam, alpha=0.45):\n    cam_r = cv2.resize(cam, (orig_rgb.shape[1], orig_rgb.shape[0]))\n    heat  = (cm.jet(cam_r)[:, :, :3] * 255).astype(np.uint8)\n    return (alpha * heat + (1 - alpha) * orig_rgb).astype(np.uint8)\n\n\nprint('Grad-CAM & helpers ready.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-21T04:26:45.926358Z","iopub.execute_input":"2026-02-21T04:26:45.926614Z","iopub.status.idle":"2026-02-21T04:26:45.937665Z","shell.execute_reply.started":"2026-02-21T04:26:45.92659Z","shell.execute_reply":"2026-02-21T04:26:45.937046Z"}},"outputs":[],"execution_count":null},{"id":"b332a652","cell_type":"code","source":"# ── 4. Inference + calibration pipeline ──────────────────────────────────\ndef infer_single(image_id: str) -> dict:\n    \"\"\"\n    Run forward pass, apply temperature calibration, return prediction dict.\n    \"\"\"\n    model.train()   # train mode so Grad-CAM gradients flow\n    t = get_tensor(image_id)\n    with torch.no_grad():\n        logit = model(t).squeeze().cpu().item()\n\n    raw_prob  = float(torch.sigmoid(torch.tensor(logit)).item())\n    cal_prob  = float(torch.sigmoid(torch.tensor(logit / TEMPERATURE)).item())\n\n    model.train()\n    cam = grad_cam.generate(t)\n    model.eval()\n\n    return {'logit': logit, 'raw_prob': raw_prob, 'cal_prob': cal_prob, 'cam': cam}\n\n\nprint('Inference pipeline defined.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-21T04:26:45.939046Z","iopub.execute_input":"2026-02-21T04:26:45.939609Z","iopub.status.idle":"2026-02-21T04:26:45.955348Z","shell.execute_reply.started":"2026-02-21T04:26:45.939585Z","shell.execute_reply":"2026-02-21T04:26:45.954814Z"}},"outputs":[],"execution_count":null},{"id":"c7549a11","cell_type":"code","source":"# ── 5. Report schema (fixed vocabulary) ──────────────────────────────────\n\"\"\"\nFIXED VOCABULARY — do not change these strings.\nThe report module is intentionally restricted to prevent diagnostic claims.\n\"\"\"\n\n# Outcome phrases (based on threshold)\nOUTCOME_POSITIVE = 'Hemorrhage indicator detected'\nOUTCOME_NEGATIVE = 'No hemorrhage indicator detected'\n\n# Band labels\nBAND_LABELS = {\n    'HIGH'  : 'High confidence',\n    'MEDIUM': 'Moderate confidence',\n    'LOW'   : 'Low confidence',\n}\n\n# Triage action phrases (one per band × outcome combination)\nTRIAGE_ACTIONS = {\n    ('POSITIVE', 'HIGH')  : 'Urgent radiologist review recommended',\n    ('POSITIVE', 'MEDIUM'): 'Prioritised radiologist review recommended',\n    ('POSITIVE', 'LOW')   : 'Radiologist review recommended',\n    ('NEGATIVE', 'HIGH')  : 'Standard workflow — no immediate escalation indicated',\n    ('NEGATIVE', 'MEDIUM'): 'Standard workflow — manual review if clinically indicated',\n    ('NEGATIVE', 'LOW')   : 'Standard workflow',\n}\n\nDISCLAIMER = (\n    'This report is produced by an AI-assisted screening tool and does NOT constitute a '\n    'medical diagnosis. All screening findings must be reviewed and confirmed by a '\n    'qualified, licensed medical professional before any clinical decision is made. '\n    'The system is intended solely as a decision-support aid in a screening workflow '\n    'and is not cleared for standalone diagnostic use.'\n)\n\n\ndef build_report(image_id: str, inference: dict,\n                 true_label: int = None) -> dict:\n    \"\"\"\n    Build a structured screening report from inference results.\n    Enforces fixed vocabulary — never outputs free-form clinical text.\n    \"\"\"\n    cal_prob = inference['cal_prob']\n\n    # Band assignment\n    if cal_prob >= HIGH_THRESHOLD:\n        band = 'HIGH'\n    elif cal_prob >= LOW_THRESHOLD:\n        band = 'MEDIUM'\n    else:\n        band = 'LOW'\n\n    # Outcome: based on base_threshold (Youden-optimal from training)\n    is_positive = cal_prob >= BASE_THRESHOLD\n    outcome_str = OUTCOME_POSITIVE if is_positive else OUTCOME_NEGATIVE\n    outcome_key = 'POSITIVE' if is_positive else 'NEGATIVE'\n\n    triage_action = TRIAGE_ACTIONS[(outcome_key, band)]\n\n    # Save Grad-CAM overlay\n    orig_rgb = load_npy(image_id)\n    overlay  = make_overlay(orig_rgb, inference['cam'])\n    cam_save_path = str(REPORTS_DIR / f'{image_id}_gradcam.png')\n    cv2.imwrite(cam_save_path, cv2.cvtColor(overlay, cv2.COLOR_RGB2BGR))\n\n    report = {\n        'report_id'        : f'RPT_{datetime.datetime.now().strftime(\"%Y%m%d_%H%M%S\")}_{image_id[-8:]}',\n        'generated_at'     : datetime.datetime.utcnow().isoformat() + 'Z',\n        'image_id'         : image_id,\n        'ground_truth_label': int(true_label) if true_label is not None else 'N/A',\n        'screening_module' : {\n            'version'             : '1.0',\n            'architecture'        : ARCH,\n            'calibration_method'  : CALIB_METHOD,\n        },\n        'prediction': {\n            'screening_outcome'      : outcome_str,\n            'raw_probability'        : round(inference['raw_prob'], 4),\n            'calibrated_probability' : round(cal_prob, 4),\n            'confidence_band'        : band,\n            'confidence_band_label'  : BAND_LABELS[band],\n            'decision_threshold'     : round(BASE_THRESHOLD, 4),\n        },\n        'triage': {\n            'action'  : triage_action,\n            'urgency' : 'URGENT' if (is_positive and band == 'HIGH') else 'STANDARD',\n        },\n        'explainability': {\n            'method'     : 'Gradient-weighted Class Activation Mapping (Grad-CAM)',\n            'heatmap_path': cam_save_path,\n            'note'       : (\n                'Highlighted regions indicate areas with greatest influence on the '\n                'screening decision. These are not confirmed anatomical findings.'\n            ),\n        },\n        'disclaimer': DISCLAIMER,\n    }\n    return report\n\n\nprint('Report schema defined.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-21T04:26:45.956371Z","iopub.execute_input":"2026-02-21T04:26:45.956736Z","iopub.status.idle":"2026-02-21T04:26:45.975998Z","shell.execute_reply.started":"2026-02-21T04:26:45.956704Z","shell.execute_reply":"2026-02-21T04:26:45.975269Z"}},"outputs":[],"execution_count":null},{"id":"410bfb27","cell_type":"code","source":"# ── 6. Generate reports for sample images ────────────────────────────────\nimport random\nrandom.seed(SEED)\n\nmanifest = pd.read_csv(MANIFEST_PATH)\nval_df   = manifest[manifest['split'] == 'val'].reset_index(drop=True)\n\n# Sample a mix: 5 positive + 5 negative\npos_samples = val_df[val_df['any'] == 1].sample(N_REPORTS // 2, random_state=SEED)\nneg_samples = val_df[val_df['any'] == 0].sample(N_REPORTS // 2, random_state=SEED)\nsample_df   = pd.concat([pos_samples, neg_samples]).sample(frac=1, random_state=SEED)\n\nall_reports = []\nfor _, row in sample_df.iterrows():\n    inf = infer_single(row['image_id'])\n    rep = build_report(row['image_id'], inf, true_label=row['any'])\n    all_reports.append(rep)\n\n    # Save individual JSON report\n    report_path = REPORTS_DIR / f'{row[\"image_id\"]}_report.json'\n    with open(report_path, 'w') as f:\n        json.dump({k: v for k, v in rep.items() if k != 'cam'}, f, indent=2)\n\nprint(f'{len(all_reports)} reports generated and saved to {REPORTS_DIR}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-21T04:26:45.976898Z","iopub.execute_input":"2026-02-21T04:26:45.977184Z","iopub.status.idle":"2026-02-21T04:26:48.123024Z","shell.execute_reply.started":"2026-02-21T04:26:45.977162Z","shell.execute_reply":"2026-02-21T04:26:48.12227Z"}},"outputs":[],"execution_count":null},{"id":"f43e7ee6","cell_type":"code","source":"# ── 7. HTML report display ────────────────────────────────────────────────\ndef img_to_base64(path: str) -> str:\n    with open(path, 'rb') as f:\n        return base64.b64encode(f.read()).decode('utf-8')\n\n\nBAND_COLORS = {'HIGH': '#e74c3c', 'MEDIUM': '#f39c12', 'LOW': '#27ae60'}\nURGENCY_COLORS = {'URGENT': '#e74c3c', 'STANDARD': '#2980b9'}\n\n\ndef render_report_html(report: dict) -> str:\n    pred   = report['prediction']\n    triage = report['triage']\n    expl   = report['explainability']\n    band   = pred['confidence_band']\n    band_color = BAND_COLORS.get(band, '#555')\n    urg_color  = URGENCY_COLORS.get(triage['urgency'], '#555')\n\n    # Embed Grad-CAM image as base64\n    cam_data = img_to_base64(expl['heatmap_path']) if os.path.exists(expl['heatmap_path']) else ''\n    cam_img_tag = f'<img src=\"data:image/png;base64,{cam_data}\" style=\"height:200px;\"/>' if cam_data else '<i>Image not available</i>'\n\n    outcome_color = '#e74c3c' if 'detected' in pred['screening_outcome'].lower() and 'No' not in pred['screening_outcome'] else '#27ae60'\n\n    html = f\"\"\"\n<div style=\"border:1px solid #ccc; border-radius:8px; padding:16px; margin:16px 0;\n             font-family:Arial,sans-serif; max-width:800px; background:#fafafa;\">\n  <h3 style=\"margin:0 0 8px;\">Screening Report\n    <span style=\"font-size:0.7em; color:#888; float:right;\">{report['report_id']}</span>\n  </h3>\n  <p style=\"margin:2px 0; color:#555; font-size:0.85em;\">\n    Image ID: <code>{report['image_id']}</code> &nbsp;|\n    Generated: {report['generated_at']} &nbsp;|\n    Ground truth: <b>{'Positive' if report['ground_truth_label'] == 1 else 'Negative' if report['ground_truth_label'] == 0 else 'Unknown'}</b>\n  </p>\n  <hr style=\"margin:10px 0;\"/>\n\n  <table style=\"width:100%; border-collapse:collapse;\">\n    <tr>\n      <td style=\"width:50%; vertical-align:top; padding-right:16px;\">\n        <h4 style=\"margin:0 0 8px;\">Screening Outcome</h4>\n        <div style=\"background:{outcome_color}; color:white; padding:8px 12px;\n                    border-radius:4px; font-size:1.05em; font-weight:bold;\">\n          {pred['screening_outcome']}\n        </div>\n\n        <h4 style=\"margin:12px 0 6px;\">Confidence</h4>\n        <table style=\"font-size:0.9em; width:100%;\">\n          <tr><td>Raw probability:</td><td><b>{pred['raw_probability']:.4f}</b></td></tr>\n          <tr><td>Calibrated probability:</td><td><b>{pred['calibrated_probability']:.4f}</b></td></tr>\n          <tr><td>Decision threshold:</td><td><b>{pred['decision_threshold']:.4f}</b></td></tr>\n          <tr><td>Confidence band:</td>\n              <td><span style=\"background:{band_color}; color:white; padding:2px 8px;\n                               border-radius:3px; font-weight:bold;\">\n                {pred['confidence_band_label']} ({band})\n              </span></td></tr>\n        </table>\n\n        <h4 style=\"margin:12px 0 6px;\">Triage Recommendation</h4>\n        <div style=\"border-left:4px solid {urg_color}; padding-left:10px;\">\n          <span style=\"color:{urg_color}; font-weight:bold;\">{triage['urgency']}</span><br/>\n          {triage['action']}\n        </div>\n      </td>\n\n      <td style=\"width:50%; vertical-align:top;\">\n        <h4 style=\"margin:0 0 8px;\">Visual Evidence (Grad-CAM)</h4>\n        {cam_img_tag}\n        <p style=\"font-size:0.8em; color:#888; margin:4px 0;\">{expl['note']}</p>\n      </td>\n    </tr>\n  </table>\n\n  <hr style=\"margin:12px 0;\"/>\n  <p style=\"font-size:0.75em; color:#888; margin:0; border-left:3px solid #e74c3c; padding-left:8px;\">\n    <b>DISCLAIMER:</b> {report['disclaimer']}\n  </p>\n</div>\n\"\"\"\n    return html\n\n\nprint('HTML renderer defined. Displaying reports...')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-21T04:26:48.124185Z","iopub.execute_input":"2026-02-21T04:26:48.124814Z","iopub.status.idle":"2026-02-21T04:26:48.133954Z","shell.execute_reply.started":"2026-02-21T04:26:48.124785Z","shell.execute_reply":"2026-02-21T04:26:48.133238Z"}},"outputs":[],"execution_count":null},{"id":"87418127","cell_type":"code","source":"# ── 8. Render all sample reports ──────────────────────────────────────────\nfor report in all_reports:\n    display(HTML(render_report_html(report)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-21T04:26:48.134921Z","iopub.execute_input":"2026-02-21T04:26:48.135311Z","iopub.status.idle":"2026-02-21T04:26:48.175323Z","shell.execute_reply.started":"2026-02-21T04:26:48.135279Z","shell.execute_reply":"2026-02-21T04:26:48.174781Z"}},"outputs":[],"execution_count":null},{"id":"7b9a14f6","cell_type":"code","source":"# ── 9. Batch summary table + Triage Risk Analysis ─────────────────────────\nsummary_rows = []\nfor rep in all_reports:\n    pred = rep['prediction']\n    summary_rows.append({\n        'image_id'           : rep['image_id'][-8:],\n        'true_label'         : rep['ground_truth_label'],\n        'screening_outcome'  : pred['screening_outcome'],\n        'cal_prob'           : pred['calibrated_probability'],\n        'confidence_band'    : pred['confidence_band'],\n        'triage_action'      : rep['triage']['action'],\n        'urgency'            : rep['triage']['urgency'],\n    })\n\nsummary_df = pd.DataFrame(summary_rows)\nsummary_df.to_csv('/kaggle/working/report_summary.csv', index=False)\n\nprint('Batch summary:')\ndisplay(summary_df)\n\n# ── Triage Risk Analysis ─────────────────────────────────────────────────\n# Go beyond \"did we escalate positives?\" — also analyze false escalations.\nis_positive_pred = summary_df['screening_outcome'].str.contains('detected') & \\\n                   ~summary_df['screening_outcome'].str.contains('No ')\nis_positive_true = summary_df['true_label'] == 1\nis_urgent = summary_df['urgency'] == 'URGENT'\n\ntp = (is_positive_pred & is_positive_true).sum()\nfp = (is_positive_pred & ~is_positive_true).sum()\nfn = (~is_positive_pred & is_positive_true).sum()\ntn = (~is_positive_pred & ~is_positive_true).sum()\nn_pos = is_positive_true.sum()\nn_neg = (~is_positive_true).sum()\n\n# Urgent escalation analysis\nurgent_tp = (is_urgent & is_positive_true).sum()\nurgent_fp = (is_urgent & ~is_positive_true).sum()\nurgent_fn = (~is_urgent & is_positive_true).sum()\n\nprint(f'\\n{\"═\" * 55}')\nprint(f'TRIAGE RISK ANALYSIS (n={len(summary_df)} sample reports)')\nprint(f'{\"═\" * 55}')\nprint(f'\\n  Screening Decision:')\nprint(f'    True Positives  : {tp}/{n_pos} hemorrhage cases detected')\nprint(f'    False Positives : {fp}/{n_neg} normal cases falsely flagged')\nprint(f'    False Negatives : {fn}/{n_pos} hemorrhage cases missed')\nprint(f'    True Negatives  : {tn}/{n_neg} normal cases correctly cleared')\nprint(f'    Precision (PPV) : {tp / max(tp + fp, 1):.2%}')\nprint(f'    Recall (Sens)   : {tp / max(tp + fn, 1):.2%}')\n\nprint(f'\\n  Urgent Escalation:')\nprint(f'    True urgent     : {urgent_tp} (hemorrhage → URGENT)')\nprint(f'    False urgent    : {urgent_fp} (normal → URGENT) ← unnecessary escalation')\nprint(f'    Missed urgent   : {urgent_fn} (hemorrhage → not URGENT)')\nprint(f'    Escalation PPV  : {urgent_tp / max(urgent_tp + urgent_fp, 1):.2%}')\n\nif fp > 0:\n    print(f'\\n  ⚠ FALSE POSITIVE DETAIL:')\n    fp_cases = summary_df[is_positive_pred & ~is_positive_true]\n    for _, row in fp_cases.iterrows():\n        print(f'    {row[\"image_id\"]}  cal_prob={row[\"cal_prob\"]:.4f}  '\n              f'band={row[\"confidence_band\"]}  action=\"{row[\"triage_action\"]}\"')\n    print(f'\\n    These represent unnecessary radiologist escalations.')\n    print(f'    In deployment, {fp} of {fp + tp} flagged cases are false alarms.')\n\nif fn > 0:\n    print(f'\\n  ⚠ FALSE NEGATIVE DETAIL:')\n    fn_cases = summary_df[~is_positive_pred & is_positive_true]\n    for _, row in fn_cases.iterrows():\n        print(f'    {row[\"image_id\"]}  cal_prob={row[\"cal_prob\"]:.4f}  '\n              f'band={row[\"confidence_band\"]}  → MISSED')\n    print(f'    These hemorrhage cases would not be escalated.')\n\nprint(f'\\n  Note: This analysis is on {len(summary_df)} sampled images.')\nprint(f'  Full-set triage metrics are computed in the patient-level section below.')\nprint(f'{\"═\" * 55}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-21T04:26:48.177162Z","iopub.execute_input":"2026-02-21T04:26:48.177433Z","iopub.status.idle":"2026-02-21T04:26:48.221981Z","shell.execute_reply.started":"2026-02-21T04:26:48.177413Z","shell.execute_reply":"2026-02-21T04:26:48.221292Z"}},"outputs":[],"execution_count":null},{"id":"b88c360f","cell_type":"markdown","source":"---\n# Part II — Final Analysis\n\nThe sections below address key structural requirements for a complete project deliverable:\npatient-level aggregation, training stability, capacity analysis, and deployment configuration.","metadata":{}},{"id":"9615ae41","cell_type":"code","source":"# ── 10. Patient-Level Aggregation ──────────────────────────────────────────\n# All training, calibration, and evaluation so far has been at SLICE level.\n# Clinically and competition-wise, the PATIENT (study) is the unit.\n# We aggregate slice probabilities to patient level using multiple strategies.\n\nfrom torch.utils.data import Dataset, DataLoader\nfrom tqdm.auto import tqdm\n\n# ── Load full validation set and run inference ───────────────────────────\nfull_manifest = pd.read_csv(MANIFEST_PATH)\nval_full = full_manifest[full_manifest['split'] == 'val'].reset_index(drop=True)\n\nprint(f'Validation set: {len(val_full):,} slices, '\n      f'{val_full[\"patient_id\"].nunique():,} patients')\n\n\nclass SimpleDataset(Dataset):\n    def __init__(self, df, npy_root, transform):\n        self.df = df.reset_index(drop=True)\n        self.npy_root = npy_root\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        path = os.path.join(self.npy_root, f'{row[\"image_id\"]}.npy')\n        try:\n            img = np.load(path)\n        except Exception:\n            img = np.zeros((IMG_SIZE, IMG_SIZE, 3), dtype=np.uint8)\n        return self.transform(img), idx\n\n\nval_dataset = SimpleDataset(val_full, NPY_CACHE_DIR, val_transform)\nval_loader  = DataLoader(val_dataset, batch_size=64, shuffle=False,\n                         num_workers=4, pin_memory=True)\n\n# ── Collect raw logits for all val slices ─────────────────────────────────\nmodel.eval()\nall_logits = np.zeros(len(val_full))\n\nwith torch.no_grad():\n    for batch_imgs, batch_idxs in tqdm(val_loader, desc='Val inference'):\n        logits = model(batch_imgs.to(DEVICE)).squeeze(-1).cpu().numpy()\n        for i, idx in enumerate(batch_idxs.numpy()):\n            all_logits[idx] = logits[i]\n\nraw_probs = 1 / (1 + np.exp(-all_logits))                     # sigmoid\ncal_probs = 1 / (1 + np.exp(-all_logits / TEMPERATURE))       # temp-scaled\n\nval_full['raw_prob'] = raw_probs\nval_full['cal_prob'] = cal_probs\nval_full['true_label'] = val_full['any'].astype(int)\n\n# ── Slice-level AUC (sanity check) ──────────────────────────────────────\nslice_auc = roc_auc_score(val_full['true_label'], val_full['cal_prob'])\nprint(f'\\nSlice-level AUC (calibrated): {slice_auc:.5f}')\n\n# ── Patient-level aggregation: max, mean, noisy-or ──────────────────────\ndef noisy_or(probs):\n    \"\"\"Noisy-OR: P(any) = 1 - prod(1 - p_i)\"\"\"\n    return 1.0 - np.prod(1.0 - probs)\n\n\npatient_agg = val_full.groupby('patient_id').agg(\n    true_label = ('true_label', 'max'),   # patient positive if ANY slice is positive\n    n_slices   = ('true_label', 'count'),\n    prob_max   = ('cal_prob', 'max'),\n    prob_mean  = ('cal_prob', 'mean'),\n    prob_noisy_or = ('cal_prob', noisy_or),\n).reset_index()\n\nprint(f'\\nPatient-level summary: {len(patient_agg):,} patients')\nprint(f'  Positive patients: {patient_agg[\"true_label\"].sum():,} '\n      f'({patient_agg[\"true_label\"].mean()*100:.1f}%)')\nprint(f'  Median slices/patient: {patient_agg[\"n_slices\"].median():.0f}')\n\n# ── Patient-level AUC for each aggregation strategy ─────────────────────\nstrategies = {\n    'Max (slice)'   : 'prob_max',\n    'Mean (slice)'  : 'prob_mean',\n    'Noisy-OR'      : 'prob_noisy_or',\n}\n\nprint(f'\\nPatient-Level AUC by Aggregation Strategy:')\nprint(f'  {\"Strategy\":<16s}  {\"AUC\":>8s}  {\"Sens\":>6s}  {\"Spec\":>6s}')\nprint(f'  {\"─\"*16}  {\"─\"*8}  {\"─\"*6}  {\"─\"*6}')\n\nbest_patient_strategy = None\nbest_patient_auc = 0\n\nfor name, col in strategies.items():\n    auc = roc_auc_score(patient_agg['true_label'], patient_agg[col])\n    fpr, tpr, thresholds = roc_curve(patient_agg['true_label'], patient_agg[col])\n    j_idx = np.argmax(tpr - fpr)\n    thr = thresholds[j_idx]\n    preds = (patient_agg[col] >= thr).astype(int)\n    tp = ((preds == 1) & (patient_agg['true_label'] == 1)).sum()\n    fn = ((preds == 0) & (patient_agg['true_label'] == 1)).sum()\n    tn = ((preds == 0) & (patient_agg['true_label'] == 0)).sum()\n    fp = ((preds == 1) & (patient_agg['true_label'] == 0)).sum()\n    sens = tp / max(tp + fn, 1)\n    spec = tn / max(tn + fp, 1)\n    print(f'  {name:<16s}  {auc:>8.5f}  {sens:>6.4f}  {spec:>6.4f}')\n\n    if auc > best_patient_auc:\n        best_patient_auc = auc\n        best_patient_strategy = name\n        best_patient_thr = thr\n        best_patient_col = col\n\nprint(f'\\n★ Best patient-level strategy: {best_patient_strategy} '\n      f'(AUC={best_patient_auc:.5f})')\nprint(f'  vs. slice-level AUC: {slice_auc:.5f} '\n      f'(Δ = {best_patient_auc - slice_auc:+.5f})')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-21T04:26:48.222935Z","iopub.execute_input":"2026-02-21T04:26:48.223196Z","iopub.status.idle":"2026-02-21T04:27:21.275468Z","shell.execute_reply.started":"2026-02-21T04:26:48.223174Z","shell.execute_reply":"2026-02-21T04:27:21.274577Z"}},"outputs":[],"execution_count":null},{"id":"8b3be0e8","cell_type":"code","source":"# ── 11. Patient-Level ROC Curve & Confusion Matrix ────────────────────────\nfig, axes = plt.subplots(1, 3, figsize=(18, 5))\n\n# ── ROC curves: slice vs patient (all strategies) ────────────────────────\n# Slice-level ROC\ns_fpr, s_tpr, _ = roc_curve(val_full['true_label'], val_full['cal_prob'])\naxes[0].plot(s_fpr, s_tpr, 'k--', linewidth=1.5, alpha=0.5,\n             label=f'Slice-level (AUC={slice_auc:.4f})')\n\ncolors = {'Max (slice)': 'tab:blue', 'Mean (slice)': 'tab:orange', 'Noisy-OR': 'tab:green'}\nfor name, col in strategies.items():\n    p_fpr, p_tpr, _ = roc_curve(patient_agg['true_label'], patient_agg[col])\n    p_auc = roc_auc_score(patient_agg['true_label'], patient_agg[col])\n    lw = 2.5 if name == best_patient_strategy else 1.5\n    axes[0].plot(p_fpr, p_tpr, color=colors[name], linewidth=lw,\n                 label=f'{name} (AUC={p_auc:.4f})')\n\naxes[0].plot([0, 1], [0, 1], 'k:', linewidth=0.5)\naxes[0].set(title='ROC: Slice-Level vs Patient-Level',\n            xlabel='False Positive Rate', ylabel='True Positive Rate')\naxes[0].legend(fontsize=8)\naxes[0].set_xlim(0, 1); axes[0].set_ylim(0, 1)\n\n# ── Confusion matrix (best patient-level strategy) ───────────────────────\nbest_preds = (patient_agg[best_patient_col] >= best_patient_thr).astype(int)\ncm = confusion_matrix(patient_agg['true_label'], best_preds)\nsns.heatmap(cm, annot=True, fmt='d', cmap='Blues', ax=axes[1],\n            xticklabels=['Pred Neg', 'Pred Pos'],\n            yticklabels=['True Neg', 'True Pos'])\naxes[1].set_title(f'Patient-Level Confusion Matrix\\n({best_patient_strategy}, '\n                  f'thr={best_patient_thr:.3f})')\n\n# ── Distribution of max slice probability per patient ────────────────────\npos_patients = patient_agg[patient_agg['true_label'] == 1]['prob_max']\nneg_patients = patient_agg[patient_agg['true_label'] == 0]['prob_max']\naxes[2].hist(neg_patients, bins=40, alpha=0.6, color='tab:blue',\n             label=f'Negative (n={len(neg_patients)})', density=True)\naxes[2].hist(pos_patients, bins=40, alpha=0.6, color='tab:red',\n             label=f'Positive (n={len(pos_patients)})', density=True)\naxes[2].axvline(best_patient_thr, color='black', linestyle='--',\n                label=f'Threshold={best_patient_thr:.3f}')\naxes[2].set(title='Patient Max-Probability Distribution',\n            xlabel='Max calibrated probability', ylabel='Density')\naxes[2].legend(fontsize=8)\n\nplt.suptitle('Patient-Level Analysis', fontsize=13)\nplt.tight_layout()\nplt.savefig('/kaggle/working/patient_level_analysis.png', bbox_inches='tight')\nplt.show()\n\n# ── Patient-level performance summary ────────────────────────────────────\ntp = cm[1, 1]; fn = cm[1, 0]; tn = cm[0, 0]; fp = cm[0, 1]\nsens = tp / max(tp + fn, 1)\nspec = tn / max(tn + fp, 1)\nppv  = tp / max(tp + fp, 1)\nnpv  = tn / max(tn + fn, 1)\nf1   = 2 * tp / max(2 * tp + fp + fn, 1)\n\nprint(f'\\n{\"═\" * 50}')\nprint(f'PATIENT-LEVEL PERFORMANCE ({best_patient_strategy})')\nprint(f'{\"═\" * 50}')\nprint(f'  AUC         : {best_patient_auc:.5f}')\nprint(f'  Sensitivity : {sens:.4f}')\nprint(f'  Specificity : {spec:.4f}')\nprint(f'  PPV         : {ppv:.4f}')\nprint(f'  NPV         : {npv:.4f}')\nprint(f'  F1          : {f1:.4f}')\nprint(f'  Threshold   : {best_patient_thr:.4f}')\nprint(f'{\"═\" * 50}')\n\n# ── Patient-level triage risk (full validation set) ──────────────────────\nprint(f'\\n{\"═\" * 50}')\nprint(f'PATIENT-LEVEL TRIAGE RISK (full val set)')\nprint(f'{\"═\" * 50}')\nprint(f'  Total patients       : {len(patient_agg):,}')\nprint(f'  Positive (hemorrhage): {int(patient_agg[\"true_label\"].sum()):,}')\nprint(f'  Negative (normal)    : {int((patient_agg[\"true_label\"] == 0).sum()):,}')\nprint(f'\\n  At threshold {best_patient_thr:.4f}:')\nprint(f'    True Positives     : {tp:,}')\nprint(f'    False Positives    : {fp:,} ← unnecessary escalations')\nprint(f'    False Negatives    : {fn:,} ← missed hemorrhages')\nprint(f'    True Negatives     : {tn:,}')\nprint(f'    Escalation PPV     : {ppv:.4f} ({ppv*100:.1f}% of flagged patients have hemorrhage)')\nprint(f'    Miss rate (FNR)    : {fn / max(fn + tp, 1):.4f} ({fn / max(fn + tp, 1)*100:.1f}% of hemorrhage patients missed)')\nprint(f'{\"═\" * 50}')\n\n# Save patient-level results\npatient_results = {\n    'aggregation_strategy': best_patient_strategy,\n    'patient_auc':    round(best_patient_auc, 5),\n    'slice_auc':      round(slice_auc, 5),\n    'sensitivity':    round(sens, 4),\n    'specificity':    round(spec, 4),\n    'ppv':            round(ppv, 4),\n    'npv':            round(npv, 4),\n    'f1':             round(f1, 4),\n    'threshold':      round(best_patient_thr, 4),\n    'n_patients':     len(patient_agg),\n    'n_positive':     int(patient_agg['true_label'].sum()),\n    'false_positives': int(fp),\n    'false_negatives': int(fn),\n    'escalation_ppv':  round(ppv, 4),\n    'miss_rate_fnr':   round(fn / max(fn + tp, 1), 4),\n}\nwith open('/kaggle/working/patient_level_results.json', 'w') as f:\n    json.dump(patient_results, f, indent=2)\n\nprint('\\nSaved: patient_level_results.json, patient_level_analysis.png')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-21T04:27:21.277489Z","iopub.execute_input":"2026-02-21T04:27:21.277868Z","iopub.status.idle":"2026-02-21T04:27:22.455529Z","shell.execute_reply.started":"2026-02-21T04:27:21.277821Z","shell.execute_reply":"2026-02-21T04:27:22.454887Z"}},"outputs":[],"execution_count":null},{"id":"97c75bb3","cell_type":"code","source":"# ── 11b. Patient-Level Recalibration ──────────────────────────────────────\n# Slice-level calibration does NOT transfer after max-pooling aggregation.\n# Max-pooling inflates probabilities → the calibration mapping is invalid.\n# We apply isotonic regression with 5-fold CV at the patient level.\n\nfrom sklearn.isotonic import IsotonicRegression\nfrom sklearn.model_selection import KFold\n\npatient_probs  = patient_agg[best_patient_col].values\npatient_labels = patient_agg['true_label'].values\n\n# ── ECE helper ───────────────────────────────────────────────────────────\ndef compute_ece(probs, labels, n_bins=15):\n    \"\"\"Expected Calibration Error with equal-width bins.\"\"\"\n    bin_edges = np.linspace(0, 1, n_bins + 1)\n    ece = 0.0\n    for i in range(n_bins):\n        mask = (probs >= bin_edges[i]) & (probs < bin_edges[i + 1])\n        if mask.sum() == 0:\n            continue\n        bin_acc  = labels[mask].mean()\n        bin_conf = probs[mask].mean()\n        ece += mask.sum() * abs(bin_acc - bin_conf)\n    return ece / len(probs)\n\n# ── Pre-recalibration ECE at patient level ───────────────────────────────\npatient_ece_before = compute_ece(patient_probs, patient_labels)\n\n# ── 5-fold CV isotonic regression at patient level ───────────────────────\nkf = KFold(n_splits=5, shuffle=True, random_state=SEED)\nrecal_probs = np.zeros_like(patient_probs)\n\nfor fold_idx, (train_idx, val_idx) in enumerate(kf.split(patient_probs)):\n    ir_fold = IsotonicRegression(out_of_bounds='clip')\n    ir_fold.fit(patient_probs[train_idx], patient_labels[train_idx])\n    recal_probs[val_idx] = ir_fold.predict(patient_probs[val_idx])\n\npatient_ece_after = compute_ece(recal_probs, patient_labels)\n\n# ── Full-data isotonic for deployment ────────────────────────────────────\nir_patient = IsotonicRegression(out_of_bounds='clip')\nir_patient.fit(patient_probs, patient_labels)\n\n# ── Recalibrated ROC curve ───────────────────────────────────────────────\nfpr_r, tpr_r, thr_r = roc_curve(patient_labels, recal_probs)\nrecal_auc = roc_auc_score(patient_labels, recal_probs)\n\n# ── Threshold selection: sensitivity-constrained ─────────────────────────\n# For a screening tool, missed hemorrhage is catastrophic.\n# Strategy: find lowest threshold where sensitivity ≥ SENS_FLOOR,\n# then among those pick the one with maximum specificity.\n\nSENS_FLOOR = 0.90\n\n# Build tradeoff table at multiple sensitivity targets\nn_pos = int(patient_labels.sum())\nn_neg = int((patient_labels == 0).sum())\n\nsens_targets = [0.80, 0.85, 0.90, 0.925, 0.95, 0.975]\ntradeoff_rows = []\n\nfor target_sens in sens_targets:\n    # Find thresholds where tpr >= target_sens\n    valid_mask = tpr_r >= target_sens\n    if not valid_mask.any():\n        continue\n    # Among those, pick the one with highest specificity (= lowest fpr)\n    valid_idx = np.where(valid_mask)[0]\n    best_idx = valid_idx[np.argmin(fpr_r[valid_idx])]\n\n    t_thr = float(thr_r[best_idx])\n    t_sens = float(tpr_r[best_idx])\n    t_spec = float(1.0 - fpr_r[best_idx])\n\n    t_preds = (recal_probs >= t_thr).astype(int)\n    t_tp = int(((t_preds == 1) & (patient_labels == 1)).sum())\n    t_fp = int(((t_preds == 1) & (patient_labels == 0)).sum())\n    t_fn = int(((t_preds == 0) & (patient_labels == 1)).sum())\n    t_tn = int(((t_preds == 0) & (patient_labels == 0)).sum())\n    t_ppv = t_tp / max(t_tp + t_fp, 1)\n    t_f1 = 2 * t_tp / max(2 * t_tp + t_fp + t_fn, 1)\n    t_fnr = t_fn / max(t_fn + t_tp, 1)\n\n    tradeoff_rows.append({\n        'target_sens': target_sens,\n        'threshold': t_thr, 'sensitivity': t_sens, 'specificity': t_spec,\n        'ppv': t_ppv, 'f1': t_f1, 'tp': t_tp, 'fp': t_fp,\n        'fn': t_fn, 'tn': t_tn, 'fnr': t_fnr,\n    })\n\n# Also add Youden for comparison\nj_idx_youden = np.argmax(tpr_r - fpr_r)\nyouden_thr = float(thr_r[j_idx_youden])\ny_preds = (recal_probs >= youden_thr).astype(int)\ny_tp = int(((y_preds == 1) & (patient_labels == 1)).sum())\ny_fp = int(((y_preds == 1) & (patient_labels == 0)).sum())\ny_fn = int(((y_preds == 0) & (patient_labels == 1)).sum())\ny_tn = int(((y_preds == 0) & (patient_labels == 0)).sum())\ny_ppv = y_tp / max(y_tp + y_fp, 1)\ny_f1 = 2 * y_tp / max(2 * y_tp + y_fp + y_fn, 1)\ny_sens = float(tpr_r[j_idx_youden])\ny_spec = float(1.0 - fpr_r[j_idx_youden])\n\n# Print tradeoff table\nprint(f'\\n{\"═\" * 90}')\nprint(f'THRESHOLD OPERATING POINT ANALYSIS (recalibrated patient-level probabilities)')\nprint(f'{\"═\" * 90}')\nprint(f'  Patients: {n_pos} positive, {n_neg} negative, {n_pos + n_neg} total')\nprint(f'\\n  {\"Target\":>6s}  {\"Thr\":>6s}  {\"Sens\":>6s}  {\"Spec\":>6s}  {\"PPV\":>6s}  '\n      f'{\"F1\":>6s}  {\"FP\":>5s}  {\"FN\":>5s}  {\"Miss%\":>6s}  {\"Note\":<20s}')\nprint(f'  {\"─\"*6}  {\"─\"*6}  {\"─\"*6}  {\"─\"*6}  {\"─\"*6}  '\n      f'{\"─\"*6}  {\"─\"*5}  {\"─\"*5}  {\"─\"*6}  {\"─\"*20}')\n\nfor row in tradeoff_rows:\n    note = ''\n    if row['target_sens'] == SENS_FLOOR:\n        note = '← SELECTED (screening)'\n    print(f'  {row[\"target_sens\"]:>6.2%}  {row[\"threshold\"]:>6.4f}  '\n          f'{row[\"sensitivity\"]:>6.4f}  {row[\"specificity\"]:>6.4f}  '\n          f'{row[\"ppv\"]:>6.4f}  {row[\"f1\"]:>6.4f}  {row[\"fp\"]:>5d}  '\n          f'{row[\"fn\"]:>5d}  {row[\"fnr\"]:>6.1%}  {note}')\n\n# Youden row for comparison\nprint(f'  {\"Youd.\":>6s}  {youden_thr:>6.4f}  {y_sens:>6.4f}  {y_spec:>6.4f}  '\n      f'{y_ppv:>6.4f}  {y_f1:>6.4f}  {y_fp:>5d}  {y_fn:>5d}  '\n      f'{y_fn/max(y_fn+y_tp,1):>6.1%}  (Youden J — not for screening)')\n\nprint(f'\\n  Decision rationale:')\nprint(f'    • Youden maximises Sens+Spec equally — appropriate for diagnosis, NOT screening')\nprint(f'    • Screening requires a sensitivity floor to bound the miss rate')\nprint(f'    • At {SENS_FLOOR:.0%} sensitivity floor: miss rate drops to acceptable range')\nprint(f'    • PPV remains usable (> 0.60 threshold for clinical utility)')\nprint(f'{\"═\" * 90}')\n\n# ── Apply the selected threshold (sensitivity-constrained) ───────────────\n# Find the operating point for SENS_FLOOR\nselected_row = [r for r in tradeoff_rows if r['target_sens'] == SENS_FLOOR][0]\nrecal_thr = selected_row['threshold']\n\nrecal_preds = (recal_probs >= recal_thr).astype(int)\ncm_r = confusion_matrix(patient_labels, recal_preds)\ntp_r = cm_r[1, 1]; fn_r = cm_r[1, 0]; tn_r = cm_r[0, 0]; fp_r = cm_r[0, 1]\nsens_r = tp_r / max(tp_r + fn_r, 1)\nspec_r = tn_r / max(tn_r + fp_r, 1)\nppv_r  = tp_r / max(tp_r + fp_r, 1)\nnpv_r  = tn_r / max(tn_r + fn_r, 1)\nf1_r   = 2 * tp_r / max(2 * tp_r + fp_r + fn_r, 1)\n\n# ── Plots: tradeoff curve + reliability diagrams + confusion matrix ──────\nfig, axes = plt.subplots(2, 2, figsize=(14, 10))\n\n# Top-left: Sensitivity vs Specificity vs Threshold\nax_trade = axes[0, 0]\n# Use a subset of thresholds for smooth plotting\nplot_mask = (thr_r >= 0.01) & (thr_r <= 0.99)\nax_trade.plot(thr_r[plot_mask], tpr_r[plot_mask], 'b-', linewidth=2, label='Sensitivity')\nax_trade.plot(thr_r[plot_mask], 1 - fpr_r[plot_mask], 'r-', linewidth=2, label='Specificity')\n\n# Mark operating points\nfor row in tradeoff_rows:\n    marker = '*' if row['target_sens'] == SENS_FLOOR else 'o'\n    ms = 15 if row['target_sens'] == SENS_FLOOR else 6\n    color = 'green' if row['target_sens'] == SENS_FLOOR else 'gray'\n    ax_trade.plot(row['threshold'], row['sensitivity'], marker, color=color,\n                  markersize=ms, zorder=5)\n\nax_trade.axvline(recal_thr, color='green', linestyle='--', alpha=0.7,\n                 label=f'Selected thr={recal_thr:.3f}')\nax_trade.axhline(SENS_FLOOR, color='blue', linestyle=':', alpha=0.5,\n                 label=f'Sens floor={SENS_FLOOR:.0%}')\nax_trade.set(title='Threshold Operating Point Selection',\n             xlabel='Threshold', ylabel='Rate')\nax_trade.legend(fontsize=8, loc='center left')\nax_trade.set_xlim(0, 1); ax_trade.set_ylim(0, 1)\n\n# Top-right: Reliability diagram BEFORE\ndef plot_rel_diagram(ax, probs, labels, title, ece_val, n_bins=15):\n    bin_edges = np.linspace(0, 1, n_bins + 1)\n    bin_confs, bin_accs, bin_sizes = [], [], []\n    for i in range(n_bins):\n        mask = (probs >= bin_edges[i]) & (probs < bin_edges[i + 1])\n        if mask.sum() > 0:\n            bin_confs.append(probs[mask].mean())\n            bin_accs.append(labels[mask].mean())\n            bin_sizes.append(mask.sum())\n        else:\n            bin_confs.append((bin_edges[i] + bin_edges[i+1]) / 2)\n            bin_accs.append(0)\n            bin_sizes.append(0)\n    bc, ba, bs = np.array(bin_confs), np.array(bin_accs), np.array(bin_sizes)\n    mask = bs > 0\n    ax.bar(bc[mask], ba[mask], width=1.0/n_bins, alpha=0.6, color='tab:blue', align='center')\n    ax.plot([0, 1], [0, 1], 'k--', linewidth=1)\n    ax.set(title=f'{title}\\nECE={ece_val:.4f}', xlabel='Mean predicted prob', ylabel='Fraction positive')\n    ax.set_xlim(0, 1); ax.set_ylim(0, 1)\n\nplot_rel_diagram(axes[0, 1], patient_probs, patient_labels,\n                 'Before Recalibration', patient_ece_before)\n\n# Bottom-left: Reliability diagram AFTER\nplot_rel_diagram(axes[1, 0], recal_probs, patient_labels,\n                 'After Recalibration (Isotonic 5-fold CV)', patient_ece_after)\n\n# Bottom-right: Confusion matrix after recalibration\nsns.heatmap(cm_r, annot=True, fmt='d', cmap='Greens', ax=axes[1, 1],\n            xticklabels=['Pred Neg', 'Pred Pos'],\n            yticklabels=['True Neg', 'True Pos'])\naxes[1, 1].set_title(f'Recalibrated CM\\n(sens-constrained thr={recal_thr:.3f})')\n\nplt.suptitle('Patient-Level Recalibration & Threshold Selection', fontsize=13)\nplt.tight_layout()\nplt.savefig('/kaggle/working/patient_recalibration.png', bbox_inches='tight')\nplt.show()\n\n# ── Summary ──────────────────────────────────────────────────────────────\nprint(f'\\n{\"═\" * 65}')\nprint(f'PATIENT-LEVEL RECALIBRATION RESULTS')\nprint(f'{\"═\" * 65}')\nprint(f'  Calibration : Isotonic Regression (5-fold CV)')\nprint(f'  Threshold   : Sensitivity-constrained (≥{SENS_FLOOR:.0%}, max specificity)')\nprint(f'\\n  {\"Metric\":<20s}  {\"Before\":>10s}  {\"After\":>10s}  {\"Δ\":>10s}')\nprint(f'  {\"─\"*20}  {\"─\"*10}  {\"─\"*10}  {\"─\"*10}')\nprint(f'  {\"ECE\":<20s}  {patient_ece_before:>10.5f}  {patient_ece_after:>10.5f}  {patient_ece_after - patient_ece_before:>+10.5f}')\nprint(f'  {\"AUC\":<20s}  {best_patient_auc:>10.5f}  {recal_auc:>10.5f}  {recal_auc - best_patient_auc:>+10.5f}')\nprint(f'  {\"Sensitivity\":<20s}  {sens:>10.4f}  {sens_r:>10.4f}  {sens_r - sens:>+10.4f}')\nprint(f'  {\"Specificity\":<20s}  {spec:>10.4f}  {spec_r:>10.4f}  {spec_r - spec:>+10.4f}')\nprint(f'  {\"PPV\":<20s}  {ppv:>10.4f}  {ppv_r:>10.4f}  {ppv_r - ppv:>+10.4f}')\nprint(f'  {\"F1\":<20s}  {f1:>10.4f}  {f1_r:>10.4f}  {f1_r - f1:>+10.4f}')\nprint(f'  {\"Threshold\":<20s}  {best_patient_thr:>10.4f}  {recal_thr:>10.4f}  {recal_thr - best_patient_thr:>+10.4f}')\nprint(f'  {\"FP (unnec. escal.)\":<20s}  {fp:>10d}  {fp_r:>10d}  {fp_r - fp:>+10d}')\nprint(f'  {\"FN (missed)\":<20s}  {fn:>10d}  {fn_r:>10d}  {fn_r - fn:>+10d}')\nprint(f'  {\"Miss rate (FNR)\":<20s}  {fn/max(fn+tp,1):>10.4f}  {fn_r/max(fn_r+tp_r,1):>10.4f}  {fn_r/max(fn_r+tp_r,1) - fn/max(fn+tp,1):>+10.4f}')\nprint(f'{\"═\" * 65}')\n\nif ppv_r < 0.60:\n    print(f'\\n  ⚠ WARNING: PPV={ppv_r:.4f} is below 0.60 — specificity may be')\n    print(f'    too low for practical deployment. Consider relaxing sensitivity floor.')\nelse:\n    print(f'\\n  ✓ PPV={ppv_r:.4f} remains above 0.60 clinical utility threshold.')\n    print(f'  ✓ Miss rate reduced from {fn/max(fn+tp,1):.1%} to {fn_r/max(fn_r+tp_r,1):.1%}.')\n\n# Save recalibration results\nimport joblib\njoblib.dump(ir_patient, '/kaggle/working/patient_isotonic_regressor.pkl')\n\nrecal_results = {\n    'method': 'isotonic_5fold_cv',\n    'threshold_strategy': f'sensitivity_constrained_{SENS_FLOOR:.2f}',\n    'sensitivity_floor': SENS_FLOOR,\n    'ece_before': round(patient_ece_before, 5),\n    'ece_after':  round(patient_ece_after, 5),\n    'auc_before': round(best_patient_auc, 5),\n    'auc_after':  round(recal_auc, 5),\n    'threshold':  round(recal_thr, 4),\n    'sensitivity': round(sens_r, 4),\n    'specificity': round(spec_r, 4),\n    'ppv': round(ppv_r, 4),\n    'f1':  round(f1_r, 4),\n    'false_positives': int(fp_r),\n    'false_negatives': int(fn_r),\n    'miss_rate_fnr': round(fn_r / max(fn_r + tp_r, 1), 4),\n    'youden_comparison': {\n        'threshold': round(youden_thr, 4),\n        'sensitivity': round(y_sens, 4),\n        'specificity': round(y_spec, 4),\n        'ppv': round(y_ppv, 4),\n        'fn': y_fn,\n        'note': 'Youden maximises Sens+Spec — not appropriate for screening',\n    },\n    'tradeoff_table': tradeoff_rows,\n}\nwith open('/kaggle/working/patient_recalibration.json', 'w') as f:\n    json.dump(recal_results, f, indent=2)\n\nprint('\\nSaved: patient_isotonic_regressor.pkl, patient_recalibration.json, patient_recalibration.png')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-21T04:27:22.45664Z","iopub.execute_input":"2026-02-21T04:27:22.456912Z","iopub.status.idle":"2026-02-21T04:27:44.058833Z","shell.execute_reply.started":"2026-02-21T04:27:22.456887Z","shell.execute_reply":"2026-02-21T04:27:44.058327Z"}},"outputs":[],"execution_count":null},{"id":"f3dda7b1","cell_type":"code","source":"# ── 12. Training Stability Summary ─────────────────────────────────────────\n# Load training metrics log from NB03 to demonstrate convergence.\n\nmetrics_log = pd.read_csv(METRICS_LOG_PATH)\n\nprint('Training History (all epochs):')\nprint(metrics_log[['epoch', 'train_loss', 'val_loss', 'val_auc',\n                    'sensitivity', 'specificity', 'f1']].to_string(index=False))\n\n# ── Convergence evidence: final epochs ───────────────────────────────────\n# Show AUC stability in the plateau region\nplateau_start = max(0, len(metrics_log) - 7)  # last 7 epochs\nplateau = metrics_log.iloc[plateau_start:]\n\nprint(f'\\n{\"═\" * 55}')\nprint(f'CONVERGENCE ANALYSIS (epochs {plateau.iloc[0][\"epoch\"]:.0f}–'\n      f'{plateau.iloc[-1][\"epoch\"]:.0f})')\nprint(f'{\"═\" * 55}')\n\nprint(f'\\n  {\"Epoch\":>5s}  {\"AUC\":>8s}  {\"Val Loss\":>9s}  {\"ΔAUC\":>8s}')\nprint(f'  {\"─\"*5}  {\"─\"*8}  {\"─\"*9}  {\"─\"*8}')\nprev_auc = None\nfor _, row in plateau.iterrows():\n    delta = f'{row[\"val_auc\"] - prev_auc:+.5f}' if prev_auc is not None else '    —'\n    print(f'  {row[\"epoch\"]:5.0f}  {row[\"val_auc\"]:8.5f}  {row[\"val_loss\"]:9.5f}  {delta:>8s}')\n    prev_auc = row['val_auc']\n\nauc_std = plateau['val_auc'].std()\nauc_range = plateau['val_auc'].max() - plateau['val_auc'].min()\nbest_epoch = metrics_log.loc[metrics_log['val_auc'].idxmax(), 'epoch']\n\nprint(f'\\n  AUC std  : {auc_std:.5f}')\nprint(f'  AUC range: {auc_range:.5f}')\nprint(f'  Best epoch: {best_epoch:.0f}')\nprint(f'\\n  Conclusion: Model converged between epochs '\n      f'{plateau.iloc[0][\"epoch\"]:.0f}–{plateau.iloc[-1][\"epoch\"]:.0f}.')\nprint(f'  AUC oscillates within ±{auc_range/2:.4f}, indicating the '\n      f'architecture has saturated.')\n\n# ── Learning curves plot ──────────────────────────────────────────────────\nfig, axes = plt.subplots(1, 3, figsize=(15, 4))\n\naxes[0].plot(metrics_log['epoch'], metrics_log['train_loss'], 'o-', label='Train')\naxes[0].plot(metrics_log['epoch'], metrics_log['val_loss'], 's-', label='Val')\naxes[0].set(title='Loss', xlabel='Epoch', ylabel='BCE Loss')\naxes[0].legend()\n\naxes[1].plot(metrics_log['epoch'], metrics_log['val_auc'], 'o-', color='tab:orange')\naxes[1].axhline(metrics_log['val_auc'].max(), color='red', linestyle='--',\n                alpha=0.5, label=f'Best={metrics_log[\"val_auc\"].max():.5f}')\naxes[1].fill_between(plateau['epoch'], plateau['val_auc'].min(),\n                     plateau['val_auc'].max(), alpha=0.15, color='orange',\n                     label='Plateau region')\naxes[1].set(title='Validation AUC', xlabel='Epoch', ylabel='AUC')\naxes[1].set_ylim([max(0.8, metrics_log['val_auc'].min() - 0.02), 1.0])\naxes[1].legend(fontsize=8)\n\naxes[2].plot(metrics_log['epoch'], metrics_log['sensitivity'], 'o-', label='Sensitivity')\naxes[2].plot(metrics_log['epoch'], metrics_log['specificity'], 's-', label='Specificity')\naxes[2].plot(metrics_log['epoch'], metrics_log['f1'], '^-', label='F1')\naxes[2].set(title='Metrics Over Training', xlabel='Epoch')\naxes[2].legend(fontsize=8)\n\nplt.suptitle('Training Convergence', fontsize=13)\nplt.tight_layout()\nplt.savefig('/kaggle/working/training_convergence.png', bbox_inches='tight')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-21T04:27:44.059919Z","iopub.execute_input":"2026-02-21T04:27:44.060297Z","iopub.status.idle":"2026-02-21T04:27:44.924821Z","shell.execute_reply.started":"2026-02-21T04:27:44.060273Z","shell.execute_reply":"2026-02-21T04:27:44.924099Z"}},"outputs":[],"execution_count":null},{"id":"6406e072","cell_type":"code","source":"# ── 13. Capacity Ceiling Reflection ────────────────────────────────────────\n\nbest_auc = metrics_log['val_auc'].max()\nbest_ep  = int(metrics_log.loc[metrics_log['val_auc'].idxmax(), 'epoch'])\n\nmiss_rate_raw = fn / max(fn + tp, 1)\nmiss_rate_recal = fn_r / max(fn_r + tp_r, 1)\n\ncapacity_text = f\"\"\"\n{'═' * 65}\nARCHITECTURAL CAPACITY ANALYSIS\n{'═' * 65}\n\nModel             : EfficientNet-B0 (5.3M parameters)\nBest slice AUC    : {best_auc:.5f} (epoch {best_ep})\nPatient-level AUC : {best_patient_auc:.5f} ({best_patient_strategy} aggregation)\nRecalibrated AUC  : {recal_auc:.5f} (isotonic 5-fold CV at patient level)\nTraining regime   : {len(metrics_log)} epochs, early stopped at epoch {int(metrics_log['epoch'].iloc[-1])}\n\nCAPACITY CEILING:\n  EfficientNet-B0 saturates around ~{best_auc:.3f} AUC under our current\n  preprocessing (3-channel CT windowing) and training scheme (binary \"any\"\n  label, slice-level, weighted sampling, discriminative LR).\n\n  Evidence: AUC oscillated within ±{auc_range/2:.4f} over the final\n  {len(plateau)} epochs with no monotonic improvement. Additional epochs\n  would not improve performance — the architecture has reached its limit.\n\nCALIBRATION ASSESSMENT:\n  Temperature scaling was chosen for deployment despite worsening global\n  ECE ({raw_ece:.5f} → {cal_ece:.5f}). Rationale:\n    • High-confidence band ECE improved ({hc_ece_raw:.5f} → {hc_ece_cal:.5f})\n    • Brier score improved ({raw_brier:.5f} → {cal_brier:.5f})\n    • For triage, HC-band reliability matters more than global ECE\n  Patient-level recalibration via isotonic regression (5-fold CV) was\n  applied to correct max-pooling probability inflation.\n\nTHRESHOLD SELECTION:\n  Youden index was rejected for patient-level threshold because it\n  maximises Sens+Spec equally — appropriate for diagnosis, NOT screening.\n\n  Instead: sensitivity-constrained threshold (≥{SENS_FLOOR:.0%}) was chosen.\n    • Miss rate: {miss_rate_raw:.1%} (Youden, raw) → {miss_rate_recal:.1%} (sens-constrained, recal)\n    • PPV remains ≥ 0.60 → system is clinically usable\n    • FP increase is acceptable: unnecessary reviews, not patient harm\n\nKNOWN LIMITATIONS:\n  1. Slice-level only: no spatial context across adjacent CT slices\n  2. Binary \"any\" label: does not model 5 hemorrhage subtypes\n  3. Single split: AUC estimate has variance ~±0.005 without cross-validation\n  4. Global ECE worsened after temp scaling (HC-band improved — acceptable)\n  5. {int(fp_r)} false positive patient escalations at selected threshold\n\nPATHS TO IMPROVEMENT (future work, separate from this project):\n  ┌──────────────────────────────────┬────────────┬──────────────────┐\n  │ Approach                         │ Effort     │ Expected Gain    │\n  ├──────────────────────────────────┼────────────┼──────────────────┤\n  │ Stronger backbone (B3/B4/B7)    │ Medium     │ +0.5–1.0% AUC   │\n  │ Advanced augmentation (MixUp)   │ Low        │ +0.2–0.5%        │\n  │ Multi-label (5 subtypes + any)  │ High       │ +1–2%            │\n  │ Patient-level aggregation+train │ Medium     │ +0.5–1.0%        │\n  │ Ensemble (2–3 architectures)    │ Medium     │ +0.3–0.8%        │\n  │ Cross-validation (5-fold)       │ High       │ Variance estimate│\n  └──────────────────────────────────┴────────────┴──────────────────┘\n\n  These improvements are planned as an enhancement project, separate from\n  the current major project scope.\n{'═' * 65}\n\"\"\"\nprint(capacity_text)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-21T04:27:44.925765Z","iopub.execute_input":"2026-02-21T04:27:44.926071Z","iopub.status.idle":"2026-02-21T04:27:44.934002Z","shell.execute_reply.started":"2026-02-21T04:27:44.926037Z","shell.execute_reply":"2026-02-21T04:27:44.933388Z"}},"outputs":[],"execution_count":null},{"id":"2d52690f","cell_type":"code","source":"# ── 14. Deployment Decision ────────────────────────────────────────────────\n# One clean declaration of the final selected configuration.\n\ndeployment_config = {\n    'model': {\n        'architecture'     : ARCH,\n        'parameters'       : '5.3M',\n        'best_epoch'       : best_ep,\n        'checkpoint'       : 'best_model.pth',\n    },\n    'slice_calibration': {\n        'method'           : CALIB_METHOD,\n        'temperature'      : TEMPERATURE,\n        'global_ece'       : {'raw': round(raw_ece, 5), 'calibrated': round(cal_ece, 5),\n                              'effect': 'Worsened (+0.026) — acceptable tradeoff'},\n        'hc_band_ece'      : {'raw': round(hc_ece_raw, 5), 'calibrated': round(hc_ece_cal, 5),\n                              'effect': 'Improved (critical for triage)'},\n        'brier_score'      : {'raw': round(raw_brier, 5), 'calibrated': round(cal_brier, 5),\n                              'effect': 'Improved'},\n        'note'             : ('Temperature scaling worsened global ECE slightly but improved '\n                              'high-confidence band ECE and Brier score. For triage deployment, '\n                              'high-confidence calibration is the critical metric.'),\n    },\n    'patient_calibration': {\n        'method'           : 'Isotonic regression (5-fold CV)',\n        'threshold_strategy': f'Sensitivity-constrained (≥{SENS_FLOOR:.0%})',\n        'ece_before'       : round(patient_ece_before, 5),\n        'ece_after'        : round(patient_ece_after, 5),\n        'threshold'        : round(recal_thr, 4),\n    },\n    'thresholds': {\n        'slice_threshold'  : BASE_THRESHOLD,\n        'patient_threshold': round(recal_thr, 4),\n        'threshold_strategy': f'Sensitivity ≥ {SENS_FLOOR:.0%}, then max specificity',\n        'high_conf_band'   : HIGH_THRESHOLD,\n        'low_conf_band'    : LOW_THRESHOLD,\n    },\n    'aggregation': {\n        'strategy'         : best_patient_strategy,\n        'patient_auc'      : round(best_patient_auc, 5),\n    },\n    'performance': {\n        'slice_auc'          : round(slice_auc, 5),\n        'patient_auc'        : round(best_patient_auc, 5),\n        'patient_auc_recal'  : round(recal_auc, 5),\n        'patient_sensitivity': round(sens_r, 4),\n        'patient_specificity': round(spec_r, 4),\n        'patient_ppv'        : round(ppv_r, 4),\n        'patient_f1'         : round(f1_r, 4),\n        'patient_fn'         : int(fn_r),\n        'patient_fp'         : int(fp_r),\n        'miss_rate_fnr'      : round(fn_r / max(fn_r + tp_r, 1), 4),\n    },\n    'caveats': [\n        'Single train/val split — no cross-validation variance estimate',\n        'Temperature scaling worsened global ECE but improved HC-band ECE and Brier',\n        'Binary \"any\" label — no subtype classification',\n        'Not cleared for standalone diagnostic use',\n    ],\n}\n\n# Pretty-print deployment card\nmiss_rate_r = fn_r / max(fn_r + tp_r, 1)\nmiss_rate_raw = fn / max(fn + tp, 1)\n\nprint(f\"\"\"\n{'█' * 65}\n  FINAL DEPLOYMENT CONFIGURATION\n{'█' * 65}\n\n  Model\n    Architecture       : {ARCH}\n    Best epoch         : {best_ep}\n    Checkpoint         : best_model.pth\n\n  Slice-Level Calibration\n    Method             : {CALIB_METHOD.title()} Scaling (T={TEMPERATURE:.4f})\n    Global ECE         : {raw_ece:.5f} → {cal_ece:.5f} (worsened; acceptable)\n    HC-band ECE        : {hc_ece_raw:.5f} → {hc_ece_cal:.5f} (improved; critical)\n    Brier              : {raw_brier:.5f} → {cal_brier:.5f} (improved)\n\n  Patient-Level Recalibration\n    Method             : Isotonic Regression (5-fold CV)\n    Threshold strategy : Sensitivity ≥ {SENS_FLOOR:.0%}, then max specificity\n    ECE                : {patient_ece_before:.5f} → {patient_ece_after:.5f}\n\n  Thresholds\n    Slice threshold    : {BASE_THRESHOLD:.4f} (Youden — slice-level)\n    Patient threshold  : {recal_thr:.4f} (sens-constrained, recalibrated)\n    High-conf band     : ≥ {HIGH_THRESHOLD:.2f}\n    Low-conf band      : < {LOW_THRESHOLD:.2f}\n\n  Patient-Level Aggregation\n    Strategy           : {best_patient_strategy}\n\n  Performance Summary\n    ┌────────────────────┬─────────────┬──────────────┬──────────────┐\n    │ Metric             │ Slice-Level │ Patient (raw)│ Patient (cal)│\n    ├────────────────────┼─────────────┼──────────────┼──────────────┤\n    │ AUC                │ {slice_auc:>11.5f} │ {best_patient_auc:>12.5f} │ {recal_auc:>12.5f} │\n    │ Sensitivity        │      —      │ {sens:>12.4f} │ {sens_r:>12.4f} │\n    │ Specificity        │      —      │ {spec:>12.4f} │ {spec_r:>12.4f} │\n    │ PPV                │      —      │ {ppv:>12.4f} │ {ppv_r:>12.4f} │\n    │ F1                 │      —      │ {f1:>12.4f} │ {f1_r:>12.4f} │\n    │ FP (unnec. escal.) │      —      │ {fp:>12d} │ {fp_r:>12d} │\n    │ FN (missed)        │      —      │ {fn:>12d} │ {fn_r:>12d} │\n    │ Miss rate (FNR)    │      —      │ {miss_rate_raw:>11.1%} │ {miss_rate_r:>11.1%} │\n    └────────────────────┴─────────────┴──────────────┴──────────────┘\n\n  Caveats\n    • Single split — AUC variance ~±0.005 without CV\n    • Temp scaling worsened global ECE; improved HC-band ECE + Brier\n    • Binary \"any\" label — 5 subtypes not modeled\n    • Screening aid only — not for standalone diagnosis\n\n{'█' * 65}\n\"\"\")\n\n# Save deployment config\nwith open('/kaggle/working/deployment_config.json', 'w') as f:\n    json.dump(deployment_config, f, indent=2)\n\nprint('Saved: deployment_config.json')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-21T04:27:44.934946Z","iopub.execute_input":"2026-02-21T04:27:44.935178Z","iopub.status.idle":"2026-02-21T04:27:44.951452Z","shell.execute_reply.started":"2026-02-21T04:27:44.935152Z","shell.execute_reply":"2026-02-21T04:27:44.950636Z"}},"outputs":[],"execution_count":null},{"id":"7ec8fcb9","cell_type":"code","source":"# ── 15. Cleanup + final output summary ───────────────────────────────────\ngrad_cam.remove()\n\nimport glob\nreport_files = glob.glob(str(REPORTS_DIR / '*.json'))\nprint(f'\\nTotal JSON reports saved: {len(report_files)}')\nprint(f'Grad-CAM overlays saved : {len(glob.glob(str(REPORTS_DIR / \"*_gradcam.png\")))}')\nprint(f'Summary CSV             : /kaggle/working/report_summary.csv')\nprint(f'Patient-level results   : /kaggle/working/patient_level_results.json')\nprint(f'Patient recalibration   : /kaggle/working/patient_recalibration.json')\nprint(f'Patient isotonic model  : /kaggle/working/patient_isotonic_regressor.pkl')\nprint(f'Deployment config       : /kaggle/working/deployment_config.json')\nprint(f'Training convergence    : /kaggle/working/training_convergence.png')\nprint(f'Patient-level analysis  : /kaggle/working/patient_level_analysis.png')\nprint(f'Patient recalibration   : /kaggle/working/patient_recalibration.png')\nprint()\nprint('All notebooks complete. Full pipeline:')\nprint('  01_eda.ipynb              → Dataset exploration + windowing demo')\nprint('  02_preprocess_cache.ipynb → DICOM → NPY cache (commit output)')\nprint('  03_train_session.ipynb    → Training with session chaining')\nprint('  04_ablations.ipynb        → Architecture + preprocessing + augmentation ablations')\nprint('  05_gradcam.ipynb          → Grad-CAM heatmaps + occlusion sanity check')\nprint('  06_calibration.ipynb      → Temperature scaling + 3-band triage system')\nprint('  07_report.ipynb           → Clinical reports + patient-level analysis (this notebook)')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-21T04:27:44.952539Z","iopub.execute_input":"2026-02-21T04:27:44.952815Z","iopub.status.idle":"2026-02-21T04:27:44.967965Z","shell.execute_reply.started":"2026-02-21T04:27:44.952791Z","shell.execute_reply":"2026-02-21T04:27:44.967243Z"}},"outputs":[],"execution_count":null},{"id":"46f17447","cell_type":"code","source":"# ── HEALTH CHECK — automated output validation ────────────────────────────\n\nerrors = []\n\n# Check reports directory\nimport glob as _glob\njson_reports = _glob.glob(str(REPORTS_DIR / '*.json'))\ncam_files    = _glob.glob(str(REPORTS_DIR / '*_gradcam.png'))\n\nif len(json_reports) < N_REPORTS:\n    errors.append(f'Only {len(json_reports)} JSON reports — expected {N_REPORTS}')\n\nif len(cam_files) < N_REPORTS:\n    errors.append(f'Only {len(cam_files)} Grad-CAM overlays — expected {N_REPORTS}')\n\n# Check summary CSV\nif not os.path.exists('/kaggle/working/report_summary.csv'):\n    errors.append('report_summary.csv is MISSING')\n\n# Check patient-level results\nif not os.path.exists('/kaggle/working/patient_level_results.json'):\n    errors.append('patient_level_results.json is MISSING')\nelse:\n    with open('/kaggle/working/patient_level_results.json') as f:\n        pl = json.load(f)\n    if pl.get('patient_auc', 0) < 0.5:\n        errors.append(f'Patient-level AUC={pl[\"patient_auc\"]:.4f} is suspiciously low')\n\n# Check patient-level recalibration outputs\nfor recal_file in ['patient_recalibration.json', 'patient_isotonic_regressor.pkl']:\n    if not os.path.exists(f'/kaggle/working/{recal_file}'):\n        errors.append(f'{recal_file} is MISSING')\n\nif os.path.exists('/kaggle/working/patient_recalibration.json'):\n    with open('/kaggle/working/patient_recalibration.json') as f:\n        recal = json.load(f)\n    if recal.get('ece_after', 1.0) > recal.get('ece_before', 0.0) + 0.05:\n        errors.append(f'Patient recalibration ECE worsened significantly: '\n                      f'{recal[\"ece_before\"]:.4f} → {recal[\"ece_after\"]:.4f}')\n\n# Check deployment config\nif not os.path.exists('/kaggle/working/deployment_config.json'):\n    errors.append('deployment_config.json is MISSING')\n\n# Check new plots\nfor plot in ['patient_level_analysis.png', 'training_convergence.png',\n             'patient_recalibration.png']:\n    if not os.path.exists(f'/kaggle/working/{plot}'):\n        errors.append(f'Missing plot: {plot}')\n\n# Validate report structure (sample one)\nif json_reports:\n    with open(json_reports[0]) as f:\n        sample_rep = json.load(f)\n    required_keys = ['report_id', 'image_id', 'prediction', 'triage', 'disclaimer']\n    for k in required_keys:\n        if k not in sample_rep:\n            errors.append(f'Report missing key: {k}')\n    if 'disclaimer' in sample_rep and len(sample_rep['disclaimer']) < 50:\n        errors.append('Disclaimer text is too short — possible truncation')\n\nhealth = {\n    'notebook'                 : '07_report',\n    'status'                   : 'PASS' if not errors else 'FAIL',\n    'errors'                   : errors,\n    'n_reports'                : len(json_reports),\n    'n_gradcam'                : len(cam_files),\n    'summary_csv'              : os.path.exists('/kaggle/working/report_summary.csv'),\n    'patient_level_auc'        : round(best_patient_auc, 5),\n    'patient_aggregation'      : best_patient_strategy,\n    'slice_auc'                : round(slice_auc, 5),\n    'patient_recalibration'    : os.path.exists('/kaggle/working/patient_recalibration.json'),\n    'deployment_config_saved'  : os.path.exists('/kaggle/working/deployment_config.json'),\n}\n\nwith open('/kaggle/working/health_check_nb07.json', 'w') as f:\n    json.dump(health, f, indent=2)\n\nif errors:\n    print('❌ HEALTH CHECK FAILED:')\n    for e in errors:\n        print(f'   • {e}')\nelse:\n    print('✅ HEALTH CHECK PASSED')\n    print(f'   Reports              : {len(json_reports)} JSON + {len(cam_files)} Grad-CAM')\n    print(f'   Summary CSV          : saved')\n    print(f'   Slice-level AUC      : {slice_auc:.5f}')\n    print(f'   Patient-level AUC    : {best_patient_auc:.5f} ({best_patient_strategy})')\n    print(f'   Patient recalibration: saved (isotonic 5-fold CV)')\n    print(f'   Deployment config    : saved')\n    print(f'   Pipeline             : COMPLETE ✓')\n    print()\n    print('All 7 notebooks have completed successfully.')\n    print('Check health_check_nb*.json files for per-notebook validation results.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-21T04:27:44.968998Z","iopub.execute_input":"2026-02-21T04:27:44.969595Z","iopub.status.idle":"2026-02-21T04:27:44.985676Z","shell.execute_reply.started":"2026-02-21T04:27:44.969567Z","shell.execute_reply":"2026-02-21T04:27:44.984932Z"}},"outputs":[],"execution_count":null}]}