{"cells":[{"cell_type":"markdown","id":"e8abb9f2","metadata":{},"source":"# RSNA Knee V3 report-image teacher gate\n\nThis is a validation and checkpoint-training notebook, not a submission notebook. It\ntrains a DINOv2-base six-slot model against a report/image-OOF teacher, selects the epoch\non a fixed 250-study report holdout, and opens the 58-study image-label gold set only once\nat the final audit. A checkpoint is eligible for inference work only when the final audit\nsays `kaggle_blend_allowed=true`."},{"cell_type":"code","execution_count":null,"id":"3a3db754","metadata":{},"outputs":[],"source":"from __future__ import annotations\n\nimport json\nimport os\nimport shutil\nimport subprocess\nimport sys\nimport time\nfrom pathlib import Path\n\nimport torch\n\n\nSEED = 6203\nDEBUG = False\nEPOCHS = 12\nSTEPS_PER_EPOCH = 900\nCACHE_WORKERS = 4\n\nWORK = Path(\"/kaggle/working\")\nTEMP = Path(\"/kaggle/temp\")\nCACHE = TEMP / \"rsna-knee-v3-six-slot-cache\"\nRUN = WORK / f\"v3_gate_seed{SEED}\"\nTEMP.mkdir(parents=True, exist_ok=True)\nWORK.mkdir(parents=True, exist_ok=True)\n\n\ndef log(message: str) -> None:\n    print(time.strftime(\"[%Y-%m-%d %H:%M:%S]\"), message, flush=True)\n\n\ndef run(command: list[str]) -> None:\n    log(\"RUN \" + \" \".join(command))\n    subprocess.run(command, check=True)\n\n\ndef find_competition_root() -> Path:\n    candidates = [\n        Path(\"/kaggle/input/competitions/rsna-knee-abnormality-detection\"),\n        Path(\"/kaggle/input/rsna-knee-abnormality-detection\"),\n    ]\n    for path in candidates:\n        if (path / \"train.csv\").is_file() and (path / \"train_series.csv\").is_file():\n            return path\n    raise FileNotFoundError(\"Attach the RSNA Knee Abnormality Detection competition\")\n\n\ndef find_asset(filename: str) -> Path:\n    preferred = Path(\"/kaggle/input/rsna-knee-v3-gate-assets\") / filename\n    if preferred.is_file():\n        return preferred\n    for root, directories, files in os.walk(\"/kaggle/input\"):\n        directories[:] = [\n            name for name in directories if name not in (\"train_series\", \"test_series\")\n        ]\n        if filename in files:\n            return Path(root) / filename\n    raise FileNotFoundError(\n        f\"Missing {filename}. Attach the private rsna-knee-v3-gate-assets Dataset.\"\n    )\n\n\ndef find_dinov2_base() -> Path:\n    candidates = [\n        Path(\"/kaggle/input/models/metaresearch/dinov2/pytorch/base/1\"),\n        Path(\"/kaggle/input/dinov2/pytorch/base/1\"),\n    ]\n    for path in candidates:\n        if (path / \"config.json\").is_file():\n            return path\n    for root, directories, files in os.walk(\"/kaggle/input\"):\n        directories[:] = [\n            name for name in directories if name not in (\"train_series\", \"test_series\")\n        ]\n        if \"config.json\" in files and \"dinov2\" in root.lower() and \"base\" in root.lower():\n            return Path(root)\n    raise FileNotFoundError(\n        \"Attach Kaggle Model metaresearch/dinov2, framework PyTorch, variation base, version 1\"\n    )\n\n\nif not torch.cuda.is_available():\n    raise RuntimeError(\"Enable a Kaggle GPU accelerator before running this notebook\")\n\nROOT = find_competition_root()\nMODEL = find_dinov2_base()\nBUILDER = find_asset(\"build_kaggle_cache.py\")\nTRAINER = find_asset(\"train_kaggle_v3.py\")\nAUDITOR = find_asset(\"audit_locked_gold.py\")\nLABELS = find_asset(\"report_image_oof_teacher.csv\")\nSPLIT = find_asset(\"split.csv\")\n\nlog(f\"GPU={torch.cuda.get_device_name(0)} torch={torch.__version__}\")\nlog(f\"competition={ROOT}\")\nlog(f\"model={MODEL}\")\nlog(f\"assets={LABELS.parent}\")\nlog(f\"temporary free={shutil.disk_usage(TEMP).free / 1024**3:.1f} GiB\")"},{"cell_type":"markdown","id":"0e29a5e1","metadata":{},"source":"## Build temporary cache\n\nThe competition input remains read-only. Every `.npz` cache file is written under\n`/kaggle/temp`, so the read-only-filesystem failure from earlier notebooks cannot recur."},{"cell_type":"code","execution_count":null,"id":"fbc62d83","metadata":{},"outputs":[],"source":"cache_command = [\n    sys.executable,\n    \"-u\",\n    str(BUILDER),\n    \"--competition\",\n    str(ROOT),\n    \"--output\",\n    str(CACHE),\n    \"--workers\",\n    str(CACHE_WORKERS),\n]\nif DEBUG:\n    raise RuntimeError(\"DEBUG cache slicing is intentionally unsupported; run the real gate\")\nrun(cache_command)"},{"cell_type":"markdown","id":"78fd9bdf","metadata":{},"source":"## Train and select on the fixed holdout"},{"cell_type":"code","execution_count":null,"id":"a41eadb0","metadata":{},"outputs":[],"source":"RUN.mkdir(parents=True, exist_ok=True)\ntrain_command = [\n    sys.executable,\n    \"-u\",\n    str(TRAINER),\n    \"--cache\",\n    str(CACHE),\n    \"--model\",\n    str(MODEL),\n    \"--labels\",\n    str(LABELS),\n    \"--split\",\n    str(SPLIT),\n    \"--output\",\n    str(RUN),\n    \"--seed\",\n    str(SEED),\n    \"--epochs\",\n    str(EPOCHS),\n    \"--steps-per-epoch\",\n    str(STEPS_PER_EPOCH),\n    \"--batch-studies\",\n    \"2\",\n    \"--groups-per-train\",\n    \"2\",\n    \"--eval-batch\",\n    \"2\",\n    \"--unfreeze-last\",\n    \"6\",\n    \"--lr-backbone\",\n    \"5e-6\",\n    \"--lr-head\",\n    \"3e-4\",\n    \"--weight-decay\",\n    \"0.05\",\n    \"--rank-weight\",\n    \"0.12\",\n    \"--label-smoothing\",\n    \"0.01\",\n    \"--ema-decay\",\n    \"0.995\",\n    \"--softpool-beta\",\n    \"6.0\",\n    \"--encode-chunk\",\n    \"8\",\n    \"--io-workers\",\n    \"4\",\n]\nif (RUN / \"last.pt\").is_file():\n    train_command.append(\"--resume\")\nrun(train_command)"},{"cell_type":"markdown","id":"790ced7b","metadata":{},"source":"## Locked-gold audit"},{"cell_type":"code","execution_count":null,"id":"f284472f","metadata":{},"outputs":[],"source":"run(\n    [\n        sys.executable,\n        \"-u\",\n        str(AUDITOR),\n        \"--run\",\n        str(RUN),\n        \"--train-csv\",\n        str(ROOT / \"train.csv\"),\n    ]\n)\n\nsummary = json.loads((RUN / \"training_summary.json\").read_text())\naudit = json.loads((RUN / \"locked_gold_audit.json\").read_text())\ndecision = {\n    \"best_holdout_auc\": summary[\"best_holdout_auc\"],\n    \"best_epoch\": summary[\"best_epoch\"],\n    \"locked_gold_macro_auc\": audit[\"macro_auc\"],\n    \"kaggle_blend_allowed\": audit[\"kaggle_blend_allowed\"],\n    \"next_step\": (\n        \"package checkpoint and test only target-level OOF-selected blends\"\n        if audit[\"kaggle_blend_allowed\"]\n        else \"reject checkpoint; keep the 0.936 public parent\"\n    ),\n}\n(WORK / \"V3_GATE_DECISION.json\").write_text(json.dumps(decision, indent=2) + \"\\n\")\nprint(json.dumps(decision, indent=2))\n\n# Do not persist the 20+ GiB cache as a notebook output.\nshutil.rmtree(CACHE, ignore_errors=True)\nif (RUN / \"last.pt\").is_file():\n    (RUN / \"last.pt\").unlink()\nlog(\"removed temporary cache and resumable optimizer checkpoint\")"},{"cell_type":"markdown","id":"f9fc92d4","metadata":{},"source":"The notebook intentionally does not create `submission.csv`. A locked-gold pass is a\nprerequisite for building inference, not evidence that a blind leaderboard blend wins."}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"}},"nbformat":4,"nbformat_minor":5}