{"cells":[{"cell_type":"code","execution_count":null,"metadata":{},"outputs":[],"source":"# RSNA Knee — band32 adjacency leg, efficiency track\n#\n# One DINOv2-small slot-attention family, 10 seeds, on a crop130 / 224 px / 32-slice cache.\n# The 32-slice geometry is the point: from a 30-slice series a 16-slice cache makes the three\n# input channels every-OTHER slice, i.e. spatially aliased, and fixing that was worth +0.0060\n# at 10/10 seeds — the largest confirmed effect in this project's log.\n#\n# Runtime is the other half of the score here: ~1.0 s/study for one member plus 0.14 s/study\n# per extra seed, measured. Against the 9 h cap on 1,322 studies that is a ~21x margin.\nimport os\nimport sys\n\nos.environ.setdefault(\"HF_HUB_OFFLINE\", \"1\")\nos.environ.setdefault(\"TRANSFORMERS_OFFLINE\", \"1\")\nos.environ[\"RSNA_OUT\"] = \"/kaggle/working/submission.csv\"\nos.environ[\"RSNA_SPLIT\"] = \"test\"\n"},{"cell_type":"code","execution_count":null,"metadata":{},"outputs":[],"source":"# Member count, DECLARED not inferred — Sec 8.81. It comes from the weights dataset's own\n# members.json, so the guard compares the manifest against what is found on disk. Depth is\n# searched rather than assumed: this environment mounts datasets at\n# /kaggle/input/datasets/<owner>/<slug>/, four levels down, and a single-depth glob found\n# nothing (Sec 8.147e).\nimport glob as _eff_glob\nimport json as _eff_json\nfrom pathlib import Path as _EffPath\n\n_EFF_IN = os.environ.get(\"KAGGLE_INPUT_ROOT\", \"/kaggle/input\")\n_eff_manifests = []\nfor _eff_d in range(1, 6):\n    _eff_manifests += sorted(_eff_glob.glob(_EFF_IN + \"/\" + \"*/\" * _eff_d + \"members.json\"))\n_eff_manifests = [p for p in _eff_manifests if \"series\" not in p]\nif not _eff_manifests:\n    raise SystemExit(\"[X] no members.json in the inputs — attach the weights dataset\")\n_eff_n = len(_eff_json.loads(\n    _EffPath(_eff_manifests[0]).read_text(encoding=\"utf-8\"))[\"members\"])\nos.environ[\"RSNA_N_EXPECTED\"] = str(_eff_n)\nprint(f\"[eff] {_eff_manifests[0]} declares {_eff_n} member(s)\", flush=True)\n"},{"cell_type":"code","execution_count":null,"metadata":{},"outputs":[],"source":"\"\"\"Submission notebook v3 — fine-tuned DINOv2 + per-diagnosis slot attention.\n\nFeature extraction here MUST match `kaggle_train_dinov2.py` exactly — same slice\ncount, same CLS+mean pooling, same mean+max over slices, same slot order, same\nstandardisation. A mismatch produces plausible-looking predictions computed from\nfeatures the head never saw, with no error anywhere.\n\"\"\"\n\nimport glob\nimport os\nimport re\nimport time\nfrom concurrent.futures import ThreadPoolExecutor\n\nimport numpy as np\nimport pandas as pd\nimport pydicom\n\nOUT = os.environ.get(\"RSNA_OUT\", \"/kaggle/working/submission.csv\")\nSPLIT = os.environ.get(\"RSNA_SPLIT\", \"test\")  # 'train' when testing locally\n\n\ndef find_data_dir():\n    \"\"\"Locate the competition data instead of hardcoding a mount path.\n\n    Kaggle's input folder name is not guaranteed to equal the competition slug, and\n    a wrong guess fails with a bare FileNotFoundError that says nothing useful. Search\n    for the directory that actually contains the split CSV, and if none does, print\n    what IS mounted so the cause is obvious.\n    \"\"\"\n    if os.environ.get(\"RSNA_DATA\"):\n        return os.environ[\"RSNA_DATA\"]\n\n    # Competition data can be mounted at /kaggle/input/<slug>/ or nested under\n    # /kaggle/input/competitions/<slug>/ — search both rather than betting on one.\n    root = os.environ.get(\"KAGGLE_INPUT_ROOT\", \"/kaggle/input\")\n    # Depth-bounded rather than fixed: Kaggle nests inputs by type and owner, and\n    # the depth has varied. Not recursive — that would walk millions of DICOMs.\n    candidates = []\n    for d in range(5):\n        candidates += sorted(glob.glob(f\"{root}/\" + \"*/\" * d))\n        for c in sorted(glob.glob(f\"{root}/\" + \"*/\" * d)):\n            if os.path.isdir(c) and os.path.exists(os.path.join(c, f\"{SPLIT}.csv\")):\n                print(f\"[data] found competition data at {c.rstrip('/')}\")\n                return c.rstrip(\"/\")\n\n    print(\"[X] Could not find the competition data.\")\n    print(f\"    Looked for '{SPLIT}.csv' two levels deep under {root}\")\n    if candidates:\n        print(\"    Mounted inputs:\")\n        for d in candidates:\n            if os.path.isdir(d):\n                print(f\"      {d}  ->  {sorted(os.listdir(d))[:8]}\")\n    else:\n        print(f\"    NOTHING is mounted under {root}.\")\n        print(\"    Fix: notebook editor -> Add Input -> Competitions -> \"\n              \"RSNA Knee Abnormality Detection\")\n    raise SystemExit(1)\n\n\nDATA = find_data_dir()\n\nLABEL_COLUMNS = [\n    \"ACL\", \"MCL\", \"Medial Meniscus\", \"Lateral Meniscus\", \"Medial OA\", \"Lateral OA\",\n    \"PF OA\", \"Effusion\", \"Synovitis\", \"Baker's\", \"Contusion\", \"Fracture\",\n]\nSLOTS = [\"Sagittal_fs\", \"Sagittal_nfs\", \"Coronal_fs\", \"Coronal_nfs\", \"Axial_fs\"]\n\nSIZE, N_SLICES, WORKERS = 224, 16, 4\nGROUP = 3\nSLICE_BAND = (0.20, 0.80)\nTTA = True\nCHUNK = 32                  # studies preprocessed+scored per pass; bounds memory\nFEAT_BATCH = 64             # slices per DINOv2 forward\n# Members this submission must load; see the count guard below. Sec 8.81 is why the guard\n# exists and why the number is DECLARED rather than inferred from what was found — a\n# loader that silently assembles 2 of 3 members produces a valid-looking submission that\n# scores wrong. It is settable so the same script can serve a different member count\n# (a parity check on one seed, a deployment on several) without editing the literal, but\n# it is never defaulted to len(MODELS).\nN_EXPECTED = int(os.environ.get(\"RSNA_N_EXPECTED\", \"3\"))\n# Diagnostic only: a directory to write each study's assembled volume into, for\n# scripts/diff_preprocess_paths.py. Unset in any real run.\nDUMP_VOL = os.environ.get(\"RSNA_DUMP_VOL\", \"\")\nif DUMP_VOL:\n    os.makedirs(DUMP_VOL, exist_ok=True)\n    print(f\"[!] RSNA_DUMP_VOL set — writing assembled volumes to {DUMP_VOL}\")\nCANONICAL_SIDE = \"R\"        # mirror left knees; medial/lateral swap between sides\nLAT_MIN_OFFSET_MM = 20.0\nCROP_CENTER = \"image\"       # OVERRIDDEN below from the checkpoint too\nCROP_MM = 130.0             # default; OVERRIDDEN below from the checkpoint's\n                            # pipeline_version, so inference cannot silently crop\n                            # differently from training (the §5.6 parity failure)\nRUNTIME_LIMIT_S = 9 * 3600\n\n# ---------------------------------------------------------------- preprocessing\n# Self-contained copy of src/data/dicom.py so no repo upload is needed.\n\n\ndef slice_position(ds):\n    \"\"\"Project ImagePositionPatient onto the slice normal.\n\n    Filenames are SOP-UID-based and carry no order — sorting by filename yields a\n    spatially scrambled volume (measured: adjacent-slice correlation 0.64 vs 0.88).\n    \"\"\"\n    iop = getattr(ds, \"ImageOrientationPatient\", None)\n    ipp = getattr(ds, \"ImagePositionPatient\", None)\n    if iop is None or ipp is None or len(iop) != 6 or len(ipp) != 3:\n        return None\n    row, col = np.array(iop[:3], float), np.array(iop[3:], float)\n    return float(np.dot(np.array(ipp, float), np.cross(row, col)))\n\n\ndef lr_axis(ds):\n    \"\"\"Which axis of `volume` (n_slices, H, W) runs medial-to-lateral.\n\n    Byte-for-byte the logic of `src/data/dicom.py: lr_axis`. It has to be here because\n    this file reimplements preprocessing rather than importing it, and without it the\n    mirror below flips the wrong axis on sagittal series: Sec 8.79 measured that a bare\n    `vol[:, :, ::-1]` puts the patella at the BACK of the knee on the 46.4% of studies\n    that are left knees, while leaving medial/lateral — the entire point of the mirror —\n    un-normalised. src/data/ was fixed; this file was not, so the shipped models were\n    being served sagittal volumes they had never been trained on.\n    \"\"\"\n    iop = getattr(ds, \"ImageOrientationPatient\", None)\n    if iop is None or len(iop) != 6:\n        return None\n    row = np.array(iop[:3], dtype=float)\n    normal = np.cross(row, np.array(iop[3:], dtype=float))\n    # x is the patient's left-right axis in DICOM's LPS convention.\n    if abs(row[0]) >= abs(normal[0]):\n        return \"columns\" if abs(row[0]) > 0.5 else None\n    return \"slices\" if abs(normal[0]) > 0.5 else None\n\n\ndef detect_side(headers):\n    \"\"\"'L'/'R'/None. Mirrors src/data/dicom.py — see its docstring for why.\"\"\"\n    for ds in headers:\n        for tag in (\"ImageLaterality\", \"Laterality\"):\n            v = getattr(ds, tag, None)\n            if v and str(v).strip().upper() in (\"L\", \"R\"):\n                return str(v).strip().upper()\n    xs = []\n    for ds in headers:\n        ipp = getattr(ds, \"ImagePositionPatient\", None)\n        iop = getattr(ds, \"ImageOrientationPatient\", None)\n        cols = int(getattr(ds, \"Columns\", 0) or 0)\n        sp = getattr(ds, \"PixelSpacing\", None)\n        if ipp is None or iop is None or not cols or sp is None:\n            continue\n        centre = np.array(ipp, float) + np.array(iop[:3], float) * float(sp[1]) * cols / 2.0\n        xs.append(centre[0])\n    if not xs:\n        return None\n    x = float(np.median(xs))\n    return None if abs(x) < LAT_MIN_OFFSET_MM else (\"L\" if x > 0 else \"R\")\n\n\ndef load_series(series_dir, size=SIZE, n_slices=None, crops=None):\n    \"\"\"Decode ONCE, then emit one volume per crop in `crops`.\n\n    Returns {crop_mm: uint8 volume}. Decoding dominates cost, so producing both\n    crops here rather than reading the series twice is most of why an ensemble is\n    affordable.\n\n    `n_slices` resolves at CALL time, not definition time. It was `n_slices=N_SLICES`,\n    which binds the value at import — before the checkpoint guard below can raise it. The\n    guard would then report \"training and inference now agree\" while preprocessing quietly\n    kept using 16, which is worse than having no guard at all.\n    \"\"\"\n    n_slices = N_SLICES if n_slices is None else n_slices\n    crops = crops or [CROP_MM]\n    import cv2\n\n    files = sorted(os.listdir(series_dir))\n    files = [f for f in files if f.endswith(\".dcm\")]\n    if not files:\n        return None\n\n    # Two-pass read, matching src/data/dicom.py exactly. Pass 1 reads geometry tags\n    # only (no pixel decode) to establish slice order; pass 2 decodes just the slices\n    # kept. Roughly 1.6x faster, and pixel decode is the dominant cost.\n    hdr = [pydicom.dcmread(os.path.join(series_dir, f), stop_before_pixels=True)\n           for f in files]\n    side = detect_side(hdr)\n    pos = [slice_position(d) for d in hdr]\n    if all(p is not None for p in pos):\n        order = sorted(range(len(hdr)), key=lambda i: (pos[i],\n                       int(getattr(hdr[i], \"InstanceNumber\", 0))))\n    else:\n        order = sorted(range(len(hdr)),\n                       key=lambda i: int(getattr(hdr[i], \"InstanceNumber\", i)))\n\n    # Select slices BEFORE decoding, so normalization sees the same window the\n    # training pipeline used. Order matters: subsample, then normalize.\n    if len(order) > n_slices:\n        sel = np.linspace(0, len(order) - 1, n_slices).round().astype(int)\n        order = [order[i] for i in sel]\n\n    ds = [pydicom.dcmread(os.path.join(series_dir, files[i])) for i in order]\n    order = range(len(ds))\n\n    spacing = getattr(ds[0], \"PixelSpacing\", None)\n    arrs = [ds[i].pixel_array for i in order]\n    shapes = {a.shape for a in arrs}\n    if len(shapes) > 1:\n        tgt = max(shapes, key=lambda s: sum(a.shape == s for a in arrs))\n        arrs = [a if a.shape == tgt\n                else cv2.resize(a.astype(np.float32), (tgt[1], tgt[0])) for a in arrs]\n    vol = np.stack(arrs).astype(np.float32)\n\n    # The medial-lateral axis is a property of the ACQUISITION, so it is read off the\n    # first ordered-and-selected dataset — the same dataset src/data/dicom.py reads it\n    # from, so the two paths make the same decision on the same study.\n    axis = lr_axis(ds[0])\n\n    return {c: _finish(vol.copy(), spacing, side, c, size, n_slices, axis)\n            for c in crops}\n\n\ndef _finish(vol, spacing, side, CROP_MM, size, n_slices, lr=\"columns\"):\n    \"\"\"Crop -> mirror -> normalize -> sample -> resize, for ONE crop value.\n    Order matters and matches src/data/dicom.py: crop and mirror precede normalize,\n    and normalize precedes slice selection.\"\"\"\n    import cv2\n\n    # Physical-scale crop, before normalization — matches src/data/dicom.py.\n    if CROP_MM and spacing is not None:\n        ph, pw = float(spacing[0]), float(spacing[1])\n        if ph > 0 and pw > 0:\n            ch, cw = int(round(CROP_MM / ph)), int(round(CROP_MM / pw))\n            h, w = vol.shape[1], vol.shape[2]\n            if not (ch >= h and cw >= w):\n                ch, cw = min(ch, h), min(cw, w)\n                if CROP_CENTER == \"anatomy\":\n                    # byte-for-byte the same rule as src/data/dicom.py:tissue_centroid\n                    mid = vol[len(vol) // 2].astype(np.float32)\n                    lo_, hi_ = float(mid.min()), float(mid.max())\n                    if hi_ <= lo_:\n                        cy, cx = h / 2.0, w / 2.0\n                    else:\n                        m_ = mid > (lo_ + 0.12 * (hi_ - lo_))\n                        if m_.sum() < 0.01 * m_.size:\n                            cy, cx = h / 2.0, w / 2.0\n                        else:\n                            ys_, xs_ = np.nonzero(m_)\n                            cy, cx = float(ys_.mean()), float(xs_.mean())\n                    top = int(round(min(max(cy - ch / 2.0, 0), h - ch)))\n                    left = int(round(min(max(cx - cw / 2.0, 0), w - cw)))\n                else:\n                    top, left = (h - ch) // 2, (w - cw) // 2\n                vol = vol[:, top:top + ch, left:left + cw]\n\n    # Canonicalise laterality BEFORE normalization, matching src/data/dicom.py — and now\n    # ON THE SAME AXIS as it. `vol` is (n_slices, H, W); which axis runs medial-to-lateral\n    # depends on the plane, so the mirror cannot be a fixed `[:, :, ::-1]`:\n    #\n    #   plane      slice axis            columns axis        mirror on\n    #   Axial      superior-inferior     medial-lateral      columns\n    #   Coronal    anterior-posterior    medial-lateral      columns\n    #   Sagittal   MEDIAL-LATERAL        anterior-posterior  SLICES\n    #\n    # `lr == None` is an oblique acquisition or missing geometry: there is no left-right\n    # axis to reflect through, so the series passes unmirrored rather than scrambled.\n    if CANONICAL_SIDE is not None and side is not None and side != CANONICAL_SIDE:\n        if lr == \"columns\":\n            vol = vol[:, :, ::-1].copy()\n        elif lr == \"slices\":\n            vol = vol[::-1].copy()\n\n    # 12-bit data in a uint16 container: percentile-normalize, never /65535.\n    lo, hi = np.percentile(vol, 0.5), np.percentile(vol, 99.5)\n    vol = np.zeros_like(vol) if hi <= lo else np.clip((vol - lo) / (hi - lo), 0, 1)\n\n    if vol.shape[0] != n_slices:\n        idx = np.linspace(0, vol.shape[0] - 1, n_slices).round().astype(int)\n        vol = vol[idx]\n    if vol.shape[1] != size:\n        vol = np.stack([cv2.resize(s, (size, size), interpolation=cv2.INTER_AREA)\n                        for s in vol])\n    # .round() before cast, matching src/data/preprocess.py. A bare astype()\n    # truncates and drifts one grey level from the training pipeline.\n    return (vol * 255).round().astype(np.uint8)\n\n\ndef build_study(uid, meta):\n    \"\"\"Assemble one study into (5, N_SLICES, SIZE, SIZE) plus a presence mask.\"\"\"\n    d = os.path.join(SERIES_DIR, uid)\n    vols = {c: np.zeros((len(SLOTS), N_SLICES, SIZE, SIZE), np.uint8) for c in CROPS}\n    mask = np.zeros(len(SLOTS), bool)\n    rows = meta[meta.StudyInstanceUID == uid]\n    for i, slot in enumerate(SLOTS):\n        plane, fs = slot.rsplit(\"_\", 1)\n        sel = rows[(rows.Anatomical_Plane == plane)\n                   & (rows.Fluid_Sensitive == (1 if fs == \"fs\" else 0))]\n        best, best_n = None, -1\n        for suid in sel.SeriesInstanceUID:\n            p = os.path.join(d, suid)\n            if os.path.isdir(p):\n                n = len([f for f in os.listdir(p) if f.endswith(\".dcm\")])\n                if n > best_n:\n                    best, best_n = p, n\n        if best is None:\n            continue\n        try:\n            v = load_series(best, crops=CROPS)     # one decode, every crop\n            if v is not None:\n                for c in CROPS:\n                    vols[c][i] = v[c]\n                mask[i] = True\n        except Exception as exc:  # noqa: BLE001 — never let one study kill the run\n            print(f\"  [warn] {uid[-8:]} {slot}: {type(exc).__name__}: {exc}\")\n    if DUMP_VOL:\n        # Off unless asked. This file reimplements the preprocessing in\n        # src/data/{dicom,preprocess}.py rather than importing it, because the\n        # submission has to be a single self-contained notebook cell — and Sec 5.6\n        # records that divergence as a trap already sprung once here. Dumping the\n        # assembled volume is what lets scripts/diff_preprocess_paths.py compare the\n        # two implementations on pixels instead of on predictions.\n        np.savez_compressed(os.path.join(DUMP_VOL, f\"{uid}.npz\"),\n                            mask=mask, **{f\"crop{c:g}\": vols[c] for c in CROPS})\n    return vols, mask\n\n\n# ---------------------------------------------------------------- run\ntest = pd.read_csv(f\"{DATA}/{SPLIT}.csv\")\nmeta = pd.read_csv(f\"{DATA}/{SPLIT}_series.csv\")\n\n# The DICOM tree is expected at <DATA>/<split>_series/, but don't assume: find the\n# directory whose children are the study UIDs.\nSERIES_DIR = os.path.join(DATA, f\"{SPLIT}_series\")\nif not os.path.isdir(SERIES_DIR):\n    wanted = set(test.StudyInstanceUID.astype(str))\n    for name in sorted(os.listdir(DATA)):\n        cand = os.path.join(DATA, name)\n        if os.path.isdir(cand) and wanted.intersection(os.listdir(cand)):\n            SERIES_DIR = cand\n            break\n    print(f\"[data] series directory: {SERIES_DIR}\")\n\n# Prefer sample_submission.csv as the row template — it IS the required format, so\n# matching it exactly removes any doubt about column names and row order.\nSAMPLE = os.path.join(DATA, \"sample_submission.csv\")\ntemplate = pd.read_csv(SAMPLE) if os.path.exists(SAMPLE) else None\nif template is not None:\n    print(f\"[data] using sample_submission.csv as the template ({len(template)} rows)\")\n# Locally we point at the train split and only have a few studies on disk.\nuids = [u for u in test.StudyInstanceUID\n        if os.path.isdir(os.path.join(SERIES_DIR, u))] or test.StudyInstanceUID.tolist()\nprint(f\"test set: {len(uids)} studies, {len(meta)} series\")\n\n\ndef write_submission(predictions=None):\n    \"\"\"Write submission.csv. Called immediately, then again with real predictions.\n\n    Writing a valid fallback BEFORE any heavy work is deliberate: if preprocessing\n    raises, or the kernel is killed on time or memory, a scoreable file still exists.\n    Kaggle rejects a version that produces no submission.csv at all, so an empty\n    output is strictly worse than a mediocre one.\n    \"\"\"\n    base = template if template is not None else test\n    df = pd.DataFrame({\"StudyInstanceUID\": base.StudyInstanceUID})\n    for j, c in enumerate(LABEL_COLUMNS):\n        df[c] = 0.5 if predictions is None else predictions[:, j]\n    df.to_csv(OUT, index=False)\n    return df\n\n\nos.makedirs(os.path.dirname(OUT) or \".\", exist_ok=True)\n_fallback = write_submission()\nprint(f\"[safety] wrote fallback {OUT} before starting work \"\n      f\"({len(_fallback)} rows, all 0.5)\")\n\nt_boot = time.time()\nstats = {\"ok\": 0, \"failed\": 0, \"slots\": 0}\nt_pre = t_inf = 0.0\n\nt_boot = time.time()\nstats = {\"ok\": 0, \"failed\": 0, \"slots\": 0}\nt_pre = t_inf = 0.0\n\n# ---------------------------------------------------------------- model\nimport torch  # noqa: E402\nimport torch.nn as nn  # noqa: E402\nfrom transformers import AutoModel  # noqa: E402\n\nDEV = \"cuda\" if torch.cuda.is_available() else \"cpu\"\nprint(f\"[model] device {DEV}\")\n\n\ndef _find(rel, max_depth=9, first_only=True):\n    \"\"\"Per-root depth cap: shallow inside an input holding a DICOM tree, deep\n    elsewhere. See kaggle_train_v3.py for why both halves are needed.\"\"\"\n    root = os.environ.get(\"KAGGLE_INPUT_ROOT\", \"/kaggle/input\")\n    found = []\n    for top in sorted(glob.glob(f\"{root}/*\")):\n        if not os.path.isdir(top):\n            continue\n        dicom = bool(glob.glob(os.path.join(top, \"*_series\"))\n                     or glob.glob(os.path.join(top, \"*\", \"*_series\")))\n        for d in range(3 if dicom else max_depth):\n            h = sorted(glob.glob(f\"{top}/\" + \"*/\" * d + rel))\n            if h:\n                if first_only:\n                    return h\n                found += h\n                break\n    return found\n\n\n_ckpt = _find(\"model_v3.pt\", first_only=False)\n_dino = [os.path.dirname(h) for h in _find(\"config.json\", first_only=False)\n         if \"dinov2\" in h.lower()]\nMODEL_OK = bool(_ckpt and _dino)\nMODELS = []          # (crop_mm, nn.Module) — filled after the class is defined\n_NS_SEEN = []        # n_slices each checkpoint was trained at; must be unanimous\nCROPS = [CROP_MM]    # every distinct crop the loaded models need\nif not MODEL_OK:\n    print(f\"[!] model inputs missing — model_v3.pt {'ok' if _ckpt else 'MISSING'}, \"\n          f\"dinov2 {'ok' if _dino else 'MISSING'}. Falling back to constants.\")\nelse:\n    ck = torch.load(_ckpt[0], map_location=\"cpu\", weights_only=False)\n    print(f\"[model] {ck.get('arch')} trained on {ck.get('pipeline_version')}, \"\n          f\"val macro AUC {ck.get('val_macro_auc'):.4f}\")\n    _pv = ck.get(\"pipeline_version\", \"\")\n    CROP_CENTER = \"anatomy\" if re.search(r\"crop\\d+(?:\\.\\d+)?c\\b\", _pv) else \"image\"\n    print(f\"[model] crop centring: {CROP_CENTER}  (from pipeline_version {_pv!r})\")\n    _m = re.search(r\"crop(\\d+(?:\\.\\d+)?)\", _pv)\n    if _m:\n        _ck_crop = float(_m.group(1))\n        if _ck_crop != CROP_MM:\n            print(f\"[model] crop {CROP_MM:.0f} mm -> {_ck_crop:.0f} mm, taken from the \"\n                  f\"checkpoint. Training and inference now agree.\")\n            CROP_MM = _ck_crop\n    else:\n        print(f\"[!] checkpoint pipeline_version {_pv!r} carries no crop; assuming \"\n              f\"{CROP_MM:.0f} mm. If training used another value the features are wrong.\")\n    GROUP = ck.get(\"group\", GROUP)\n    SLICE_BAND = tuple(ck.get(\"band\", SLICE_BAND))\n    HIDDEN = ck.get(\"hidden\", 256)\n    UNFREEZE_LAST = ck.get(\"unfreeze_last\", 6)\n\n    class SlotAttentionHead(nn.Module):\n        \"\"\"Byte-identical to the training definition; see kaggle_train_v3.py.\"\"\"\n\n        def __init__(self, dim, n_slot, n_out, hidden, p=0.2):\n            super().__init__()\n            self.proj = nn.Sequential(nn.LayerNorm(dim), nn.Linear(dim, hidden), nn.GELU())\n            self.slot_emb = nn.Parameter(torch.randn(n_slot, hidden) * 0.02)\n            self.query = nn.Parameter(torch.randn(n_out, hidden) * 0.02)\n            self.drop = nn.Dropout(p)\n            self.w = nn.Parameter(torch.randn(n_out, hidden) * 0.02)\n            self.b = nn.Parameter(torch.zeros(n_out))\n            self.register_buffer(\"prior\", torch.zeros(n_out, n_slot))\n\n        def forward(self, feats, mask):\n            h = self.proj(feats) + self.slot_emb\n            att = torch.einsum(\"oh,bsh->bos\", self.query, h) / (self.slot_emb.shape[1] ** 0.5)\n            att = att + self.prior.unsqueeze(0)\n            att = att.masked_fill(~mask.unsqueeze(1), float(\"-inf\")).softmax(-1)\n            pooled = self.drop(torch.einsum(\"bos,bsh->boh\", att, h))\n            return (pooled * self.w).sum(-1) + self.b, att\n\n    def _dino_path(width):\n        \"\"\"Pick the DINOv2 of the width this checkpoint was trained with.\n\n        Every model used to be built from _dino[0]. That was fine while every checkpoint\n        was small, and silently wrong the moment a base checkpoint joined: base weights\n        into a small backbone is a shape mismatch, and the failure would surface deep in\n        load_state_dict rather than here where the cause is obvious.\n        \"\"\"\n        hits = [d for d in _dino if f\"/{width}/\" in d.replace(\"\\\\\", \"/\").lower()]\n        if not hits:\n            raise SystemExit(\n                f\"[X] a checkpoint needs DINOv2-{width} but only {_dino} are attached. \"\n                f\"Add Input -> Models -> metaresearch/dinov2/PyTorch/{width}.\")\n        return hits[0]\n\n    class Model(nn.Module):\n        def __init__(self, width=\"small\"):\n            super().__init__()\n            self.bb = AutoModel.from_pretrained(_dino_path(width))\n            dim = self.bb.config.hidden_size * 2\n            self.head = SlotAttentionHead(dim, len(SLOTS), len(LABEL_COLUMNS), HIDDEN)\n\n        def forward(self, x, mask):\n            h = self.bb(pixel_values=x).last_hidden_state\n            f = torch.cat([h[:, 0], h[:, 1:].mean(1)], -1)\n            f = f.reshape(mask.shape[0], mask.shape[1], -1)\n            return self.head(f, mask)\n\n    # Build one model per checkpoint. Each carries the crop it was TRAINED at, so a\n    # crop130 model is never fed crop95 pixels — that would not raise, it would just\n    # score badly and look like the ensemble failing.\n    for _p in _ckpt:\n        _c = torch.load(_p, map_location=\"cpu\", weights_only=False)\n        _cpv = _c.get(\"pipeline_version\", \"\")\n        _cm = re.search(r\"crop(\\d+(?:\\.\\d+)?)\", _cpv)\n        _crop = float(_cm.group(1)) if _cm else CROP_MM\n        _w = _c.get(\"backbone\", \"small\")        # v12/base16 record this; older ones do not\n        _m2 = Model(_w)\n        _m2.load_state_dict(_c[\"state_dict\"])   # strict: a width mismatch must raise here\n        _m2.eval().to(DEV)\n        MODELS.append((_crop, _m2))\n        _NS_SEEN.append(int(_c.get(\"n_slices\", 16)))\n        print(f\"[model] + {os.path.basename(os.path.dirname(os.path.dirname(_p)))[:28]:<30}\"\n              f\"{_w:<5} crop {_crop:.0f} mm, val {_c.get('val_macro_auc', float('nan')):.4f}\")\n\n    # SLICE-COUNT GUARD. Sec 8.141c measured +0.0060 (10/10 seeds) from moving the cache\n    # from 16 to 32 slices, and the mechanism is ADJACENCY: from a 30-slice series a\n    # 16-slice cache makes the 3 channels every-other-slice, i.e. spatially aliased.\n    # A model trained on 32 and served on 16 sees the aliased triplet it was never trained\n    # on, which would silently give back the entire gain and look like the finding not\n    # replicating. Unlike crop, N_SLICES is NOT per-model: preprocessing builds ONE volume\n    # per crop shared by every member, so the members must agree.\n    if len(set(_NS_SEEN)) > 1:\n        print(f\"[X] checkpoints disagree on n_slices: {sorted(set(_NS_SEEN))}. \"\n              \"Preprocessing produces one volume per crop for all members, so this cannot \"\n              \"be satisfied without preprocessing twice. Refusing to write a submission \"\n              \"that would feed some members the wrong slice spacing.\")\n        raise SystemExit(1)\n    if _NS_SEEN and _NS_SEEN[0] != N_SLICES:\n        print(f\"[model] n_slices {N_SLICES} -> {_NS_SEEN[0]}, taken from the checkpoint. \"\n              \"Training and inference now agree.\")\n        N_SLICES = _NS_SEEN[0]\n    elif not _NS_SEEN:\n        print(f\"[!] no checkpoint records n_slices; assuming {N_SLICES}. A model trained \"\n              \"at a different slice count will be served aliased triplets (Sec 8.141c).\")\n\n    CROPS = sorted({c for c, _ in MODELS})\n    print(f\"[model] ensembling {len(MODELS)} model(s) over crops {CROPS} mm\")\n    # COUNT GUARD. Sec 8.81: the dual notebook silently assembled 2 of 3 members because its\n    # loader searched for one filename, and ONLY its count guard caught it — the submission\n    if len(MODELS) != N_EXPECTED:\n        print(f\"[X] expected {N_EXPECTED} models, found {len(MODELS)}. Attach Notebooks -> \"\n              \"all members listed in this kernel's inputs. Refusing to write a submission \"\n              \"that would look valid and score wrong.\")\n        raise SystemExit(1)\n    if len({c for c, _ in MODELS}) == 1 and len(MODELS) > 1:\n        print(\"[!] all models share one crop — the holdout says crop diversity is worth \"\n              \"2x seed diversity (+0.0046 vs +0.0023). Consider mixing crops.\")\n    IMEAN = torch.tensor([0.485, 0.456, 0.406]).view(1, 3, 1, 1).to(DEV)\n    ISTD = torch.tensor([0.229, 0.224, 0.225]).view(1, 3, 1, 1).to(DEV)\n    BAND = (int(SLICE_BAND[0] * (N_SLICES - 1)), int(SLICE_BAND[1] * (N_SLICES - 1)))\n    STARTS = list(range(BAND[0], max(BAND[0], BAND[1] - GROUP + 1) + 1)) if TTA \\\n        else [(BAND[0] + BAND[1] - GROUP + 1) // 2]\n    print(f\"[model] group {GROUP}, band {BAND}, {len(STARTS)} TTA window(s)\")\n\n# Focal findings are \"present on SOME window\" — a max. Diffuse ones are an average.\n# Same principle as the mean+max slice pooling, applied across TTA windows; the\n# public 0.899 fork gained +0.008 doing exactly this.\nTTA_POOL = {\"Fracture\": \"max\", \"Contusion\": \"max\", \"Medial Meniscus\": \"max\",\n            \"Lateral Meniscus\": \"max\", \"Baker's\": \"max\", \"ACL\": \"top3\", \"MCL\": \"top3\"}\n\n\n@torch.no_grad()\ndef _predict_one(v, m, model):\n    \"\"\"One model, TTA-pooled. (B,5,16,H,W) uint8 -> (B,12).\"\"\"\n    wins = []\n    for st in STARTS:\n        w = v[:, :, st:st + GROUP]\n        x = torch.from_numpy(np.ascontiguousarray(w)).to(DEV).float().div_(255.0)\n        b, s = x.shape[0], x.shape[1]\n        x = ((x.reshape(b * s, GROUP, *x.shape[-2:])) - IMEAN) / ISTD\n        logit, _ = model(x, m)\n        wins.append(torch.sigmoid(logit.float()))\n    P = torch.stack(wins)                              # (W, B, 12)\n    out = P.mean(0)\n    for lab, mode in TTA_POOL.items():\n        j = LABEL_COLUMNS.index(lab)\n        if mode == \"max\":\n            out[:, j] = P[:, :, j].max(0).values\n        elif mode == \"top3\":\n            k = min(3, P.shape[0])\n            out[:, j] = P[:, :, j].topk(k, dim=0).values.mean(0)\n    return out\n\n\n@torch.no_grad()\ndef predict(vols, masks):\n    \"\"\"Average every model, each reading the crop it was trained at.\n\n    `vols` is a list of {crop: volume} dicts — one decode per study, reused across\n    models. A plain mean: the holdout showed no benefit from weighting, and weights\n    fitted on 870 studies would be one more thing to overfit.\n    \"\"\"\n    m = torch.from_numpy(np.stack(masks)).to(DEV)\n    per_crop = {c: np.stack([d[c] for d in vols]) for c in CROPS}\n    outs = [_predict_one(per_crop[c], m, mdl) for c, mdl in MODELS]\n    return (sum(outs) / len(outs)).cpu().numpy()\n\n\nSTARTUP = time.time() - t_boot\nprint(f\"[time] startup (imports + model load + CUDA init): {STARTUP:.0f}s\")\n\nt0 = time.time()\npreds = {}\ntry:\n    with ThreadPoolExecutor(max_workers=WORKERS) as pool:\n        buf_uid, buf_vol, buf_mask = [], [], []\n        for i, uid in enumerate(uids, 1):\n            _t = time.time()\n            try:\n                vol, mask = build_study(uid, meta)\n                stats[\"ok\"] += 1\n                stats[\"slots\"] += int(mask.sum())\n            except Exception as exc:  # noqa: BLE001\n                stats[\"failed\"] += 1\n                print(f\"  [FAIL] {uid[-8:]}: {exc}\")\n                vol = {c: np.zeros((len(SLOTS), N_SLICES, SIZE, SIZE), np.uint8)\n                       for c in CROPS}\n                mask = np.zeros(len(SLOTS), bool)\n            t_pre += time.time() - _t\n            buf_uid.append(uid); buf_vol.append(vol); buf_mask.append(mask)\n\n            if MODEL_OK and (len(buf_uid) == CHUNK or i == len(uids)):\n                _t = time.time()\n                p = predict(buf_vol, buf_mask)\n                t_inf += time.time() - _t\n                for u, row in zip(buf_uid, p):\n                    preds[u] = row\n                buf_uid, buf_vol, buf_mask = [], [], []\n            elif not MODEL_OK and len(buf_uid) >= CHUNK:\n                buf_uid, buf_vol, buf_mask = [], [], []\n\n            if i % 50 == 0 or i == len(uids):\n                el = time.time() - t0\n                print(f\"  {i}/{len(uids)}  {el:.0f}s  ({el / i:.2f}s/study)\", flush=True)\nexcept Exception as exc:  # noqa: BLE001\n    print(f\"\\n[!] inference aborted: {type(exc).__name__}: {exc}\")\n    print(\"    the fallback submission.csv written earlier still stands.\")\n\nelapsed = time.time() - t0\n\n# ---------------------------------------------------------------- submission\nbase = template if template is not None else test\nif preds:\n    P = np.full((len(base), len(LABEL_COLUMNS)), 0.5, dtype=float)\n    for r, u in enumerate(base.StudyInstanceUID):\n        if u in preds:\n            P[r] = preds[u]\n    missing = len(base) - sum(u in preds for u in base.StudyInstanceUID)\n    if missing:\n        print(f\"[!] {missing} studies had no prediction; left at 0.5\")\n    sub = write_submission(P)\n    print(f\"[ok] wrote model predictions for {len(preds)} studies\")\nelse:\n    sub = write_submission()\n    print(\"[!] no predictions — submission is the constant fallback\")\n\nprint(\"\\n\" + \"=\" * 62)\nprint(\"RESULT\")\nprint(\"=\" * 62)\nprint(f\"  studies processed : {stats['ok']}/{len(uids)}  ({stats['failed']} failed)\")\nif stats[\"ok\"]:\n    print(f\"  mean slots present: {stats['slots'] / stats['ok']:.2f} / {len(SLOTS)}\")\nprint(f\"  wrote {OUT}: {sub.shape}\")\nassert list(sub.columns) == [\"StudyInstanceUID\", *LABEL_COLUMNS], \"column mismatch!\"\nexpected = len(template) if template is not None else len(test)\nassert len(sub) == expected, \"submission row count does not match the template!\"\nassert sub[LABEL_COLUMNS].notna().all().all(), \"NaNs in submission!\"\nassert sub.StudyInstanceUID.is_unique, \"duplicate StudyInstanceUID!\"\nassert ((sub[LABEL_COLUMNS] >= 0) & (sub[LABEL_COLUMNS] <= 1)).all().all(), \\\n    \"predictions outside [0, 1]!\"\n\n# On Kaggle every row of the split's CSV must be predicted. Locally we deliberately\n# run on the handful of studies present on disk, so only enforce this for real.\nLOCAL = not DATA.startswith(\"/kaggle/input\")\nif LOCAL:\n    print(f\"  [local mode] preprocessed {len(uids)} of {len(test)} studies; \"\n          \"submission still covers every row\")\nprint(\"  submission columns / NaN / range / uniqueness checks: PASSED\")\n\n# The fallback keeps a failed run scoreable, but a 0.5 submission looks identical to\n# a working one from the outside. Say so LOUDLY, as the last thing in the log.\nif not preds:\n    print(\"\\n\" + \"!\" * 62)\n    print(\"!!  THIS SUBMISSION IS THE CONSTANT FALLBACK — IT WILL SCORE 0.500\")\n    print(\"!!  The model did not run. Check the [!] lines above before submitting.\")\n    print(\"!\" * 62)\nelse:\n    spread = float(np.std([v for v in preds.values()]))\n    print(f\"\\n[check] prediction spread (std across all values): {spread:.4f}\")\n    if spread < 1e-6:\n        print(\"!\" * 62)\n        print(\"!!  PREDICTIONS ARE CONSTANT — this will score 0.500 regardless of\")\n        print(\"!!  the model. Something is wrong upstream.\")\n        print(\"!\" * 62)\n\nprint(\"\\n\" + \"=\" * 62)\nprint(\"TIMING BUDGET  <- the number this whole run exists to produce\")\nprint(\"=\" * 62)\nn = max(len(uids), 1)\nper_pre, per_inf = t_pre / n, t_inf / n\nper_work = per_pre + per_inf\nprint(f\"  startup (one-off)  : {STARTUP:>8.0f}s\")\nprint(f\"  preprocessing      : {t_pre:>8.0f}s  ({per_pre:.2f}s/study)\")\nprint(f\"  model inference    : {t_inf:>8.0f}s  ({per_inf:.2f}s/study)\")\nprint(f\"  steady state       : {per_work:.2f}s/study  <- use THIS to extrapolate\")\nprint()\nbudget = RUNTIME_LIMIT_S - STARTUP\nprint(f\"  9h limit minus startup = {budget:.0f}s\")\nif per_work > 0:\n    cap = int(budget / per_work)\n    print(f\"  capacity at this rate: ~{cap:,} studies\")\n    if cap < 5000:\n        print(f\"  [!] the training set is 4,407 studies. If the private test set is\")\n        print(f\"      comparable, {cap:,} is NOT enough headroom — reduce work per study\")\n        print(f\"      (fewer slices, smaller input) or overlap CPU and GPU.\")\nprint()\nprint(\"  Extrapolating from a 3-study stub is dominated by startup; the steady-state\")\nprint(\"  line is the honest number, and even it is noisy at n=3.\")\n"},{"cell_type":"code","execution_count":null,"metadata":{},"outputs":[],"source":"# ---------------------------------------------------------------- verify\n# A submission that silently fell back to constants looks identical from the outside, and a\n# constant column is all ties, which macro AUC scores at 0.5. Check, loudly.\nimport pandas as _eff_pd\n\n_eff_sub = _eff_pd.read_csv(\"/kaggle/working/submission.csv\",\n                            dtype={\"StudyInstanceUID\": str})\n_EFF_TARGETS = [\"ACL\", \"MCL\", \"Medial Meniscus\", \"Lateral Meniscus\", \"Medial OA\",\n                \"Lateral OA\", \"PF OA\", \"Effusion\", \"Synovitis\", \"Baker's\",\n                \"Contusion\", \"Fracture\"]\nassert _eff_sub.columns.tolist() == [\"StudyInstanceUID\", *_EFF_TARGETS], \"schema drift\"\nassert _eff_sub[_EFF_TARGETS].notna().all().all(), \"nulls in submission\"\n_eff_sd = _eff_sub[_EFF_TARGETS].to_numpy(float).std(axis=0)\nprint(f\"[eff] {_eff_sub.shape}, per-column sd min {_eff_sd.min():.4f} \"\n      f\"max {_eff_sd.max():.4f}\", flush=True)\nif float(_eff_sd.min()) < 1e-9:\n    raise SystemExit(\"[X] a constant column — the 0.5 fallback fired; this would score 0.5 \"\n                     \"on that label and must not be submitted\")\nprint(\"[eff] submission verified\", flush=True)\n"}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python"}},"nbformat":4,"nbformat_minor":5}