{"cells":[{"cell_type":"markdown","metadata":{},"source":"# RSNA Knee Abnormality Detection — Metadata Baseline (fast submission)\n\nSeries composition alone scores ~0.5954 macro AUC (public EDA, measured).\nThis notebook fits a ridge on manifest features (plane counts, fluid\nfractions, slot presence) against report-derived soft targets and writes\n`submission.csv`. No DICOM, no GPU, seconds of runtime.\n\nThis is the E0 floor: the DINOv3 image pipeline must beat it on the official\nEfficiency Score, not just on AUC.\n"},{"cell_type":"code","metadata":{},"source":"# ===== Cell 1: config =====\nimport os\nfor _v in (\"OMP_NUM_THREADS\", \"OPENBLAS_NUM_THREADS\", \"MKL_NUM_THREADS\",\n           \"NUMEXPR_NUM_THREADS\", \"VECLIB_MAXIMUM_THREADS\"):\n    os.environ.setdefault(_v, \"4\")\n\nSEED = 20260819\nimport numpy as np\nimport pandas as pd\nimport time\nfrom pathlib import Path\n\nALPHA = 1.0\nGOLD_WEIGHT = 8.0\n","outputs":[],"execution_count":null},{"cell_type":"code","metadata":{},"source":"# ===== Cell 2: environment =====\nimport os as _os\n\ndef detect_input_root():\n    # Local override for testing (KAGGLE_INPUT_ROOT=/path/to/data).\n    env = _os.environ.get(\"KAGGLE_INPUT_ROOT\")\n    if env and (Path(env) / \"train.csv\").is_file():\n        return Path(env)\n    candidates = [\n        Path(\"/kaggle/input/rsna-knee-abnormality-detection\"),\n        Path(\"/kaggle/input/competitions/rsna-knee-abnormality-detection\"),\n        Path(\".\"),\n    ]\n    for c in candidates:\n        try:\n            if (c / \"train.csv\").is_file():\n                return c\n        except OSError:\n            continue\n    raise FileNotFoundError(\"no Kaggle input root\")\n\nROOT = detect_input_root()\nIS_KAGGLE = str(ROOT).startswith(\"/kaggle/\")\nWORK = Path(\"/kaggle/working\") if IS_KAGGLE else Path(\".\")\nprint(\"root:\", ROOT, \"| kaggle:\", IS_KAGGLE)\n","outputs":[],"execution_count":null},{"cell_type":"code","metadata":{},"source":"# ===== Cell 3: embed tested metadata module =====\n","outputs":[],"execution_count":null},{"cell_type":"code","metadata":{},"source":"# --- embedded from knee_efficiency/utils/labels.py (locally TDD-tested) ---\n\"\"\"Label and constant definitions for the competition.\"\"\"\nfrom __future__ import annotations\n\nTARGET_LABELS: list[str] = [\n    \"ACL\",\n    \"MCL\",\n    \"Medial Meniscus\",\n    \"Lateral Meniscus\",\n    \"Medial OA\",\n    \"Lateral OA\",\n    \"PF OA\",\n    \"Effusion\",\n    \"Synovitis\",\n    \"Baker's\",\n    \"Contusion\",\n    \"Fracture\",\n]\n\n# Submission column header (spaces exactly as the competition expects).\nSUBMISSION_HEADER = [\"StudyInstanceUID\", *TARGET_LABELS]\n","outputs":[],"execution_count":null},{"cell_type":"code","metadata":{},"source":"# --- embedded from knee_efficiency/pipeline/metadata.py (locally TDD-tested) ---\n\"\"\"Metadata-only baseline: predict from series composition (no DICOM, no GPU).\n\nMotivation (measured by public EDA, 2026-08-08): series composition alone\nscores ~0.5954 macro AUC — which sequences a radiographer chose to acquire\nalready leaks a little about what they were looking for.\n\nThis is the fastest possible real submission while the DINOv3 image pipeline\nis being prepared. Features come only from train_series.csv / test_series.csv,\nwhich the hidden test set ships. Training targets are report-derived soft\nlabels (gold studies carry true labels, weighted higher).\n\"\"\"\nfrom __future__ import annotations\n\nimport numpy as np\nimport pandas as pd\n\nPLANES = [\"Sagittal\", \"Coronal\", \"Axial\"]\n\n\ndef normalize_manifest(df: pd.DataFrame) -> pd.DataFrame:\n    \"\"\"Coerce a raw series manifest into the typed columns the features need.\"\"\"\n    if df is None or df.empty:\n        return pd.DataFrame()\n    out = df.copy()\n    for c in [\"StudyInstanceUID\", \"SeriesInstanceUID\", \"Anatomical_Plane\",\n              \"Fluid_Sensitive\", \"Fat_Suppression\"]:\n        if c not in out.columns:\n            out[c] = np.nan\n    out[\"StudyInstanceUID\"] = out[\"StudyInstanceUID\"].astype(str)\n    out[\"SeriesInstanceUID\"] = out[\"SeriesInstanceUID\"].astype(str)\n    out[\"Anatomical_Plane\"] = out[\"Anatomical_Plane\"].astype(str).str.strip().str.title()\n    out[\"Fluid_Sensitive\"] = pd.to_numeric(out[\"Fluid_Sensitive\"], errors=\"coerce\").fillna(0).astype(int)\n    out[\"Fat_Suppression\"] = pd.to_numeric(out[\"Fat_Suppression\"], errors=\"coerce\").fillna(0).astype(int)\n    return out\n\n\ndef build_series_features(man: pd.DataFrame) -> tuple[np.ndarray, list[str]]:\n    \"\"\"Per-study features from the series manifest only.\n\n    Returns (float32 array [n_studies, n_features], feature names).\n    \"\"\"\n    if man.empty:\n        return np.zeros((0, 0), np.float32), []\n    by_study = {s: g for s, g in man.groupby(\"StudyInstanceUID\", sort=False)}\n    rows, names = [], []\n    n_total = man[\"StudyInstanceUID\"].nunique()\n    for sid, g in by_study.items():\n        total = len(g)\n        r = {\n            \"log_n_series\": float(np.log1p(total)),\n            \"fluid_frac\": float(g[\"Fluid_Sensitive\"].mean()) if total else 0.0,\n        }\n        for p in PLANES:\n            r[f\"n_{p}\"] = float((g[\"Anatomical_Plane\"] == p).sum())\n        for p in PLANES:\n            r[f\"fluid_{p}\"] = float(((g[\"Anatomical_Plane\"] == p) & (g[\"Fluid_Sensitive\"] == 1)).sum())\n        rows.append(r)\n    names = sorted(rows[0].keys())\n    out = np.array([[r[n] for n in names] for r in rows], np.float32)\n    return out, names\n\n\ndef _soft_targets(train: pd.DataFrame, labels) -> tuple[np.ndarray, np.ndarray]:\n    \"\"\"Soft targets: true labels for gold studies, else per-label prevalence.\n\n    Returns (Y float32 [n, 12], W float32 [n] sample weights).\n    \"\"\"\n    gold = train[labels].notna().all(axis=1)\n    Y = pd.DataFrame(np.nan, index=train[\"StudyInstanceUID\"].astype(str), columns=labels)\n    for lab in labels:\n        prev = train.loc[gold, lab].mean() if gold.any() else 0.5\n        Y[lab] = prev\n    Y.loc[train.loc[gold, \"StudyInstanceUID\"].astype(str), labels] = train.loc[gold, labels].values\n    W = np.where(train[\"StudyInstanceUID\"].astype(str).isin(\n        set(train.loc[gold, \"StudyInstanceUID\"].astype(str))), 8.0, 1.0)\n    return Y.values.astype(np.float32), W.astype(np.float32)\n\n\ndef fit_ridge(X_tr: np.ndarray, y_tr: np.ndarray, X_te: np.ndarray,\n              w_tr: np.ndarray | None = None, alpha: float = 1.0) -> np.ndarray:\n    \"\"\"Ridge fit on standardized features; returns test predictions.\n\n    Closed-form ridge in pure numpy (no sklearn dependency): the metadata\n    baseline must stay dependency-light so the notebook runs offline with the\n    Kaggle image's stdlib.\n    \"\"\"\n    if X_tr.shape[1] == 0:\n        return np.zeros((X_te.shape[0], y_tr.shape[1]), np.float32)\n    mu = X_tr.mean(axis=0, keepdims=True)\n    sd = X_tr.std(axis=0, keepdims=True)\n    sd[sd < 1e-8] = 1.0\n    Xa = (X_tr - mu) / sd\n    Xb = (X_te - mu) / sd\n    n, d = Xa.shape\n    Wd = np.sqrt(w_tr) if w_tr is not None else np.ones(n)\n    Xw = Xa * Wd[:, None]\n    yw = y_tr * Wd[:, None]\n    A = Xw.T @ Xw + alpha * np.eye(d)\n    B = Xw.T @ yw\n    coef = np.linalg.solve(A, B)\n    return (Xb @ coef).astype(np.float32)\n\n\ndef make_metadata_predictions(train_feats: np.ndarray, y: np.ndarray, test_feats: np.ndarray,\n                              w: np.ndarray | None = None) -> np.ndarray:\n    \"\"\"Ridge predictions on standardized manifest features -> [n_test, 12].\n\n    train_feats/test_feats come from build_series_features; y are the soft\n    targets; w optional sample weights (gold studies weighted higher).\n    \"\"\"\n    preds = fit_ridge(train_feats, y, test_feats, w)\n    return np.clip(preds, 0.0, 1.0)\n","outputs":[],"execution_count":null},{"cell_type":"code","metadata":{},"source":"# --- embedded from knee_efficiency/pipeline/submission.py (locally TDD-tested) ---\n\"\"\"Submission file writer (exact competition format).\"\"\"\nfrom __future__ import annotations\n\nimport csv\nfrom pathlib import Path\n\n\n\ndef write_submission(path: str | Path, study_probs: dict[str, list[float]]) -> None:\n    \"\"\"Write a submission.csv from {study_uid: [12 probabilities]}.\n\n    Values are clipped to [0, 1]; the row order follows the dict insertion order\n    (callers pass a deterministic order, e.g. from test.csv).\n    \"\"\"\n    n_labels = len(TARGET_LABELS)\n    for uid, probs in study_probs.items():\n        if len(probs) != n_labels:\n            raise ValueError(f\"study {uid}: expected {n_labels} probabilities, got {len(probs)}\")\n    path = Path(path)\n    path.parent.mkdir(parents=True, exist_ok=True)\n    with open(path, \"w\", newline=\"\", encoding=\"utf-8\") as f:\n        writer = csv.writer(f)\n        writer.writerow(SUBMISSION_HEADER)\n        for uid, probs in study_probs.items():\n            clipped = [max(0.0, min(1.0, float(p))) for p in probs]\n            writer.writerow([uid, *clipped])\n","outputs":[],"execution_count":null},{"cell_type":"code","metadata":{},"source":"# ===== Cell 4: load data =====\nt0 = time.time()\ntrain = pd.read_csv(ROOT / \"train.csv\")\ntrain_series = pd.read_csv(ROOT / \"train_series.csv\")\ntest = pd.read_csv(ROOT / \"test.csv\")\ntest_series = pd.read_csv(ROOT / \"test_series.csv\")\nprint(\"train:\", train.shape, \"test:\", test.shape,\n      \"| series:\", len(train_series), len(test_series))\nLABELS = TARGET_LABELS\n","outputs":[],"execution_count":null},{"cell_type":"code","metadata":{},"source":"# ===== Cell 5: manifest features =====\ntr_man = normalize_manifest(train_series)\nte_man = normalize_manifest(test_series)\nX_tr, names = build_series_features(tr_man)\nX_te, _ = build_series_features(te_man)\nprint(\"train features:\", X_tr.shape, \"| test features:\", X_te.shape)\nprint(\"feature names:\", names)\n","outputs":[],"execution_count":null},{"cell_type":"code","metadata":{},"source":"# ===== Cell 6: soft targets (gold weighted) =====\ngold = train[LABELS].notna().all(axis=1)\nprint(\"gold studies:\", int(gold.sum()))\nY = pd.DataFrame(np.nan, index=train[\"StudyInstanceUID\"].astype(str), columns=LABELS)\nfor lab in LABELS:\n    prev = train.loc[gold, lab].mean() if gold.any() else 0.5\n    Y[lab] = prev\nY.loc[train.loc[gold, \"StudyInstanceUID\"].astype(str), LABELS] = train.loc[gold, LABELS].values\nW = np.where(train[\"StudyInstanceUID\"].astype(str).isin(\n    set(train.loc[gold, \"StudyInstanceUID\"].astype(str))), GOLD_WEIGHT, 1.0)\ny_tr = Y.values.astype(np.float32)\nprint(\"targets:\", y_tr.shape, \"| positives at 0.5:\", int((y_tr >= 0.5).sum()))\n","outputs":[],"execution_count":null},{"cell_type":"code","metadata":{},"source":"# ===== Cell 7: fit + predict =====\npreds = make_metadata_predictions(X_tr, y_tr, X_te, w=W)\nprint(\"predictions:\", preds.shape, \"| mean:\", preds.mean().round(4))\n","outputs":[],"execution_count":null},{"cell_type":"code","metadata":{},"source":"# ===== Cell 8: write submission (test.csv row order) =====\nordered = {str(r.StudyInstanceUID): preds[i].tolist() for i, r in enumerate(test.itertuples())}\nsub_path = WORK / \"submission.csv\"\nwrite_submission(sub_path, ordered)\nprint(\"wrote\", sub_path, \"|\", len(ordered), \"rows |\", round(time.time() - t0, 1), \"s total\")\n","outputs":[],"execution_count":null},{"cell_type":"code","metadata":{},"source":"# ===== Cell 9: versions =====\nimport platform\nprint(\"python\", platform.python_version())\nprint(\"numpy\", np.__version__, \"| pandas\", pd.__version__)\n","outputs":[],"execution_count":null}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.12"}},"nbformat":4,"nbformat_minor":5}