{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\n# Use the kagglehub client library to attach Kaggle resources like competitions, datasets, and models to your session\n# Learn more about kagglehub: https://github.com/Kaggle/kagglehub/blob/main/README.md\n\nimport kagglehub\n# kagglehub.dataset_download('<owner>/<dataset-slug>')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from __future__ import annotations\n\nimport gc\nimport json\nimport logging\nimport os\nimport re\nimport sys\nimport time\nimport traceback\nimport unicodedata\nimport warnings\nfrom concurrent.futures import ThreadPoolExecutor, as_completed\nfrom dataclasses import dataclass, field\nfrom pathlib import Path\nfrom typing import Optional\n\nimport numpy as np\nimport pandas as pd\n\nwarnings.filterwarnings(\"ignore\")\n\n# Cap BLAS/OpenMP thread pools before any numeric library spins up its own\n# pool; on Kaggle's shared CPUs, unbounded thread pools are a common cause\n# of silent slowdowns and occasional OOM under contention.\nfor _var in (\"OMP_NUM_THREADS\", \"OPENBLAS_NUM_THREADS\", \"MKL_NUM_THREADS\", \"NUMEXPR_NUM_THREADS\"):\n    os.environ.setdefault(_var, \"4\")\n\nlogging.basicConfig(\n    level=logging.INFO,\n    format=\"[%(asctime)s] %(message)s\",\n    datefmt=\"%H:%M:%S\",\n    stream=sys.stdout,\n)\nlog = logging.getLogger(\"rsna_knee\")\n\nT_START = time.time()\n\n\ndef elapsed() -> float:\n    return time.time() - T_START\n\n\n# ---------------------------------------------------------------------------\n# Configuration\n# ---------------------------------------------------------------------------\n\nTARGETS = [\n    \"ACL\", \"MCL\", \"Medial Meniscus\", \"Lateral Meniscus\", \"Medial OA\",\n    \"Lateral OA\", \"PF OA\", \"Effusion\", \"Synovitis\", \"Baker's\",\n    \"Contusion\", \"Fracture\",\n]\n\n\n@dataclass\nclass Config:\n    # --- paths -------------------------------------------------------\n    data_root: Optional[str] = None          # auto-detected if None\n    output_dir: str = \"/kaggle/working\"       # falls back to cwd if absent\n    submission_name: str = \"submission.csv\"\n\n    # --- runtime / memory budget --------------------------------------\n    time_budget_sec: float = 8.0 * 3600       # leave headroom under the 9h cap\n    max_series_per_study: int = 6             # bound per-study I/O\n    max_slices_per_series: int = 3            # central slices only\n    image_size: int = 128                     # embedding input resolution\n    embed_batch_size: int = 64\n    header_threads: int = 8\n    pixel_threads: int = 8\n\n    # --- modelling -----------------------------------------------------\n    n_folds: int = 5\n    random_seed: int = 2026\n    gold_sample_weight: float = 3.0           # weight of true annotations vs. derived labels\n    min_positive_for_auc: int = 2             # below this, a fold AUC is undefined -> skipped\n\n    # --- filled in at runtime -------------------------------------------\n    have_torch: bool = field(default=False, init=False)\n    have_lightgbm: bool = field(default=False, init=False)\n    device: str = field(default=\"cpu\", init=False)\n\n\nCFG = Config()\n\n\ndef resolve_output_dir() -> Path:\n    for cand in (CFG.output_dir, \"/kaggle/working\", \".\"):\n        p = Path(cand)\n        try:\n            p.mkdir(parents=True, exist_ok=True)\n            return p\n        except OSError:\n            continue\n    return Path(\".\")\n\n\n# ---------------------------------------------------------------------------\n# \n# ---------------------------------------------------------------------------\n\ndef write_benchmark_submission(root: Path, out_dir: Path) -> None:\n    try:\n        test_df = safe_read_csv(root / \"test.csv\")\n        if test_df is None or \"StudyInstanceUID\" not in test_df.columns:\n            test_df = pd.DataFrame({\"StudyInstanceUID\": []})\n    except Exception:\n        test_df = pd.DataFrame({\"StudyInstanceUID\": []})\n    sub = test_df[[\"StudyInstanceUID\"]].copy()\n    for t in TARGETS:\n        sub[t] = 0.5\n    sub.to_csv(out_dir / CFG.submission_name, index=False)\n    log.info(f\"benchmark submission written ({len(sub)} rows)\")\n\n\n# ---------------------------------------------------------------------------\n# I/O helpers\n# ---------------------------------------------------------------------------\n\ndef safe_read_csv(path: Path) -> Optional[pd.DataFrame]:\n    \"\"\"Read a CSV defensively; return None (not raise) on any failure.\"\"\"\n    try:\n        if not path.is_file():\n            log.warning(f\"missing file: {path}\")\n            return None\n        df = pd.read_csv(path)\n        return df\n    except Exception as exc:\n        log.warning(f\"failed to read {path}: {exc}\")\n        return None\n\n\ndef find_data_root() -> Path:\n    \"\"\"Locate the competition mount, trying the documented path first.\"\"\"\n    candidates = [\n        Path(\"/kaggle/input/competitions/rsna-knee-abnormality-detection\"),\n        Path(\"/kaggle/input/rsna-knee-abnormality-detection\"),\n        Path(\"data\"), Path(\".\"),\n    ]\n    if CFG.data_root:\n        candidates.insert(0, Path(CFG.data_root))\n    for c in candidates:\n        if (c / \"test.csv\").is_file():\n            return c\n    # Last resort: scan two levels under /kaggle/input, since some mounts\n    # nest one directory deeper than the documented path.\n    base = Path(\"/kaggle/input\")\n    if base.is_dir():\n        for depth1 in sorted(p for p in base.iterdir() if p.is_dir()):\n            for cand in [depth1] + sorted(p for p in depth1.iterdir() if p.is_dir()):\n                if (cand / \"test.csv\").is_file():\n                    return cand\n    raise FileNotFoundError(\n        \"Could not locate the competition data (test.csv not found under \"\n        \"/kaggle/input). Check that the dataset is attached to the notebook.\"\n    )\n\n\n# ---------------------------------------------------------------------------\n# Report-derived labels\n# ---------------------------------------------------------------------------\n\n_PRE_TRANSLATE = str.maketrans({\n    \"ı\": \"i\", \"İ\": \"i\", \"ß\": \"ss\", \"đ\": \"d\", \"Đ\": \"d\", \"ø\": \"o\", \"æ\": \"ae\",\n})\n\n\ndef normalize_text(text: str) -> str:\n    if not isinstance(text, str) or not text:\n        return \"\"\n    text = text.translate(_PRE_TRANSLATE).lower()\n    text = unicodedata.normalize(\"NFKD\", text)\n    text = \"\".join(ch for ch in text if not unicodedata.combining(ch))\n    text = re.sub(r\"[_\\-/\\\\]+\", \" \", text)\n    text = re.sub(r\"\\s+\", \" \", text).strip()\n    return text\n\n\ndef _rx(*patterns: str) -> re.Pattern:\n    return re.compile(\"|\".join(f\"(?:{p})\" for p in patterns))\n\n\nNEGATION = _rx(\n    r\"\\bno\\b\", r\"\\bnot\\b\", r\"\\bwithout\\b\", r\"\\bnegative for\\b\", r\"\\babsence\\b\",\n    r\"\\bno evidence\\b\", r\"\\bunremarkable\\b\", r\"\\bfree of\\b\", r\"\\bnone\\b\", r\"\\bnil\\b\",\n    r\"not (identified|seen|visuali[sz]ed|demonstrated|detected|present)\",\n    r\"\\bsin\\b\", r\"\\bausencia\\b\", r\"\\bningun[ao]?\\b\",\n    r\"\\bpas de\\b\", r\"\\bsans\\b\", r\"\\baucune?\\b\",\n)\nNORMALITY = _rx(\n    r\"\\bnormal\", r\"\\bintact\\b\", r\"\\bpreserved\\b\", r\"\\bwithin normal limits\\b\",\n    r\"limites normales\", r\"\\bconservad\", r\"\\bintegr\",\n)\nUNCERTAIN = _rx(\n    r\"\\bpossible\\b\", r\"\\bprobable\\b\", r\"\\bsuspicious\\b\", r\"\\bsuspected\\b\",\n    r\"cannot (be )?exclude\", r\"\\bquestionable\\b\", r\"\\bequivocal\\b\", r\"\\bposible\\b\",\n)\nTEAR = _rx(r\"\\btear\", r\"\\btorn\\b\", r\"\\brupture\", r\"\\bdisruption\\b\", r\"\\bavuls\", r\"\\brotura\\b\")\nSEV_HIGH = _rx(r\"\\blarge\\b\", r\"\\bmarked\\b\", r\"\\bmassive\\b\", r\"\\bsevere\\b\", r\"\\bextensive\\b\",\n               r\"\\bmoderate\\b\", r\"\\bsignificant\\b\")\nSEV_LOW = _rx(r\"\\bsmall\\b\", r\"\\bminimal\\b\", r\"\\btrace\\b\", r\"\\bmild\\b\", r\"\\bslight\\b\", r\"\\btiny\\b\")\n\nANATOMY = {\n    \"ACL\": _rx(r\"anterior cruciate\", r\"\\bacl\\b\", r\"cruzado anterior\", r\"\\blca\\b\",\n               r\"croise anterieur\", r\"cruciate ligaments\", r\"ligamentos cruzados\"),\n    \"MCL\": _rx(r\"medial collateral\", r\"\\bmcl\\b\", r\"tibial collateral\", r\"colateral medial\",\n               r\"collateral ligaments\", r\"ligamentos colaterales\"),\n    \"Medial Meniscus\": _rx(r\"medial meniscus\", r\"menisco medial\", r\"menisque interne\"),\n    \"Lateral Meniscus\": _rx(r\"lateral meniscus\", r\"menisco lateral\", r\"menisque externe\"),\n    \"Medial OA\": _rx(r\"medial (compartment )?osteoarthrit\", r\"medial compartment.*(narrow|degen)\"),\n    \"Lateral OA\": _rx(r\"lateral (compartment )?osteoarthrit\", r\"lateral compartment.*(narrow|degen)\"),\n    \"PF OA\": _rx(r\"patellofemoral (compartment )?osteoarthrit\", r\"\\bpf\\b.*osteoarthrit\"),\n    \"Effusion\": _rx(r\"\\beffusion\\b\", r\"derrame articular\", r\"epanchement articulaire\"),\n    \"Synovitis\": _rx(r\"synovitis\", r\"sinovitis\", r\"synovite\"),\n    \"Baker's\": _rx(r\"baker\", r\"popliteal cyst\", r\"quiste de baker\"),\n    \"Contusion\": _rx(r\"bone bruise\", r\"contusion\", r\"marrow (oedema|edema)\"),\n    \"Fracture\": _rx(r\"fractur\\w*\", r\"fractura\\w*\"),\n}\nGLOBAL_OA = _rx(\n    r\"tri ?compartment\", r\"all three compartment\", r\"\\bgonarthros\", r\"osteoarthritis of the knee\",\n)\n\n\ndef _clauses(text: str) -> list[str]:\n    \"\"\"Split a normalised report into clauses, joining `heading:` onto its value.\n\n    A line reading \"Fractures :\" followed by \"None.\" is one statement; splitting\n    on punctuation alone separates the anatomy word from its negation.\n    \"\"\"\n    norm = normalize_text(text)\n    if not norm:\n        return []\n    raw = [c.strip() for c in re.split(r\"(?<=[.;!?])\\s+\", norm) if c.strip()]\n    merged: list[str] = []\n    i = 0\n    while i < len(raw):\n        c = raw[i]\n        if c.endswith(\":\") and i + 1 < len(raw):\n            merged.append(c + \" \" + raw[i + 1])\n            i += 2\n        else:\n            merged.append(c)\n            i += 1\n    return merged\n\n\ndef _polarity(clause: str) -> str:\n    if UNCERTAIN.search(clause):\n        return \"uncertain\"\n    if NEGATION.search(clause):\n        return \"negative\"\n    if NORMALITY.search(clause):\n        if TEAR.search(clause) or re.search(r\"\\bgrade [34]\\b\", clause):\n            return \"positive\"\n        return \"negative\"\n    return \"positive\"\n\n\ndef _severity(clause: str) -> float:\n    if SEV_HIGH.search(clause):\n        return 0.85\n    if SEV_LOW.search(clause):\n        return 0.62\n    return 0.75\n\n\ndef extract_report_labels(text: str) -> dict:\n    \"\"\"Return {target: score in [0,1]} and {target+'__conf': weight in [0,1]}.\"\"\"\n    result = {}\n    clauses = _clauses(text)\n    for target, anat_rx in ANATOMY.items():\n        best_score = None\n        n_mentions = 0\n        for clause in clauses:\n            hit = anat_rx.search(clause) is not None\n            if target in (\"Medial OA\", \"Lateral OA\", \"PF OA\") and GLOBAL_OA.search(clause):\n                hit = True\n            if not hit:\n                continue\n            n_mentions += 1\n            pol = _polarity(clause)\n            if pol == \"negative\":\n                score = 0.12\n            elif pol == \"uncertain\":\n                score = 0.5\n            else:\n                score = _severity(clause)\n            best_score = score if best_score is None else max(best_score, score)\n        if best_score is None:\n            result[target] = 0.28          # weak negative prior, per §1 floor\n            result[target + \"__conf\"] = 0.0\n        else:\n            result[target] = best_score\n            result[target + \"__conf\"] = min(1.0, 0.4 + 0.3 * n_mentions)\n    return result\n\n\ndef build_report_labels(df: pd.DataFrame, report_col: str = \"Report\") -> pd.DataFrame:\n    \"\"\"Vectorised (via list comprehension) report-label extraction with a\n    hard time cap - falls back to neutral scores if it runs long or errors.\"\"\"\n    if report_col not in df.columns:\n        out = pd.DataFrame(0.28, index=df.index, columns=TARGETS)\n        for t in TARGETS:\n            out[t + \"__conf\"] = 0.0\n        return out\n    t0 = time.time()\n    try:\n        records = [extract_report_labels(r) for r in df[report_col].fillna(\"\")]\n        out = pd.DataFrame(records, index=df.index)\n        log.info(f\"extracted report labels for {len(out)} studies in {time.time() - t0:.1f}s\")\n        return out\n    except Exception as exc:\n        log.warning(f\"report label extraction failed ({exc}); using neutral fallback\")\n        out = pd.DataFrame(0.28, index=df.index, columns=TARGETS)\n        for t in TARGETS:\n            out[t + \"__conf\"] = 0.0\n        return out\n\n\n# ---------------------------------------------------------------------------\n#  DICOM series metadata features\n# ---------------------------------------------------------------------------\n\nHDR_TAGS = [\"SeriesDescription\", \"SequenceName\", \"ScanOptions\", \"RepetitionTime\",\n            \"EchoTime\", \"Laterality\", \"ImageLaterality\", \"PixelSpacing\", \"Rows\", \"Columns\"]\n\n\ndef list_series_dirs(root: Path, split: str, study_ids) -> dict:\n    \"\"\"Map StudyInstanceUID -> list of series directories, capped and validated.\"\"\"\n    base = root / split\n    out = {}\n    if not base.is_dir():\n        return out\n    study_set = set(map(str, study_ids))\n    for study_entry in os.scandir(base):\n        if not study_entry.is_dir() or study_entry.name not in study_set:\n            continue\n        series_dirs = []\n        try:\n            for series_entry in os.scandir(study_entry.path):\n                if series_entry.is_dir():\n                    series_dirs.append(Path(series_entry.path))\n        except OSError:\n            continue\n        out[study_entry.name] = series_dirs[:CFG.max_series_per_study]\n    return out\n\n\ndef probe_series(series_dir: Path) -> dict:\n    \"\"\"Read minimal header info from one series without decoding pixels.\"\"\"\n    row = {\"dir\": series_dir, \"n_slices\": 0, \"plane_guess\": None, \"fatsat\": False,\n           \"fluid\": None, \"px\": np.nan, \"files\": []}\n    try:\n        files = sorted(e.name for e in os.scandir(series_dir) if e.name.lower().endswith(\".dcm\"))\n        row[\"files\"] = files\n        row[\"n_slices\"] = len(files)\n        if not files:\n            return row\n        import pydicom\n        mid = files[len(files) // 2]\n        ds = pydicom.dcmread(str(series_dir / mid), stop_before_pixels=True, force=True)\n        desc = f\"{getattr(ds, 'SeriesDescription', '') or ''} {getattr(ds, 'SequenceName', '') or ''}\".lower()\n        opts = str(getattr(ds, \"ScanOptions\", \"\") or \"\").upper()\n        row[\"fatsat\"] = (\"fs\" in desc.split()) or (\"FS\" in opts.split(\"\\\\\")) or (\"fatsat\" in desc)\n        try:\n            tr = float(getattr(ds, \"RepetitionTime\", 0) or 0)\n            row[\"fluid\"] = tr > 800\n        except Exception:\n            row[\"fluid\"] = None\n        if \"sag\" in desc:\n            row[\"plane_guess\"] = \"Sagittal\"\n        elif \"cor\" in desc:\n            row[\"plane_guess\"] = \"Coronal\"\n        elif \"ax\" in desc or \"tra\" in desc:\n            row[\"plane_guess\"] = \"Axial\"\n        px = getattr(ds, \"PixelSpacing\", None)\n        if px:\n            try:\n                row[\"px\"] = float(px[0])\n            except Exception:\n                pass\n    except Exception:\n        pass\n    return row\n\n\ndef build_metadata_features(root: Path, split: str, study_ids, series_df: Optional[pd.DataFrame]) -> pd.DataFrame:\n    \"\"\"Aggregate per-study structural features from train/test_series.csv and headers.\"\"\"\n    study_ids = list(map(str, study_ids))\n    rows = []\n    series_map = list_series_dirs(root, split, study_ids)\n\n    # Optional plane/flags straight from the provided *_series.csv, when present.\n    plane_lookup, fluid_lookup, fatsat_lookup = {}, {}, {}\n    if series_df is not None and \"SeriesInstanceUID\" in series_df.columns:\n        for _, r in series_df.iterrows():\n            sid = str(r.get(\"SeriesInstanceUID\", \"\"))\n            if \"Plane\" in series_df.columns:\n                plane_lookup[sid] = r.get(\"Plane\")\n            if \"Fluid_Sensitive\" in series_df.columns:\n                fluid_lookup[sid] = r.get(\"Fluid_Sensitive\")\n            if \"Fat_Suppression\" in series_df.columns:\n                fatsat_lookup[sid] = r.get(\"Fat_Suppression\")\n\n    for study in study_ids:\n        dirs = series_map.get(study, [])\n        n_series = len(dirs)\n        n_slices_total = 0\n        planes = {\"Sagittal\": 0, \"Coronal\": 0, \"Axial\": 0}\n        n_fatsat, n_fluid = 0, 0\n        n_with_px = 0\n        px_vals = []\n        for d in dirs:\n            probed = probe_series(d)\n            n_slices_total += probed[\"n_slices\"]\n            sid = d.name\n            plane = plane_lookup.get(sid) or probed[\"plane_guess\"]\n            if plane in planes:\n                planes[plane] += 1\n            fatsat = fatsat_lookup.get(sid)\n            fatsat = bool(fatsat) if fatsat is not None else bool(probed[\"fatsat\"])\n            n_fatsat += int(fatsat)\n            fluid = fluid_lookup.get(sid)\n            fluid = fluid if fluid is not None else probed[\"fluid\"]\n            n_fluid += int(bool(fluid))\n            if np.isfinite(probed[\"px\"]):\n                px_vals.append(probed[\"px\"])\n                n_with_px += 1\n        rows.append({\n            \"StudyInstanceUID\": study,\n            \"n_series\": n_series,\n            \"n_slices_total\": n_slices_total,\n            \"n_sagittal\": planes[\"Sagittal\"],\n            \"n_coronal\": planes[\"Coronal\"],\n            \"n_axial\": planes[\"Axial\"],\n            \"n_fatsat\": n_fatsat,\n            \"n_fluid_sensitive\": n_fluid,\n            \"mean_pixel_spacing\": float(np.mean(px_vals)) if px_vals else np.nan,\n            \"has_missing_series_dir\": int(study not in series_map or n_series == 0),\n        })\n    return pd.DataFrame(rows)\n    \n# ---------------------------------------------------------------------------\n#  Image embeddings\n# ---------------------------------------------------------------------------\n\n\ndef _try_import_torch():\n    try:\n        import torch\n        import torchvision\n        return torch, torchvision\n    except Exception as exc:\n        log.warning(f\"torch/torchvision unavailable ({exc}); using statistical image features\")\n        return None, None\n\n\ndef _read_center_slices(series_dir: Path, files: list, n_slice: int, out_size: int) -> Optional[np.ndarray]:\n    \"\"\"Read up to n_slice slices from the central 60% of the stack, resized\n    and normalised to [0, 1] via 1st/99th percentile clipping.\"\"\"\n    if not files:\n        return None\n    n = len(files)\n    lo, hi = int(0.2 * (n - 1)), int(0.8 * (n - 1))\n    idx = np.unique(np.linspace(lo, hi, min(n_slice, n)).astype(int)) if hi > lo else np.array([n // 2])\n    try:\n        import pydicom\n        from PIL import Image\n    except Exception:\n        return None\n    planes = []\n    for i in idx:\n        try:\n            ds = pydicom.dcmread(str(series_dir / files[int(i)]), force=True)\n            arr = ds.pixel_array.astype(np.float32)\n            slope = float(getattr(ds, \"RescaleSlope\", 1) or 1)\n            intercept = float(getattr(ds, \"RescaleIntercept\", 0) or 0)\n            arr = arr * slope + intercept\n            lo_p, hi_p = np.percentile(arr, [1, 99])\n            if hi_p > lo_p:\n                arr = np.clip((arr - lo_p) / (hi_p - lo_p), 0, 1)\n            else:\n                arr = np.zeros_like(arr)\n            img = Image.fromarray((arr * 255).astype(np.uint8)).resize((out_size, out_size))\n            planes.append(np.asarray(img, dtype=np.float32) / 255.0)\n        except Exception:\n            continue\n    if not planes:\n        return None\n    return np.stack(planes)\n\n\ndef _stat_features(vol: np.ndarray, dim: int = 32) -> np.ndarray:\n    \"\"\"Cheap, fixed-length fallback features when a CNN is unavailable.\n\n    Always returns exactly `dim` values regardless of how many slices are\n    present, so it can sit in the same feature slot as the CNN embedding.\n    \"\"\"\n    flat = vol.reshape(-1)\n    if flat.size == 0:\n        return np.zeros(dim, dtype=np.float32)\n    per_slice = vol.reshape(vol.shape[0], -1)\n    feat = np.concatenate([\n        [flat.mean(), flat.std()],\n        np.percentile(flat, [5, 10, 25, 50, 75, 90, 95]),\n        per_slice.mean(axis=1),\n        per_slice.std(axis=1),\n    ]).astype(np.float32)\n    if feat.size < dim:\n        feat = np.pad(feat, (0, dim - feat.size))\n    else:\n        feat = feat[:dim]\n    return feat\n\n\nclass _CNNEmbedder:\n    \"\"\"Thin wrapper around a frozen torchvision backbone; no-op if torch is missing.\"\"\"\n\n    def __init__(self):\n        self.torch, self.torchvision = _try_import_torch()\n        self.model = None\n        self.device = \"cpu\"\n        self.out_dim = 32  # dim of the statistical fallback\n        if self.torch is None:\n            return\n        try:\n            import torch\n            from torchvision import models\n            self.device = \"cuda\" if torch.cuda.is_available() else \"cpu\"\n            weights = None\n            try:\n                weights = models.ResNet18_Weights.IMAGENET1K_V1\n            except Exception:\n                weights = None\n            backbone = models.resnet18(weights=weights)\n            backbone.fc = torch.nn.Identity()\n            backbone.eval().to(self.device)\n            for p in backbone.parameters():\n                p.requires_grad_(False)\n            self.model = backbone\n            self.out_dim = 512\n            CFG.have_torch = True\n            CFG.device = self.device\n            log.info(f\"image embedder ready: resnet18 on {self.device} \"\n                     f\"(pretrained={weights is not None})\")\n        except Exception as exc:\n            log.warning(f\"could not build CNN embedder ({exc}); using statistical fallback\")\n            self.model = None\n\n    @property\n    def dim(self) -> int:\n        return self.out_dim\n\n    def embed_batch(self, vols: list) -> np.ndarray:\n        \"\"\"vols: list of [n_slice, H, W] float arrays in [0,1]. Returns [B, dim].\"\"\"\n        if self.model is None:\n            return np.stack([_stat_features(v, self.out_dim) for v in vols])\n        torch = self.torch\n        feats = []\n        with torch.no_grad():\n            for v in vols:\n                # average the per-slice feature rather than stacking as channels,\n                # so the embedding is robust to a variable slice count\n                x = torch.from_numpy(v).unsqueeze(1).repeat(1, 3, 1, 1).float().to(self.device)\n                x = (x - 0.45) / 0.225\n                out = self.model(x)  # [n_slice, 512]\n                feats.append(out.mean(0).cpu().numpy())\n        return np.stack(feats)\n\n\ndef build_image_features(root: Path, split: str, study_ids, embedder: _CNNEmbedder,\n                          deadline: float) -> pd.DataFrame:\n    \"\"\"One row per study: a pooled embedding across up to max_series_per_study series.\"\"\"\n    study_ids = list(map(str, study_ids))\n    series_map = list_series_dirs(root, split, study_ids)\n    dim = embedder.dim\n    feats = np.zeros((len(study_ids), dim), dtype=np.float32)\n    coverage = np.zeros(len(study_ids), dtype=np.float32)\n\n    for i, study in enumerate(study_ids):\n        if time.time() > deadline:\n            log.warning(f\"{split}: image feature deadline reached at study {i}/{len(study_ids)}\")\n            break\n        dirs = series_map.get(study, [])\n        vols = []\n        for d in dirs:\n            try:\n                files = sorted(e.name for e in os.scandir(d) if e.name.lower().endswith(\".dcm\"))\n            except OSError:\n                continue\n            vol = _read_center_slices(d, files, CFG.max_slices_per_series, CFG.image_size)\n            if vol is not None:\n                vols.append(vol)\n        if not vols:\n            continue\n        try:\n            per_series = embedder.embed_batch(vols)  # [n_series_present, dim]\n            feats[i] = per_series.mean(axis=0)\n            coverage[i] = len(vols) / max(len(dirs), 1)\n        except Exception as exc:\n            log.warning(f\"embedding failed for study {study}: {exc}\")\n        if i % 200 == 0 and i > 0:\n            log.info(f\"{split}: embedded {i}/{len(study_ids)} studies \"\n                      f\"({elapsed():.0f}s elapsed)\")\n\n    cols = [f\"img_{j}\" for j in range(dim)]\n    out = pd.DataFrame(feats, columns=cols)\n    out.insert(0, \"StudyInstanceUID\", study_ids)\n    out[\"img_coverage\"] = coverage\n    return out\n\n\n# ---------------------------------------------------------------------------\n#  Feature Engineering \n# ---------------------------------------------------------------------------\n\ndef assemble_features(meta_feats: pd.DataFrame, img_feats: pd.DataFrame) -> pd.DataFrame:\n    df = meta_feats.merge(img_feats, on=\"StudyInstanceUID\", how=\"left\")\n    num_cols = [c for c in df.columns if c != \"StudyInstanceUID\"]\n    df[num_cols] = df[num_cols].apply(pd.to_numeric, errors=\"coerce\")\n    df[num_cols] = df[num_cols].fillna(df[num_cols].median(numeric_only=True))\n    df[num_cols] = df[num_cols].fillna(0.0)\n    return df\n\n\n# ---------------------------------------------------------------------------\n#  Model training and cross-validation\n# ---------------------------------------------------------------------------\n\ndef _import_model_backend():\n    try:\n        import lightgbm as lgb\n        CFG.have_lightgbm = True\n        return \"lightgbm\", lgb\n    except Exception:\n        from sklearn.ensemble import RandomForestClassifier\n        log.warning(\"lightgbm unavailable; falling back to RandomForestClassifier\")\n        return \"sklearn\", RandomForestClassifier\n\n\ndef make_estimator(backend_name: str, backend_mod, n_pos: int, n_neg: int):\n    if backend_name == \"lightgbm\":\n        import lightgbm as lgb\n        pos_weight = max(1.0, n_neg / max(n_pos, 1))\n        return lgb.LGBMClassifier(\n            n_estimators=300, learning_rate=0.03, num_leaves=15, max_depth=4,\n            min_child_samples=10, subsample=0.8, colsample_bytree=0.7,\n            reg_lambda=1.0, scale_pos_weight=min(pos_weight, 20.0),\n            random_state=CFG.random_seed, n_jobs=4, verbosity=-1,\n        )\n    return backend_mod(\n        n_estimators=300, max_depth=6, min_samples_leaf=3, class_weight=\"balanced\",\n        random_state=CFG.random_seed, n_jobs=4,\n    )\n\n\ndef hash_groups(text_series: pd.Series) -> np.ndarray:\n    \"\"\"Group id per study for CV splitting: studies sharing an identical\n    report (a template read for an unremarkable knee) must stay in the\n    same fold, or the derived label leaks across the split.\"\"\"\n    norm = text_series.fillna(\"\").map(normalize_text)\n    \n    is_empty = norm == \"\"\n    groups = norm.copy()\n    groups[is_empty] = [f\"__unique_{i}\" for i in np.flatnonzero(is_empty.values)]\n    return pd.factorize(groups)[0]\n\n\ndef macro_auc(y: np.ndarray, p: np.ndarray) -> float:\n    from sklearn.metrics import roc_auc_score\n    aucs = []\n    for j in range(y.shape[1]):\n        col = y[:, j]\n        if len(np.unique(col)) < 2:\n            continue\n        try:\n            aucs.append(roc_auc_score(col, p[:, j]))\n        except Exception:\n            continue\n    return float(np.mean(aucs)) if aucs else float(\"nan\")\n\n\ndef train_and_predict(X: pd.DataFrame, Y: pd.DataFrame, W: pd.DataFrame,\n                       X_test: pd.DataFrame, groups: np.ndarray, deadline: float):\n    \"\"\"GroupKFold training with per-target boosted trees; returns\n    (test_predictions [rank-mean over folds], per-target OOF AUC table).\"\"\"\n    from sklearn.model_selection import GroupKFold\n\n    feature_cols = [c for c in X.columns if c != \"StudyInstanceUID\"]\n    Xv = X[feature_cols].values.astype(np.float32)\n    Xtv = X_test[feature_cols].values.astype(np.float32)\n\n    backend_name, backend_mod = _import_model_backend()\n\n    n_folds = min(CFG.n_folds, len(np.unique(groups)))\n    n_folds = max(2, n_folds)\n    gkf = GroupKFold(n_splits=n_folds)\n\n    oof_pred = np.zeros((len(X), len(TARGETS)), dtype=np.float32)\n    test_rank_sum = np.zeros((len(X_test), len(TARGETS)), dtype=np.float64)\n    n_folds_run = 0\n\n    for fold, (tr_idx, va_idx) in enumerate(gkf.split(Xv, groups=groups)):\n        if time.time() > deadline:\n            log.warning(f\"training deadline reached before fold {fold}; stopping CV\")\n            break\n        fold_t0 = time.time()\n        fold_test_pred = np.zeros((len(X_test), len(TARGETS)), dtype=np.float32)\n        for j, target in enumerate(TARGETS):\n            y_tr = (Y.iloc[tr_idx, j].values > 0.5).astype(int)\n            w_tr = W.iloc[tr_idx, j].values\n            n_pos, n_neg = int(y_tr.sum()), int((1 - y_tr).sum())\n            if n_pos < 2 or n_neg < 2:\n                # degenerate target in this fold: predict the training prevalence\n                fallback = float(np.clip(y_tr.mean() if len(y_tr) else 0.28, 0.05, 0.95))\n                oof_pred[va_idx, j] = fallback\n                fold_test_pred[:, j] = fallback\n                continue\n            try:\n                est = make_estimator(backend_name, backend_mod, n_pos, n_neg)\n                est.fit(Xv[tr_idx], y_tr, sample_weight=w_tr)\n                oof_pred[va_idx, j] = est.predict_proba(Xv[va_idx])[:, 1]\n                fold_test_pred[:, j] = est.predict_proba(Xtv)[:, 1]\n            except Exception as exc:\n                log.warning(f\"fold {fold} target {target} failed ({exc}); using prevalence fallback\")\n                fallback = float(np.clip(y_tr.mean(), 0.05, 0.95))\n                oof_pred[va_idx, j] = fallback\n                fold_test_pred[:, j] = fallback\n        \n        ranks = pd.DataFrame(fold_test_pred).rank(pct=True).values\n        test_rank_sum += ranks\n        n_folds_run += 1\n        log.info(f\"fold {fold} done in {time.time() - fold_t0:.1f}s \"\n                 f\"({elapsed():.0f}s total elapsed)\")\n\n    if n_folds_run == 0:\n        log.warning(\"no folds completed; falling back to global prevalence predictions\")\n        prevalence = np.clip(Y.mean(axis=0).values, 0.05, 0.95)\n        test_pred = np.tile(prevalence, (len(X_test), 1))\n    else:\n        test_pred = (test_rank_sum / n_folds_run).astype(np.float32)\n\n    oof_auc = macro_auc((Y.values > 0.5).astype(int), oof_pred)\n    per_target = []\n    for j, t in enumerate(TARGETS):\n        col_y = (Y.values[:, j] > 0.5).astype(int)\n        if len(np.unique(col_y)) < 2:\n            per_target.append({\"target\": t, \"oof_auc\": np.nan, \"n_pos\": int(col_y.sum())})\n            continue\n        from sklearn.metrics import roc_auc_score\n        per_target.append({\n            \"target\": t, \"oof_auc\": roc_auc_score(col_y, oof_pred[:, j]),\n            \"n_pos\": int(col_y.sum()),\n        })\n    return test_pred, oof_auc, pd.DataFrame(per_target)\n\n\n# ---------------------------------------------------------------------------\n# EDA\n# ---------------------------------------------------------------------------\n\ndef run_eda(train_df: pd.DataFrame, out_dir: Path) -> None:\n    \"\"\"Save a handful of diagnostic plots; never raises - EDA is informative,\n    not load-bearing, so any failure here is caught and logged only.\"\"\"\n    try:\n        import matplotlib\n        matplotlib.use(\"Agg\")\n        import matplotlib.pyplot as plt\n\n        present_targets = [t for t in TARGETS if t in train_df.columns]\n        if present_targets:\n            prevalence = train_df[present_targets].apply(\n                lambda s: pd.to_numeric(s, errors=\"coerce\")).mean().sort_values()\n            fig, ax = plt.subplots(figsize=(8, 5))\n            ax.barh(prevalence.index, prevalence.values, color=\"#4d6d9a\")\n            ax.set_xlabel(\"Positive rate (annotated subset)\")\n            ax.set_title(\"Target prevalence\")\n            fig.tight_layout()\n            fig.savefig(out_dir / \"eda_target_prevalence.png\", dpi=120)\n            plt.close(fig)\n\n            annotated_frac = train_df[present_targets].notna().all(axis=1).mean()\n            log.info(f\"EDA: {annotated_frac:.1%} of studies carry complete annotations\")\n\n        if \"Report\" in train_df.columns:\n            lengths = train_df[\"Report\"].fillna(\"\").str.len()\n            fig, ax = plt.subplots(figsize=(7, 4))\n            ax.hist(lengths[lengths > 0], bins=30, color=\"#b4636b\")\n            ax.set_xlabel(\"Report length (characters)\")\n            ax.set_title(\"Report length distribution\")\n            fig.tight_layout()\n            fig.savefig(out_dir / \"eda_report_length.png\", dpi=120)\n            plt.close(fig)\n            log.info(f\"EDA: {int((lengths == 0).sum())} studies have an empty report\")\n    except Exception as exc:\n        log.warning(f\"EDA skipped due to error: {exc}\")\n\n\n# ---------------------------------------------------------------------------\n# Submission \n# ---------------------------------------------------------------------------\n\ndef write_submission(test_study_ids, preds: np.ndarray, out_dir: Path) -> pd.DataFrame:\n    sub = pd.DataFrame(preds, columns=TARGETS)\n    sub.insert(0, \"StudyInstanceUID\", list(map(str, test_study_ids)))\n    sub[TARGETS] = sub[TARGETS].clip(0.0, 1.0).fillna(0.5)\n    # exact column order and names required by the competition submission format\n    sub = sub[[\"StudyInstanceUID\"] + TARGETS]\n    sub.to_csv(out_dir / CFG.submission_name, index=False)\n    return sub\n\n\n# ---------------------------------------------------------------------------\n# Main\n# ---------------------------------------------------------------------------\n\ndef main() -> None:\n    out_dir = resolve_output_dir()\n    root = find_data_root()\n    log.info(f\"data root: {root}\")\n \n    write_benchmark_submission(root, out_dir)\n\n    train_df = safe_read_csv(root / \"train.csv\")\n    test_df = safe_read_csv(root / \"test.csv\")\n    if test_df is None or \"StudyInstanceUID\" not in test_df.columns:\n        raise FileNotFoundError(\"test.csv is missing or malformed (no StudyInstanceUID column)\")\n    if train_df is None or \"StudyInstanceUID\" not in train_df.columns:\n        raise FileNotFoundError(\"train.csv is missing or malformed (no StudyInstanceUID column)\")\n\n    train_series_df = safe_read_csv(root / \"train_series.csv\")\n    test_series_df = safe_read_csv(root / \"test_series.csv\")\n\n    run_eda(train_df, out_dir)\n\n    # -- labels: annotated where available, report-derived elsewhere --------\n    present_targets = [t for t in TARGETS if t in train_df.columns]\n    has_annotation = (\n        train_df[present_targets].notna().all(axis=1)\n        if present_targets else pd.Series(False, index=train_df.index)\n    )\n    report_labels = build_report_labels(train_df)\n\n    Y = pd.DataFrame(index=train_df.index, columns=TARGETS, dtype=float)\n    W = pd.DataFrame(index=train_df.index, columns=TARGETS, dtype=float)\n    for t in TARGETS:\n        annotated_vals = pd.to_numeric(train_df[t], errors=\"coerce\") if t in train_df.columns \\\n            else pd.Series(np.nan, index=train_df.index)\n        derived_vals = report_labels[t]\n        derived_conf = report_labels[t + \"__conf\"]\n        use_annotation = annotated_vals.notna()\n        Y[t] = np.where(use_annotation, annotated_vals, derived_vals)\n        W[t] = np.where(use_annotation, CFG.gold_sample_weight, np.clip(derived_conf, 0.05, 1.0))\n    Y = Y.fillna(0.28).astype(float)\n    W = W.fillna(0.05).astype(float)\n    log.info(f\"labels ready: {int(has_annotation.sum())} annotated, \"\n             f\"{len(train_df) - int(has_annotation.sum())} report-derived studies\")\n\n    # -- structural (metadata) features --------------------------------------\n    meta_train = build_metadata_features(root, \"train_series\", train_df[\"StudyInstanceUID\"], train_series_df)\n    meta_test = build_metadata_features(root, \"test_series\", test_df[\"StudyInstanceUID\"], test_series_df)\n\n    # -- image embeddings, bounded by the remaining time budget --------------\n    embedder = _CNNEmbedder()\n    deadline = T_START + CFG.time_budget_sec\n    \n    img_deadline = min(deadline, time.time() + max(60.0, (deadline - time.time()) * 0.5))\n    img_train = build_image_features(root, \"train_series\", train_df[\"StudyInstanceUID\"], embedder, img_deadline)\n    img_test = build_image_features(root, \"test_series\", test_df[\"StudyInstanceUID\"], embedder, deadline)\n    del embedder\n    gc.collect()\n\n    X_train = assemble_features(meta_train, img_train)\n    X_test = assemble_features(meta_test, img_test)\n    \n    common_cols = [c for c in X_train.columns if c in X_test.columns]\n    X_train, X_test = X_train[common_cols], X_test[common_cols]\n    log.info(f\"feature matrix: {X_train.shape[1] - 1} features, \"\n             f\"{len(X_train)} train / {len(X_test)} test studies\")\n\n    groups = hash_groups(train_df.get(\"Report\", pd.Series(\"\", index=train_df.index)))\n\n    train_deadline = T_START + CFG.time_budget_sec\n    test_pred, oof_auc, per_t\n    ","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}