{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport pandas as pd\n\n# 1. What dataset folders are actually mounted under /kaggle/input?\nprint(\"Top-level folders under /kaggle/input:\")\nfor name in os.listdir('/kaggle/input'):\n    print(' -', name)\n\n# 2. Find every CSV anywhere under /kaggle/input and show its columns\nprint(\"\\nCSV files found:\")\nfor root, dirs, files in os.walk('/kaggle/input'):\n    for f in files:\n        if f.endswith('.csv'):\n            path = os.path.join(root, f)\n            try:\n                df = pd.read_csv(path, nrows=3)\n                print(f\"\\n{path}\")\n                print(\"  columns:\", list(df.columns))\n                print(df.head(2).to_string())\n            except Exception as e:\n                print(f\"\\n{path}  (couldn't read: {e})\")\n\n# 3. Peek at the folder structure a few levels deep, to see where the\n#    actual DICOM/image files live relative to whatever the CSV references\nprint(\"\\nDirectory structure (first 3 levels):\")\nfor root, dirs, files in os.walk('/kaggle/input'):\n    depth = root[len('/kaggle/input'):].count(os.sep)\n    if depth > 3:\n        dirs[:] = []\n        continue\n    indent = '  ' * depth\n    print(f\"{indent}{os.path.basename(root) or root}/  ({len(files)} files)\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-12T07:35:47.868924Z","iopub.execute_input":"2026-09-12T07:35:47.869211Z","iopub.status.idle":"2026-09-12T07:39:59.38847Z","shell.execute_reply.started":"2026-09-12T07:35:47.869184Z","shell.execute_reply":"2026-09-12T07:39:59.387454Z"}},"outputs":[{"name":"stdout","text":"Top-level folders under /kaggle/input:\n - competitions\n\nCSV files found:\n\n/kaggle/input/competitions/rsna-knee-abnormality-detection/test_series.csv\n  columns: ['StudyInstanceUID', 'SeriesInstanceUID', 'Fluid_Sensitive', 'Fat_Suppression', 'Anatomical_Plane']\n                                                   StudyInstanceUID                                                 SeriesInstanceUID  Fluid_Sensitive  Fat_Suppression Anatomical_Plane\n0  1.2.826.0.1.3680043.8.498.10047035057544427318018579121635276191  1.2.826.0.1.3680043.8.498.11580656442259111255675562605155903947                0                0            Axial\n1  1.2.826.0.1.3680043.8.498.10047035057544427318018579121635276191  1.2.826.0.1.3680043.8.498.17811502614030631664517622518906646132                0                0         Sagittal\n\n/kaggle/input/competitions/rsna-knee-abnormality-detection/train_series.csv\n  columns: ['StudyInstanceUID', 'SeriesInstanceUID', 'Fluid_Sensitive', 'Fat_Suppression', 'Anatomical_Plane']\n                                                   StudyInstanceUID                                                 SeriesInstanceUID  Fluid_Sensitive  Fat_Suppression Anatomical_Plane\n0  1.2.826.0.1.3680043.8.498.10004873229099053869093324292195817260  1.2.826.0.1.3680043.8.498.12343110195036213483454091715412333772                1                1         Sagittal\n1  1.2.826.0.1.3680043.8.498.10004873229099053869093324292195817260  1.2.826.0.1.3680043.8.498.13821229744997220641575291927426543265                1                1            Axial\n\n/kaggle/input/competitions/rsna-knee-abnormality-detection/sample_submission.csv\n  columns: ['StudyInstanceUID', 'ACL', 'MCL', 'Medial Meniscus', 'Lateral Meniscus', 'Medial OA', 'Lateral OA', 'PF OA', 'Effusion', 'Synovitis', \"Baker's\", 'Contusion', 'Fracture']\n                                                   StudyInstanceUID  ACL  MCL  Medial Meniscus  Lateral Meniscus  Medial OA  Lateral OA  PF OA  Effusion  Synovitis  Baker's  Contusion  Fracture\n0  1.2.826.0.1.3680043.8.498.10047035057544427318018579121635276191  0.5  0.5              0.5               0.5        0.5         0.5    0.5       0.5        0.5      0.5        0.5       0.5\n1  1.2.826.0.1.3680043.8.498.10062861783145312629332250977456991776  0.5  0.5              0.5               0.5        0.5         0.5    0.5       0.5        0.5      0.5        0.5       0.5\n\n/kaggle/input/competitions/rsna-knee-abnormality-detection/train.csv\n  columns: ['StudyInstanceUID', 'Report', 'ACL', 'MCL', 'Medial Meniscus', 'Lateral Meniscus', 'Medial OA', 'Lateral OA', 'PF OA', 'Effusion', 'Synovitis', \"Baker's\", 'Contusion', 'Fracture']\n                                                   StudyInstanceUID                                                                                                                                                                                                                                                                                                                              Report  ACL  MCL  Medial Meniscus  Lateral Meniscus  Medial OA  Lateral OA  PF OA  Effusion  Synovitis  Baker's  Contusion  Fracture\n0  1.2.826.0.1.3680043.8.498.10004873229099053869093324292195817260  Técnica: RMN de la rodilla. Resultados: Rotura de menisco interno. Signo de necrosis avascular subcondral en el cóndilo femoral medial. Artrosis femorotibial medial. Derrame. . Impresión: Rotura de menisco interno. Signo de necrosis avascular subcondral en el cóndilo femoral medial. Artrosis femorotibial medial. Derrame.  NaN  NaN              NaN               NaN        NaN         NaN    NaN       NaN        NaN      NaN        NaN       NaN\n1  1.2.826.0.1.3680043.8.498.10004945927472656027199792075652399585                                                                               [DATE]: * MR Knie Rechts 15ch AA Klinische Inlichtingen: [DATE]. Diagnostische vraagstellling: Meniscusscheur/mediaal? Scanprotocol (DRB) : sag intermediair gewogen seq zonder en met fs, ax/ cor pd gewogen seq fs, cor T1 gewogen seq Bevindingen:  NaN  NaN              NaN               NaN        NaN         NaN    NaN       NaN        NaN      NaN        NaN       NaN\n\n/kaggle/input/competitions/rsna-knee-abnormality-detection/test.csv\n  columns: ['StudyInstanceUID']\n                                                   StudyInstanceUID\n0  1.2.826.0.1.3680043.8.498.10047035057544427318018579121635276191\n1  1.2.826.0.1.3680043.8.498.10062861783145312629332250977456991776\n\nDirectory structure (first 3 levels):\ninput/  (0 files)\n  competitions/  (0 files)\n    rsna-knee-abnormality-detection/  (5 files)\n      test_series/  (0 files)\n      train_series/  (0 files)\n","output_type":"stream"}],"execution_count":2},{"cell_type":"code","source":"import os\nimport pandas as pd\n\ncomp_root = '/kaggle/input/competitions/rsna-knee-abnormality-detection'\n\nprint(\"Everything directly under the competition folder:\")\nfor name in sorted(os.listdir(comp_root)):\n    print(' -', name)\n\n# Drill into whatever looks like an image folder (adjust the name below\n# once you see the listing above)\nfor candidate in ['train_images', 'train_dicoms', 'images', 'train']:\n    path = os.path.join(comp_root, candidate)\n    if os.path.isdir(path):\n        print(f\"\\nFound image-like folder: {candidate}\")\n        # one example study, drilled 3 levels deep\n        first_study = sorted(os.listdir(path))[0]\n        study_path = os.path.join(path, first_study)\n        print(f\"Contents of one study folder ({first_study}):\")\n        for root, dirs, files in os.walk(study_path):\n            depth = root[len(study_path):].count(os.sep)\n            print('  ' * depth + os.path.basename(root) + f'/  ({len(files)} files)')\n            if files:\n                print('  ' * (depth + 1) + 'example file:', files[0])\n        break\n\n# Check the actual value domain of the 12 label columns\ntrain_df = pd.read_csv(os.path.join(comp_root, 'train.csv'))\nlabel_cols = ['ACL', 'MCL', 'Medial Meniscus', 'Lateral Meniscus', 'Medial OA',\n              'Lateral OA', 'PF OA', 'Effusion', 'Synovitis', \"Baker's\",\n              'Contusion', 'Fracture']\nprint(\"\\nLabel value domains:\")\nfor c in label_cols:\n    vals = train_df[c].dropna().unique()\n    print(f\"  {c}: {sorted(vals)[:10]} (n non-null = {train_df[c].notna().sum()} / {len(train_df)})\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-12T07:39:59.683911Z","iopub.execute_input":"2026-09-12T07:39:59.68417Z","iopub.status.idle":"2026-09-12T07:39:59.770854Z","shell.execute_reply.started":"2026-09-12T07:39:59.684144Z","shell.execute_reply":"2026-09-12T07:39:59.770174Z"}},"outputs":[{"name":"stdout","text":"Everything directly under the competition folder:\n - sample_submission.csv\n - test.csv\n - test_series\n - test_series.csv\n - train.csv\n - train_series\n - train_series.csv\n\nLabel value domains:\n  ACL: [np.float64(0.0), np.float64(1.0)] (n non-null = 58 / 4407)\n  MCL: [np.float64(0.0), np.float64(1.0)] (n non-null = 58 / 4407)\n  Medial Meniscus: [np.float64(0.0), np.float64(1.0)] (n non-null = 58 / 4407)\n  Lateral Meniscus: [np.float64(0.0), np.float64(1.0)] (n non-null = 58 / 4407)\n  Medial OA: [np.float64(0.0), np.float64(1.0)] (n non-null = 58 / 4407)\n  Lateral OA: [np.float64(0.0), np.float64(1.0)] (n non-null = 58 / 4407)\n  PF OA: [np.float64(0.0), np.float64(1.0)] (n non-null = 58 / 4407)\n  Effusion: [np.float64(0.0), np.float64(1.0)] (n non-null = 58 / 4407)\n  Synovitis: [np.float64(0.0), np.float64(1.0)] (n non-null = 58 / 4407)\n  Baker's: [np.float64(0.0), np.float64(1.0)] (n non-null = 58 / 4407)\n  Contusion: [np.float64(0.0), np.float64(1.0)] (n non-null = 58 / 4407)\n  Fracture: [np.float64(0.0), np.float64(1.0)] (n non-null = 58 / 4407)\n","output_type":"stream"}],"execution_count":6},{"cell_type":"code","source":"import os\nimport pandas as pd\n\ncomp_root = '/kaggle/input/competitions/rsna-knee-abnormality-detection'\nseries_root = os.path.join(comp_root, 'train_series')\n\n# Drill into one study's folder to see the real structure and file extensions\nfirst_study = sorted(os.listdir(series_root))[0]\nstudy_path = os.path.join(series_root, first_study)\nprint(f\"Structure under one study ({first_study}):\")\nfor root, dirs, files in os.walk(study_path):\n    depth = root[len(study_path):].count(os.sep)\n    print('  ' * depth + os.path.basename(root) + f'/  ({len(files)} files)')\n    if files:\n        print('  ' * (depth + 1) + 'example files:', files[:3])\n\n# How many studies have Report text at all, and how long are the reports?\ntrain_df = pd.read_csv(os.path.join(comp_root, 'train.csv'))\nprint(f\"\\nStudies with non-null Report: {train_df['Report'].notna().sum()} / {len(train_df)}\")\n\n# Per-label positive counts among the 58 gold-labeled studies\nlabel_cols = ['ACL', 'MCL', 'Medial Meniscus', 'Lateral Meniscus', 'Medial OA',\n              'Lateral OA', 'PF OA', 'Effusion', 'Synovitis', \"Baker's\",\n              'Contusion', 'Fracture']\ngold = train_df.dropna(subset=label_cols, how='all')\nprint(f\"\\nGold-labeled subset: {len(gold)} studies\")\nprint(\"Per-finding positive count:\")\nfor c in label_cols:\n    print(f\"  {c}: {int(gold[c].sum())} positive / {len(gold)}\")\n\nany_abnormal = (gold[label_cols].sum(axis=1) > 0).astype(int)\nprint(f\"\\nPooled 'any abnormality' target: {any_abnormal.sum()} positive / {len(gold)}\")\n\n# Confirm which series (plane/sequence) exist for the gold-labeled studies\ntrain_series_df = pd.read_csv(os.path.join(comp_root, 'train_series.csv'))\ngold_series = train_series_df[train_series_df['StudyInstanceUID'].isin(gold['StudyInstanceUID'])]\nprint(f\"\\nSeries available for gold studies, by plane/fluid-sensitivity:\")\nprint(gold_series.groupby(['Anatomical_Plane', 'Fluid_Sensitive']).size())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-12T07:50:15.957038Z","iopub.execute_input":"2026-09-12T07:50:15.957333Z","iopub.status.idle":"2026-09-12T07:50:16.14311Z","shell.execute_reply.started":"2026-09-12T07:50:15.957305Z","shell.execute_reply":"2026-09-12T07:50:16.142021Z"}},"outputs":[{"name":"stdout","text":"Structure under one study (1.2.826.0.1.3680043.8.498.10004873229099053869093324292195817260):\n1.2.826.0.1.3680043.8.498.10004873229099053869093324292195817260/  (0 files)\n  1.2.826.0.1.3680043.8.498.13821229744997220641575291927426543265/  (18 files)\n    example files: ['1.2.826.0.1.3680043.8.498.12732955977266558882646974219653035510.dcm', '1.2.826.0.1.3680043.8.498.59350981989101065846464866297147921212.dcm', '1.2.826.0.1.3680043.8.498.51290044590824165461429354587015028697.dcm']\n  1.2.826.0.1.3680043.8.498.75714899997203615784077798038670363546/  (22 files)\n    example files: ['1.2.826.0.1.3680043.8.498.93854001174489428321379491457658840661.dcm', '1.2.826.0.1.3680043.8.498.39561070338793988533091874317667505814.dcm', '1.2.826.0.1.3680043.8.498.80771320912561593500846833541716910554.dcm']\n  1.2.826.0.1.3680043.8.498.40734206102458723096154687147390476697/  (16 files)\n    example files: ['1.2.826.0.1.3680043.8.498.31066553483567909142189031879812297531.dcm', '1.2.826.0.1.3680043.8.498.84940587737063172414458197536498603352.dcm', '1.2.826.0.1.3680043.8.498.99435129020823778247825836092850724032.dcm']\n  1.2.826.0.1.3680043.8.498.23084836536722595275828690293168736174/  (16 files)\n    example files: ['1.2.826.0.1.3680043.8.498.12098739564387780587308082414503153172.dcm', '1.2.826.0.1.3680043.8.498.74090003727178597144898278950251365655.dcm', '1.2.826.0.1.3680043.8.498.21821900345293357201908464058987983501.dcm']\n  1.2.826.0.1.3680043.8.498.12343110195036213483454091715412333772/  (22 files)\n    example files: ['1.2.826.0.1.3680043.8.498.10374285052733466977592225981851620435.dcm', '1.2.826.0.1.3680043.8.498.11560456494251875828657453454342564666.dcm', '1.2.826.0.1.3680043.8.498.62696444312933722854177245172284641767.dcm']\n\nStudies with non-null Report: 4407 / 4407\n\nGold-labeled subset: 58 studies\nPer-finding positive count:\n  ACL: 24 positive / 58\n  MCL: 9 positive / 58\n  Medial Meniscus: 26 positive / 58\n  Lateral Meniscus: 23 positive / 58\n  Medial OA: 15 positive / 58\n  Lateral OA: 11 positive / 58\n  PF OA: 21 positive / 58\n  Effusion: 35 positive / 58\n  Synovitis: 27 positive / 58\n  Baker's: 12 positive / 58\n  Contusion: 19 positive / 58\n  Fracture: 18 positive / 58\n\nPooled 'any abnormality' target: 58 positive / 58\n\nSeries available for gold studies, by plane/fluid-sensitivity:\nAnatomical_Plane  Fluid_Sensitive\nAxial             0                  18\n                  1                  63\nCoronal           0                  54\n                  1                  65\nSagittal          0                  72\n                  1                  64\ndtype: int64\n","output_type":"stream"}],"execution_count":8},{"cell_type":"code","source":"\"\"\"\nknee_mri_pipeline_allinone.py  (v2 - matched to real competition data)\n\nHandcrafted (HOG/LBP/GLCM) + deep CNN feature fusion, classified with\nCatBoost/AdaBoost, for the RSNA Knee Abnormality Detection competition.\n\nWHAT CHANGED FROM v1\n---------------------\nReal data exploration showed:\n  - Only 58 of 4,407 studies have radiologist-verified 0/1 labels across 12\n    findings (ACL, MCL, Medial/Lateral Meniscus, Medial/Lateral OA, PF OA,\n    Effusion, Synovitis, Baker's, Contusion, Fracture). The rest only have\n    free-text multilingual reports.\n  - All 58 gold-labeled studies are positive for >=1 finding, so a pooled\n    \"any abnormality\" target is degenerate (no negative class). This\n    pipeline instead targets ONE finding at a time (TARGET_FINDING below).\n  - Series aren't tagged by sequence name - they're tagged by Anatomical_Plane\n    plus two binary flags, Fluid_Sensitive and Fat_Suppression. Series files\n    live at train_series/{StudyInstanceUID}/{SeriesInstanceUID}/*.dcm.\n  - n=58 is small-sample territory: this version uses leave-one-out\n    cross-validation with aggregated out-of-fold metrics (standard practice\n    at this sample size) instead of 5-fold CV, and coarsens the HOG grid /\n    caps PCA components to reduce overfitting risk.\n\nHOW TO USE ON KAGGLE\n---------------------\n1. Add the competition data to your notebook.\n2. Optionally change TARGET_FINDING below (best-balanced options at n=58:\n   'ACL' 24/58, 'Medial Meniscus' 26/58, 'Effusion' 35/58, 'Synovitis' 27/58).\n3. Run the whole file. It prints the class balance for your chosen finding,\n   builds features for handcrafted/deep/fused modes, runs LOOCV with\n   CatBoost for each, and prints an aggregated-metrics summary table.\n\nInstall anything missing with: !pip install -q pydicom catboost shap\n\"\"\"\n\nfrom __future__ import annotations\n\nimport os\nimport time\nfrom dataclasses import dataclass\nfrom typing import Optional\n\nimport numpy as np\nimport pandas as pd\nfrom skimage.feature import hog, local_binary_pattern, graycomatrix, graycoprops\nfrom skimage.transform import resize\nfrom sklearn.decomposition import PCA\nfrom sklearn.model_selection import StratifiedKFold, LeaveOneOut\nfrom sklearn.metrics import roc_auc_score, f1_score, accuracy_score, confusion_matrix\nfrom sklearn.preprocessing import StandardScaler\n\ntry:\n    import pydicom\nexcept ImportError:\n    pydicom = None\n\n\n# ==========================================================================\n# CONFIG\n# ==========================================================================\n\nCOMP_ROOT = \"/kaggle/input/competitions/rsna-knee-abnormality-detection\"\nTRAIN_CSV = os.path.join(COMP_ROOT, \"train.csv\")\nTRAIN_SERIES_CSV = os.path.join(COMP_ROOT, \"train_series.csv\")\nTRAIN_SERIES_ROOT = os.path.join(COMP_ROOT, \"train_series\")\n\nLABEL_COLUMNS = [\"ACL\", \"MCL\", \"Medial Meniscus\", \"Lateral Meniscus\", \"Medial OA\",\n                  \"Lateral OA\", \"PF OA\", \"Effusion\", \"Synovitis\", \"Baker's\",\n                  \"Contusion\", \"Fracture\"]\n\n# Which finding to predict. At n=58 gold-labeled studies, the best-balanced\n# options are: ACL (24/58), Medial Meniscus (26/58), Effusion (35/58),\n# Synovitis (27/58). Starting with ACL for direct comparability to MRNet's\n# own ACL-tear benchmark numbers.\nTARGET_FINDING = \"ACL\"\n\nN_KEY_SLICES = 5\nPCA_COMPONENTS = 20     # kept small given n=58 - see note in train_and_evaluate\nUSE_LOOCV = True        # recommended at this sample size; set False to use N_FOLDS instead\nN_FOLDS = 5\n\n# Priority order for picking one series per study when multiple are\n# available. Fluid-sensitive sequences (with or without fat suppression) are\n# what radiologists lean on most for effusion/tears/marrow edema, so this is\n# a literature-grounded default, not an arbitrary one. Falls through tiers\n# if an exact match isn't available for a given study, so we don't lose\n# subjects out of a pool of only 58.\nSERIES_PRIORITY = [\n    (\"Sagittal\", 1, 1),   # sagittal, fluid-sensitive, fat-suppressed (preferred)\n    (\"Sagittal\", 1, 0),   # sagittal, fluid-sensitive, no fat-sat\n    (\"Sagittal\", 0, None),\n    (\"Coronal\", 1, None),\n    (\"Axial\", 1, None),\n]\n\n\n# ==========================================================================\n# SECTION 1: Manifest loading (joins train.csv + train_series.csv)\n# ==========================================================================\n\ndef load_gold_manifest(target_finding: str) -> \"list[dict]\":\n    \"\"\"\n    Build the exam manifest for the radiologist-verified gold subset:\n    filters train.csv to rows with at least one non-null finding label,\n    then for each study picks one series folder per SERIES_PRIORITY.\n    \"\"\"\n    train_df = pd.read_csv(TRAIN_CSV)\n    gold = train_df.dropna(subset=LABEL_COLUMNS, how=\"all\").reset_index(drop=True)\n    print(f\"Gold-labeled studies (any finding annotated): {len(gold)}\")\n\n    n_pos = int(gold[target_finding].dropna().sum())\n    n_annotated = int(gold[target_finding].notna().sum())\n    print(f\"Target finding '{target_finding}' balance: {n_pos} positive / {n_annotated} annotated\")\n\n    series_df = pd.read_csv(TRAIN_SERIES_CSV)\n\n    records, missing_series = [], []\n    for _, row in gold.iterrows():\n        label = row[target_finding]\n        if pd.isna(label):\n            continue  # this particular finding wasn't annotated for this study\n\n        study_id = row[\"StudyInstanceUID\"]\n        study_series = series_df[series_df[\"StudyInstanceUID\"] == study_id]\n\n        chosen_series_id = None\n        for plane, fluid_sens, fat_sup in SERIES_PRIORITY:\n            candidates = study_series[\n                (study_series[\"Anatomical_Plane\"] == plane)\n                & (study_series[\"Fluid_Sensitive\"] == fluid_sens)\n            ]\n            if fat_sup is not None:\n                narrowed = candidates[candidates[\"Fat_Suppression\"] == fat_sup]\n                if len(narrowed) > 0:\n                    candidates = narrowed\n            if len(candidates) > 0:\n                chosen_series_id = candidates.iloc[0][\"SeriesInstanceUID\"]\n                break\n\n        if chosen_series_id is None:\n            missing_series.append(study_id)\n            continue\n\n        series_path = os.path.join(TRAIN_SERIES_ROOT, study_id, chosen_series_id)\n        records.append({\"exam_id\": study_id, \"path\": series_path, \"label\": int(label)})\n\n    if missing_series:\n        print(f\"Warning: {len(missing_series)} studies had no matching series \"\n              f\"under SERIES_PRIORITY and were skipped: {missing_series}\")\n\n    print(f\"Final manifest: {len(records)} studies with usable series + label\\n\")\n    return records\n\n\n# ==========================================================================\n# SECTION 2: DICOM loading and slice preprocessing\n# ==========================================================================\n\n@dataclass\nclass Series:\n    exam_id: str\n    volume: np.ndarray   # shape (n_slices, H, W), float32, raw intensities\n\n\ndef load_dicom_series(folder: str, exam_id: str) -> Series:\n    \"\"\"Load one DICOM series (all .dcm files directly inside `folder`) into a\n    sorted 3D volume, ordered by InstanceNumber where available.\"\"\"\n    if pydicom is None:\n        raise ImportError(\"pydicom is required to load DICOM files: pip install pydicom\")\n\n    files = [f for f in os.listdir(folder) if f.lower().endswith(\".dcm\")]\n    if not files:\n        raise FileNotFoundError(f\"No .dcm files found in {folder}\")\n\n    datasets = [pydicom.dcmread(os.path.join(folder, f)) for f in files]\n    datasets.sort(key=lambda ds: getattr(ds, \"InstanceNumber\", 0))\n\n    slices = [ds.pixel_array.astype(np.float32) for ds in datasets]\n    min_h = min(s.shape[0] for s in slices)\n    min_w = min(s.shape[1] for s in slices)\n    slices = [s[:min_h, :min_w] for s in slices]\n\n    return Series(exam_id=exam_id, volume=np.stack(slices, axis=0))\n\n\ndef select_key_slices(volume: np.ndarray, n_slices: int = 5, method: str = \"central_variance\") -> np.ndarray:\n    \"\"\"Reduce a full slice stack to the n_slices most informative slices.\"\"\"\n    n_total = volume.shape[0]\n    if n_total <= n_slices:\n        return volume\n\n    center = n_total // 2\n    if method == \"central\":\n        half = n_slices // 2\n        start = max(0, center - half)\n        end = min(n_total, start + n_slices)\n        return volume[start:end]\n\n    if method == \"central_variance\":\n        window = min(n_total, max(n_slices * 3, 15))\n        half_window = window // 2\n        start = max(0, center - half_window)\n        end = min(n_total, start + window)\n        candidates = volume[start:end]\n        variances = candidates.reshape(candidates.shape[0], -1).var(axis=1)\n        top_idx = np.argsort(variances)[-n_slices:]\n        top_idx.sort()\n        return candidates[top_idx]\n\n    raise ValueError(f\"Unknown method: {method}\")\n\n\ndef normalize_slice(img: np.ndarray, target_size: Optional[int] = 224) -> np.ndarray:\n    \"\"\"Min-max normalize to [0, 255] uint8 and resize to a square target_size.\"\"\"\n    img = img.astype(np.float32)\n    lo, hi = np.percentile(img, 0.5), np.percentile(img, 99.5)\n    img = np.clip(img, lo, hi)\n    img = (img - lo) / (hi - lo) if hi > lo else np.zeros_like(img)\n    img = (img * 255).astype(np.uint8)\n\n    if target_size is not None:\n        img = resize(img, (target_size, target_size), preserve_range=True, anti_aliasing=True).astype(np.uint8)\n\n    return img\n\n\n# ==========================================================================\n# SECTION 3: Feature extraction (handcrafted + deep)\n# ==========================================================================\n\ndef extract_hog(img: np.ndarray) -> np.ndarray:\n    # Coarser grid than a \"default\" HOG setup - at n=58 a ~6000-dim HOG\n    # vector would dwarf LBP/GLCM/CNN dims and overfit badly. (32,32) cells\n    # brings this down to roughly 1,300 dims, still texture-rich but sane\n    # relative to the sample size.\n    return hog(img, orientations=9, pixels_per_cell=(32, 32), cells_per_block=(2, 2),\n               block_norm=\"L2-Hys\", feature_vector=True)\n\n\ndef extract_lbp(img: np.ndarray, n_points: int = 24, radius: int = 3) -> np.ndarray:\n    lbp = local_binary_pattern(img, n_points, radius, method=\"uniform\")\n    n_bins = n_points + 2\n    hist, _ = np.histogram(lbp.ravel(), bins=n_bins, range=(0, n_bins), density=True)\n    return hist\n\n\ndef extract_glcm(img: np.ndarray) -> np.ndarray:\n    img_q = (img / 32).astype(np.uint8)  # 256 -> 8 gray levels\n    glcm = graycomatrix(img_q, distances=[1, 2], angles=[0, np.pi / 4, np.pi / 2, 3 * np.pi / 4],\n                         levels=8, symmetric=True, normed=True)\n    props = [\"contrast\", \"homogeneity\", \"energy\", \"correlation\", \"ASM\"]\n    return np.concatenate([graycoprops(glcm, p).ravel() for p in props])\n\n\ndef extract_handcrafted_features(img: np.ndarray) -> np.ndarray:\n    return np.concatenate([extract_hog(img), extract_lbp(img), extract_glcm(img)])\n\n\nclass DeepFeatureExtractor:\n    \"\"\"Frozen pretrained CNN backbone used as a fixed feature extractor.\"\"\"\n\n    def __init__(self, backbone_name: str = \"resnet50\", device: str = \"cpu\"):\n        import torch\n        import torchvision.models as models\n\n        self.device = device\n        if backbone_name == \"resnet50\":\n            weights = models.ResNet50_Weights.IMAGENET1K_V2\n            net = models.resnet50(weights=weights)\n            net.fc = torch.nn.Identity()\n            self.preprocess = weights.transforms()\n        else:\n            raise ValueError(f\"Unsupported backbone: {backbone_name}\")\n\n        net.eval()\n        self.net = net.to(device)\n\n    def extract(self, img_uint8: np.ndarray) -> np.ndarray:\n        import torch\n        from PIL import Image\n\n        img_rgb = np.stack([img_uint8] * 3, axis=-1) if img_uint8.ndim == 2 else img_uint8\n        tensor = self.preprocess(Image.fromarray(img_rgb)).unsqueeze(0).to(self.device)\n\n        with torch.no_grad():\n            feats = self.net(tensor)\n\n        return feats.squeeze(0).cpu().numpy()\n\n\ndef build_exam_feature_vector(\n    key_slices: np.ndarray,\n    deep_extractor: \"DeepFeatureExtractor | None\" = None,\n    use_handcrafted: bool = True,\n    use_deep: bool = True,\n) -> np.ndarray:\n    \"\"\"Mean-pool handcrafted and/or deep features across slices, then concatenate.\"\"\"\n    handcrafted_vecs, deep_vecs = [], []\n\n    for slice_img in key_slices:\n        if use_handcrafted:\n            handcrafted_vecs.append(extract_handcrafted_features(slice_img))\n        if use_deep:\n            if deep_extractor is None:\n                raise ValueError(\"use_deep=True requires a DeepFeatureExtractor instance\")\n            deep_vecs.append(deep_extractor.extract(slice_img))\n\n    parts = []\n    if use_handcrafted:\n        parts.append(np.mean(np.stack(handcrafted_vecs), axis=0))\n    if use_deep:\n        parts.append(np.mean(np.stack(deep_vecs), axis=0))\n\n    return np.concatenate(parts)\n\n\ndef build_feature_matrix(manifest: \"list[dict]\", feature_mode: str, n_key_slices: int = 5):\n    \"\"\"feature_mode: 'handcrafted' | 'deep' | 'fused'. Returns (X, y, kept_exam_ids).\"\"\"\n    use_handcrafted = feature_mode in (\"handcrafted\", \"fused\")\n    use_deep = feature_mode in (\"deep\", \"fused\")\n    deep_extractor = DeepFeatureExtractor(backbone_name=\"resnet50\") if use_deep else None\n\n    X, y, kept_ids = [], [], []\n    for record in manifest:\n        try:\n            series = load_dicom_series(record[\"path\"], record[\"exam_id\"])\n        except Exception as e:\n            print(f\"  Skipping {record['exam_id']}: {e}\")\n            continue\n\n        key_slices = select_key_slices(series.volume, n_slices=n_key_slices)\n        key_slices = np.stack([normalize_slice(s) for s in key_slices])\n\n        vec = build_exam_feature_vector(key_slices, deep_extractor=deep_extractor,\n                                         use_handcrafted=use_handcrafted, use_deep=use_deep)\n        X.append(vec)\n        y.append(record[\"label\"])\n        kept_ids.append(record[\"exam_id\"])\n\n    return np.array(X), np.array(y), kept_ids\n\n\n# ==========================================================================\n# SECTION 4: Training / evaluation (LOOCV or K-fold, CatBoost or AdaBoost, SHAP)\n# ==========================================================================\n\n@dataclass\nclass Metrics:\n    auc: float\n    f1: float\n    accuracy: float\n    sensitivity: float\n    specificity: float\n\n\ndef compute_metrics(y_true: np.ndarray, y_pred: np.ndarray, y_prob: np.ndarray) -> Metrics:\n    auc = roc_auc_score(y_true, y_prob)\n    f1 = f1_score(y_true, y_pred)\n    acc = accuracy_score(y_true, y_pred)\n    tn, fp, fn, tp = confusion_matrix(y_true, y_pred).ravel()\n    sensitivity = tp / (tp + fn) if (tp + fn) > 0 else 0.0\n    specificity = tn / (tn + fp) if (tn + fp) > 0 else 0.0\n    return Metrics(auc=auc, f1=f1, accuracy=acc, sensitivity=sensitivity, specificity=specificity)\n\n\ndef fit_classifier(classifier: str, X_train: np.ndarray, y_train: np.ndarray):\n    if classifier == \"catboost\":\n        from catboost import CatBoostClassifier\n        # Shallower/fewer iterations than a \"default\" CatBoost setup -\n        # deliberately conservative given only ~50-57 training samples per fold.\n        model = CatBoostClassifier(iterations=300, learning_rate=0.05, depth=4,\n                                    eval_metric=\"AUC\", auto_class_weights=\"Balanced\",\n                                    random_seed=42, verbose=False)\n        model.fit(X_train, y_train)\n        return model\n    elif classifier == \"adaboost\":\n        from sklearn.ensemble import AdaBoostClassifier\n        model = AdaBoostClassifier(n_estimators=200, random_state=42)\n        model.fit(X_train, y_train)\n        return model\n    raise ValueError(f\"Unknown classifier: {classifier}\")\n\n\ndef run_shap_summary(model, X_reduced: np.ndarray, tag: str):\n    \"\"\"SHAP on post-PCA components (fit on all data, separate from the LOOCV\n    evaluation above - LOOCV gives you the honest performance estimate, this\n    gives you the interpretability plot). For per-feature-type\n    interpretability (raw HOG/LBP/GLCM/CNN dims), rerun handcrafted-only\n    without PCA - CatBoost handles high-dimensional input natively.\"\"\"\n    try:\n        import shap\n        import matplotlib.pyplot as plt\n        explainer = shap.TreeExplainer(model)\n        shap_values = explainer.shap_values(X_reduced)\n        shap.summary_plot(shap_values, X_reduced, show=False)\n        plt.savefig(f\"shap_summary_{tag}.png\", bbox_inches=\"tight\", dpi=150)\n        plt.close()\n        print(f\"Saved SHAP summary to shap_summary_{tag}.png\")\n    except ImportError:\n        print(\"shap not installed - skipping interpretability plot (pip install shap)\")\n\n\ndef train_and_evaluate(\n    X: np.ndarray, y: np.ndarray, classifier: str = \"catboost\",\n    use_loocv: bool = True, n_folds: int = 5, pca_components: int = 20,\n    run_shap: bool = True, tag: str = \"\",\n):\n    \"\"\"\n    LOOCV (default, recommended for n<~100): trains n_samples models, each\n    leaving one exam out, and aggregates all out-of-fold predictions into a\n    single set of metrics at the end - this is the standard evaluation\n    protocol at small sample sizes, since per-fold metrics with 1 validation\n    sample are meaningless.\n\n    K-fold (use_loocv=False): still aggregates out-of-fold predictions the\n    same way, plus prints per-fold AUC when both classes are present in a\n    fold's validation split.\n    \"\"\"\n    splitter = LeaveOneOut() if use_loocv else StratifiedKFold(n_splits=n_folds, shuffle=True, random_state=42)\n\n    y_true_all, y_prob_all = [], []\n    fit_times = []\n\n    for fold_idx, (train_idx, val_idx) in enumerate(splitter.split(X, y)):\n        X_train, X_val = X[train_idx], X[val_idx]\n        y_train, y_val = y[train_idx], y[val_idx]\n\n        scaler = StandardScaler()\n        X_train_scaled = scaler.fit_transform(X_train)\n        X_val_scaled = scaler.transform(X_val)\n\n        n_components = min(pca_components, X_train_scaled.shape[0] - 1, X_train_scaled.shape[1])\n        pca = PCA(n_components=n_components, random_state=42)\n        X_train_reduced = pca.fit_transform(X_train_scaled)\n        X_val_reduced = pca.transform(X_val_scaled)\n\n        start = time.time()\n        model = fit_classifier(classifier, X_train_reduced, y_train)\n        fit_times.append(time.time() - start)\n\n        y_prob = model.predict_proba(X_val_reduced)[:, 1]\n        y_true_all.extend(y_val.tolist())\n        y_prob_all.extend(y_prob.tolist())\n\n        if not use_loocv and len(set(y_val)) > 1:\n            y_pred = (y_prob >= 0.5).astype(int)\n            fold_m = compute_metrics(y_val, y_pred, y_prob)\n            print(f\"  [{tag} fold {fold_idx}] AUC={fold_m.auc:.3f} F1={fold_m.f1:.3f} \"\n                  f\"sens={fold_m.sensitivity:.3f} spec={fold_m.specificity:.3f}\")\n\n    y_true_all = np.array(y_true_all)\n    y_prob_all = np.array(y_prob_all)\n    y_pred_all = (y_prob_all >= 0.5).astype(int)\n    overall = compute_metrics(y_true_all, y_pred_all, y_prob_all)\n\n    protocol = \"LOOCV\" if use_loocv else f\"{n_folds}-fold\"\n    print(f\"[{tag}] Aggregated {protocol} results over {len(y_true_all)} out-of-fold predictions:\")\n    print(f\"  AUC={overall.auc:.3f}  F1={overall.f1:.3f}  Acc={overall.accuracy:.3f}  \"\n          f\"Sens={overall.sensitivity:.3f}  Spec={overall.specificity:.3f}\")\n    print(f\"  Mean per-fit time: {np.mean(fit_times):.3f}s  |  Total fits: {len(fit_times)}\\n\")\n\n    if run_shap and classifier == \"catboost\":\n        scaler_full = StandardScaler()\n        X_scaled_full = scaler_full.fit_transform(X)\n        n_components_full = min(pca_components, X_scaled_full.shape[0] - 1, X_scaled_full.shape[1])\n        pca_full = PCA(n_components=n_components_full, random_state=42)\n        X_reduced_full = pca_full.fit_transform(X_scaled_full)\n        full_model = fit_classifier(\"catboost\", X_reduced_full, y)\n        run_shap_summary(full_model, X_reduced_full, tag)\n\n    return overall, fit_times\n\n\n# ==========================================================================\n# MAIN\n# ==========================================================================\n\nif __name__ == \"__main__\":\n    manifest = load_gold_manifest(TARGET_FINDING)\n\n    summary = {}\n    for feature_mode in [\"handcrafted\", \"deep\", \"fused\"]:\n        print(f\"=== Feature mode: {feature_mode} ===\")\n        X, y, kept_ids = build_feature_matrix(manifest, feature_mode=feature_mode, n_key_slices=N_KEY_SLICES)\n        print(f\"  Feature matrix: {X.shape}, kept {len(kept_ids)}/{len(manifest)} exams\")\n\n        overall, fit_times = train_and_evaluate(\n            X, y, classifier=\"catboost\", use_loocv=USE_LOOCV, n_folds=N_FOLDS,\n            pca_components=PCA_COMPONENTS, run_shap=True, tag=feature_mode,\n        )\n        summary[feature_mode] = {\n            \"auc\": overall.auc, \"f1\": overall.f1, \"sensitivity\": overall.sensitivity,\n            \"specificity\": overall.specificity, \"mean_fit_time\": float(np.mean(fit_times)),\n        }\n\n    print(f\"=== Summary across feature modes (CatBoost, target={TARGET_FINDING}) ===\")\n    for mode, stats in summary.items():\n        print(f\"{mode:12s}  AUC={stats['auc']:.3f}  F1={stats['f1']:.3f}  \"\n              f\"Sens={stats['sensitivity']:.3f}  Spec={stats['specificity']:.3f}  \"\n              f\"fit_time={stats['mean_fit_time']:.3f}s\")\n\n    # Optional: AdaBoost on the fused features, as a second boosting variant\n    # for robustness of the finding.\n    # X_fused, y_fused, _ = build_feature_matrix(manifest, feature_mode=\"fused\", n_key_slices=N_KEY_SLICES)\n    # train_and_evaluate(X_fused, y_fused, classifier=\"adaboost\", use_loocv=USE_LOOCV,\n    #                     pca_components=PCA_COMPONENTS, run_shap=False, tag=\"fused_adaboost\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-12T07:56:18.35956Z","iopub.execute_input":"2026-09-12T07:56:18.359892Z","iopub.status.idle":"2026-09-12T07:59:48.357018Z","shell.execute_reply.started":"2026-09-12T07:56:18.359866Z","shell.execute_reply":"2026-09-12T07:59:48.356121Z"}},"outputs":[{"name":"stdout","text":"Gold-labeled studies (any finding annotated): 58\nTarget finding 'ACL' balance: 24 positive / 58 annotated\nFinal manifest: 58 studies with usable series + label\n\n=== Feature mode: handcrafted ===\n  Feature matrix: (58, 1362), kept 58/58 exams\n[handcrafted] Aggregated LOOCV results over 58 out-of-fold predictions:\n  AUC=0.444  F1=0.426  Acc=0.534  Sens=0.417  Spec=0.618\n  Mean per-fit time: 0.275s  |  Total fits: 58\n\nSaved SHAP summary to shap_summary_handcrafted.png\n=== Feature mode: deep ===\nDownloading: \"https://download.pytorch.org/models/resnet50-11ad3fa6.pth\" to /root/.cache/torch/hub/checkpoints/resnet50-11ad3fa6.pth\n","output_type":"stream"},{"name":"stderr","text":"100%|██████████| 97.8M/97.8M [00:00<00:00, 236MB/s] \n","output_type":"stream"},{"name":"stdout","text":"  Feature matrix: (58, 2048), kept 58/58 exams\n[deep] Aggregated LOOCV results over 58 out-of-fold predictions:\n  AUC=0.673  F1=0.462  Acc=0.638  Sens=0.375  Spec=0.824\n  Mean per-fit time: 0.251s  |  Total fits: 58\n\nSaved SHAP summary to shap_summary_deep.png\n=== Feature mode: fused ===\n  Feature matrix: (58, 3410), kept 58/58 exams\n[fused] Aggregated LOOCV results over 58 out-of-fold predictions:\n  AUC=0.556  F1=0.421  Acc=0.621  Sens=0.333  Spec=0.824\n  Mean per-fit time: 0.247s  |  Total fits: 58\n\nSaved SHAP summary to shap_summary_fused.png\n=== Summary across feature modes (CatBoost, target=ACL) ===\nhandcrafted   AUC=0.444  F1=0.426  Sens=0.417  Spec=0.618  fit_time=0.275s\ndeep          AUC=0.673  F1=0.462  Sens=0.375  Spec=0.824  fit_time=0.251s\nfused         AUC=0.556  F1=0.421  Sens=0.333  Spec=0.824  fit_time=0.247s\n","output_type":"stream"}],"execution_count":9},{"cell_type":"code","source":"\"\"\"\nknee_mri_pipeline_allinone.py  (v2 - matched to real competition data)\n\nHandcrafted (HOG/LBP/GLCM) + deep CNN feature fusion, classified with\nCatBoost/AdaBoost, for the RSNA Knee Abnormality Detection competition.\n\nWHAT CHANGED FROM v1\n---------------------\nReal data exploration showed:\n  - Only 58 of 4,407 studies have radiologist-verified 0/1 labels across 12\n    findings (ACL, MCL, Medial/Lateral Meniscus, Medial/Lateral OA, PF OA,\n    Effusion, Synovitis, Baker's, Contusion, Fracture). The rest only have\n    free-text multilingual reports.\n  - All 58 gold-labeled studies are positive for >=1 finding, so a pooled\n    \"any abnormality\" target is degenerate (no negative class). This\n    pipeline instead targets ONE finding at a time (TARGET_FINDING below).\n  - Series aren't tagged by sequence name - they're tagged by Anatomical_Plane\n    plus two binary flags, Fluid_Sensitive and Fat_Suppression. Series files\n    live at train_series/{StudyInstanceUID}/{SeriesInstanceUID}/*.dcm.\n  - n=58 is small-sample territory: this version uses leave-one-out\n    cross-validation with aggregated out-of-fold metrics (standard practice\n    at this sample size) instead of 5-fold CV, and coarsens the HOG grid /\n    caps PCA components to reduce overfitting risk.\n\nHOW TO USE ON KAGGLE\n---------------------\n1. Add the competition data to your notebook.\n2. Optionally change TARGET_FINDING below (best-balanced options at n=58:\n   'ACL' 24/58, 'Medial Meniscus' 26/58, 'Effusion' 35/58, 'Synovitis' 27/58).\n3. Run the whole file. It prints the class balance for your chosen finding,\n   builds features for handcrafted/deep/fused modes, runs LOOCV with\n   CatBoost for each, and prints an aggregated-metrics summary table.\n\nInstall anything missing with: !pip install -q pydicom catboost shap\n\"\"\"\n\nfrom __future__ import annotations\n\nimport os\nimport time\nfrom dataclasses import dataclass\nfrom typing import Optional\n\nimport numpy as np\nimport pandas as pd\nfrom skimage.feature import hog, local_binary_pattern, graycomatrix, graycoprops\nfrom skimage.transform import resize\nfrom sklearn.decomposition import PCA\nfrom sklearn.model_selection import StratifiedKFold, LeaveOneOut\nfrom sklearn.metrics import roc_auc_score, f1_score, accuracy_score, confusion_matrix\nfrom sklearn.preprocessing import StandardScaler\n\ntry:\n    import pydicom\nexcept ImportError:\n    pydicom = None\n\n\n# ==========================================================================\n# CONFIG\n# ==========================================================================\n\nCOMP_ROOT = \"/kaggle/input/competitions/rsna-knee-abnormality-detection\"\nTRAIN_CSV = os.path.join(COMP_ROOT, \"train.csv\")\nTRAIN_SERIES_CSV = os.path.join(COMP_ROOT, \"train_series.csv\")\nTRAIN_SERIES_ROOT = os.path.join(COMP_ROOT, \"train_series\")\n\nLABEL_COLUMNS = [\"ACL\", \"MCL\", \"Medial Meniscus\", \"Lateral Meniscus\", \"Medial OA\",\n                  \"Lateral OA\", \"PF OA\", \"Effusion\", \"Synovitis\", \"Baker's\",\n                  \"Contusion\", \"Fracture\"]\n\n# Which finding to predict. At n=58 gold-labeled studies, the best-balanced\n# options are: ACL (24/58), Medial Meniscus (26/58), Effusion (35/58),\n# Synovitis (27/58). Starting with ACL for direct comparability to MRNet's\n# own ACL-tear benchmark numbers.\nTARGET_FINDING = \"ACL\"\n\nN_KEY_SLICES = 5\nPCA_COMPONENTS = 20     # kept small given n=58 - see note in train_and_evaluate\nUSE_LOOCV = True        # recommended at this sample size; set False to use N_FOLDS instead\nN_FOLDS = 5\n\n# Priority order for picking one series per study when multiple are\n# available. Fluid-sensitive sequences (with or without fat suppression) are\n# what radiologists lean on most for effusion/tears/marrow edema, so this is\n# a literature-grounded default, not an arbitrary one. Falls through tiers\n# if an exact match isn't available for a given study, so we don't lose\n# subjects out of a pool of only 58.\nSERIES_PRIORITY = [\n    (\"Sagittal\", 1, 1),   # sagittal, fluid-sensitive, fat-suppressed (preferred)\n    (\"Sagittal\", 1, 0),   # sagittal, fluid-sensitive, no fat-sat\n    (\"Sagittal\", 0, None),\n    (\"Coronal\", 1, None),\n    (\"Axial\", 1, None),\n]\n\n\n# ==========================================================================\n# SECTION 1: Manifest loading (joins train.csv + train_series.csv)\n# ==========================================================================\n\ndef load_gold_manifest(target_finding: str) -> \"list[dict]\":\n    \"\"\"\n    Build the exam manifest for the radiologist-verified gold subset:\n    filters train.csv to rows with at least one non-null finding label,\n    then for each study picks one series folder per SERIES_PRIORITY.\n    \"\"\"\n    train_df = pd.read_csv(TRAIN_CSV)\n    gold = train_df.dropna(subset=LABEL_COLUMNS, how=\"all\").reset_index(drop=True)\n    print(f\"Gold-labeled studies (any finding annotated): {len(gold)}\")\n\n    n_pos = int(gold[target_finding].dropna().sum())\n    n_annotated = int(gold[target_finding].notna().sum())\n    print(f\"Target finding '{target_finding}' balance: {n_pos} positive / {n_annotated} annotated\")\n\n    series_df = pd.read_csv(TRAIN_SERIES_CSV)\n\n    records, missing_series = [], []\n    for _, row in gold.iterrows():\n        label = row[target_finding]\n        if pd.isna(label):\n            continue  # this particular finding wasn't annotated for this study\n\n        study_id = row[\"StudyInstanceUID\"]\n        study_series = series_df[series_df[\"StudyInstanceUID\"] == study_id]\n\n        chosen_series_id = None\n        for plane, fluid_sens, fat_sup in SERIES_PRIORITY:\n            candidates = study_series[\n                (study_series[\"Anatomical_Plane\"] == plane)\n                & (study_series[\"Fluid_Sensitive\"] == fluid_sens)\n            ]\n            if fat_sup is not None:\n                narrowed = candidates[candidates[\"Fat_Suppression\"] == fat_sup]\n                if len(narrowed) > 0:\n                    candidates = narrowed\n            if len(candidates) > 0:\n                chosen_series_id = candidates.iloc[0][\"SeriesInstanceUID\"]\n                break\n\n        if chosen_series_id is None:\n            missing_series.append(study_id)\n            continue\n\n        series_path = os.path.join(TRAIN_SERIES_ROOT, study_id, chosen_series_id)\n        records.append({\"exam_id\": study_id, \"path\": series_path, \"label\": int(label)})\n\n    if missing_series:\n        print(f\"Warning: {len(missing_series)} studies had no matching series \"\n              f\"under SERIES_PRIORITY and were skipped: {missing_series}\")\n\n    print(f\"Final manifest: {len(records)} studies with usable series + label\\n\")\n    return records\n\n\n# ==========================================================================\n# SECTION 2: DICOM loading and slice preprocessing\n# ==========================================================================\n\n@dataclass\nclass Series:\n    exam_id: str\n    volume: np.ndarray   # shape (n_slices, H, W), float32, raw intensities\n\n\ndef load_dicom_series(folder: str, exam_id: str) -> Series:\n    \"\"\"Load one DICOM series (all .dcm files directly inside `folder`) into a\n    sorted 3D volume, ordered by InstanceNumber where available.\"\"\"\n    if pydicom is None:\n        raise ImportError(\"pydicom is required to load DICOM files: pip install pydicom\")\n\n    files = [f for f in os.listdir(folder) if f.lower().endswith(\".dcm\")]\n    if not files:\n        raise FileNotFoundError(f\"No .dcm files found in {folder}\")\n\n    datasets = [pydicom.dcmread(os.path.join(folder, f)) for f in files]\n    datasets.sort(key=lambda ds: getattr(ds, \"InstanceNumber\", 0))\n\n    slices = [ds.pixel_array.astype(np.float32) for ds in datasets]\n    min_h = min(s.shape[0] for s in slices)\n    min_w = min(s.shape[1] for s in slices)\n    slices = [s[:min_h, :min_w] for s in slices]\n\n    return Series(exam_id=exam_id, volume=np.stack(slices, axis=0))\n\n\ndef select_key_slices(volume: np.ndarray, n_slices: int = 5, method: str = \"central_variance\") -> np.ndarray:\n    \"\"\"Reduce a full slice stack to the n_slices most informative slices.\"\"\"\n    n_total = volume.shape[0]\n    if n_total <= n_slices:\n        return volume\n\n    center = n_total // 2\n    if method == \"central\":\n        half = n_slices // 2\n        start = max(0, center - half)\n        end = min(n_total, start + n_slices)\n        return volume[start:end]\n\n    if method == \"central_variance\":\n        window = min(n_total, max(n_slices * 3, 15))\n        half_window = window // 2\n        start = max(0, center - half_window)\n        end = min(n_total, start + window)\n        candidates = volume[start:end]\n        variances = candidates.reshape(candidates.shape[0], -1).var(axis=1)\n        top_idx = np.argsort(variances)[-n_slices:]\n        top_idx.sort()\n        return candidates[top_idx]\n\n    raise ValueError(f\"Unknown method: {method}\")\n\n\ndef normalize_slice(img: np.ndarray, target_size: Optional[int] = 224) -> np.ndarray:\n    \"\"\"Min-max normalize to [0, 255] uint8 and resize to a square target_size.\"\"\"\n    img = img.astype(np.float32)\n    lo, hi = np.percentile(img, 0.5), np.percentile(img, 99.5)\n    img = np.clip(img, lo, hi)\n    img = (img - lo) / (hi - lo) if hi > lo else np.zeros_like(img)\n    img = (img * 255).astype(np.uint8)\n\n    if target_size is not None:\n        img = resize(img, (target_size, target_size), preserve_range=True, anti_aliasing=True).astype(np.uint8)\n\n    return img\n\n\n# ==========================================================================\n# SECTION 3: Feature extraction (handcrafted + deep)\n# ==========================================================================\n\ndef extract_hog(img: np.ndarray) -> np.ndarray:\n    # Coarser grid than a \"default\" HOG setup - at n=58 a ~6000-dim HOG\n    # vector would dwarf LBP/GLCM/CNN dims and overfit badly. (32,32) cells\n    # brings this down to roughly 1,300 dims, still texture-rich but sane\n    # relative to the sample size.\n    return hog(img, orientations=9, pixels_per_cell=(32, 32), cells_per_block=(2, 2),\n               block_norm=\"L2-Hys\", feature_vector=True)\n\n\ndef extract_lbp(img: np.ndarray, n_points: int = 24, radius: int = 3) -> np.ndarray:\n    lbp = local_binary_pattern(img, n_points, radius, method=\"uniform\")\n    n_bins = n_points + 2\n    hist, _ = np.histogram(lbp.ravel(), bins=n_bins, range=(0, n_bins), density=True)\n    return hist\n\n\ndef extract_glcm(img: np.ndarray) -> np.ndarray:\n    img_q = (img / 32).astype(np.uint8)  # 256 -> 8 gray levels\n    glcm = graycomatrix(img_q, distances=[1, 2], angles=[0, np.pi / 4, np.pi / 2, 3 * np.pi / 4],\n                         levels=8, symmetric=True, normed=True)\n    props = [\"contrast\", \"homogeneity\", \"energy\", \"correlation\", \"ASM\"]\n    return np.concatenate([graycoprops(glcm, p).ravel() for p in props])\n\n\ndef extract_handcrafted_features(img: np.ndarray) -> np.ndarray:\n    return np.concatenate([extract_hog(img), extract_lbp(img), extract_glcm(img)])\n\n\nclass DeepFeatureExtractor:\n    \"\"\"Frozen pretrained CNN backbone used as a fixed feature extractor.\"\"\"\n\n    def __init__(self, backbone_name: str = \"resnet50\", device: str = \"cpu\"):\n        import torch\n        import torchvision.models as models\n\n        self.device = device\n        if backbone_name == \"resnet50\":\n            weights = models.ResNet50_Weights.IMAGENET1K_V2\n            net = models.resnet50(weights=weights)\n            net.fc = torch.nn.Identity()\n            self.preprocess = weights.transforms()\n        else:\n            raise ValueError(f\"Unsupported backbone: {backbone_name}\")\n\n        net.eval()\n        self.net = net.to(device)\n\n    def extract(self, img_uint8: np.ndarray) -> np.ndarray:\n        import torch\n        from PIL import Image\n\n        img_rgb = np.stack([img_uint8] * 3, axis=-1) if img_uint8.ndim == 2 else img_uint8\n        tensor = self.preprocess(Image.fromarray(img_rgb)).unsqueeze(0).to(self.device)\n\n        with torch.no_grad():\n            feats = self.net(tensor)\n\n        return feats.squeeze(0).cpu().numpy()\n\n\ndef build_exam_feature_vector(\n    key_slices: np.ndarray,\n    deep_extractor: \"DeepFeatureExtractor | None\" = None,\n    use_handcrafted: bool = True,\n    use_deep: bool = True,\n) -> np.ndarray:\n    \"\"\"Mean-pool handcrafted and/or deep features across slices, then concatenate.\"\"\"\n    handcrafted_vecs, deep_vecs = [], []\n\n    for slice_img in key_slices:\n        if use_handcrafted:\n            handcrafted_vecs.append(extract_handcrafted_features(slice_img))\n        if use_deep:\n            if deep_extractor is None:\n                raise ValueError(\"use_deep=True requires a DeepFeatureExtractor instance\")\n            deep_vecs.append(deep_extractor.extract(slice_img))\n\n    parts = []\n    if use_handcrafted:\n        parts.append(np.mean(np.stack(handcrafted_vecs), axis=0))\n    if use_deep:\n        parts.append(np.mean(np.stack(deep_vecs), axis=0))\n\n    return np.concatenate(parts)\n\n\ndef build_feature_matrix(manifest: \"list[dict]\", feature_mode: str, n_key_slices: int = 5):\n    \"\"\"feature_mode: 'handcrafted' | 'deep' | 'fused'. Returns (X, y, kept_exam_ids).\"\"\"\n    use_handcrafted = feature_mode in (\"handcrafted\", \"fused\")\n    use_deep = feature_mode in (\"deep\", \"fused\")\n    deep_extractor = DeepFeatureExtractor(backbone_name=\"resnet50\") if use_deep else None\n\n    X, y, kept_ids = [], [], []\n    for record in manifest:\n        try:\n            series = load_dicom_series(record[\"path\"], record[\"exam_id\"])\n        except Exception as e:\n            print(f\"  Skipping {record['exam_id']}: {e}\")\n            continue\n\n        key_slices = select_key_slices(series.volume, n_slices=n_key_slices)\n        key_slices = np.stack([normalize_slice(s) for s in key_slices])\n\n        vec = build_exam_feature_vector(key_slices, deep_extractor=deep_extractor,\n                                         use_handcrafted=use_handcrafted, use_deep=use_deep)\n        X.append(vec)\n        y.append(record[\"label\"])\n        kept_ids.append(record[\"exam_id\"])\n\n    return np.array(X), np.array(y), kept_ids\n\n\n# ==========================================================================\n# SECTION 4: Training / evaluation (LOOCV or K-fold, CatBoost or AdaBoost, SHAP)\n# ==========================================================================\n\n@dataclass\nclass Metrics:\n    auc: float\n    f1: float\n    accuracy: float\n    sensitivity: float\n    specificity: float\n\n\ndef compute_metrics(y_true: np.ndarray, y_pred: np.ndarray, y_prob: np.ndarray) -> Metrics:\n    auc = roc_auc_score(y_true, y_prob)\n    f1 = f1_score(y_true, y_pred)\n    acc = accuracy_score(y_true, y_pred)\n    tn, fp, fn, tp = confusion_matrix(y_true, y_pred).ravel()\n    sensitivity = tp / (tp + fn) if (tp + fn) > 0 else 0.0\n    specificity = tn / (tn + fp) if (tn + fp) > 0 else 0.0\n    return Metrics(auc=auc, f1=f1, accuracy=acc, sensitivity=sensitivity, specificity=specificity)\n\n\ndef fit_classifier(classifier: str, X_train: np.ndarray, y_train: np.ndarray):\n    if classifier == \"catboost\":\n        from catboost import CatBoostClassifier\n        # Shallower/fewer iterations than a \"default\" CatBoost setup -\n        # deliberately conservative given only ~50-57 training samples per fold.\n        model = CatBoostClassifier(iterations=300, learning_rate=0.05, depth=4,\n                                    eval_metric=\"AUC\", auto_class_weights=\"Balanced\",\n                                    random_seed=42, verbose=False)\n        model.fit(X_train, y_train)\n        return model\n    elif classifier == \"adaboost\":\n        from sklearn.ensemble import AdaBoostClassifier\n        model = AdaBoostClassifier(n_estimators=200, random_state=42)\n        model.fit(X_train, y_train)\n        return model\n    raise ValueError(f\"Unknown classifier: {classifier}\")\n\n\ndef run_shap_summary(model, X_reduced: np.ndarray, tag: str):\n    \"\"\"SHAP on post-PCA components (fit on all data, separate from the LOOCV\n    evaluation above - LOOCV gives you the honest performance estimate, this\n    gives you the interpretability plot). For per-feature-type\n    interpretability (raw HOG/LBP/GLCM/CNN dims), rerun handcrafted-only\n    without PCA - CatBoost handles high-dimensional input natively.\"\"\"\n    try:\n        import shap\n        import matplotlib.pyplot as plt\n        explainer = shap.TreeExplainer(model)\n        shap_values = explainer.shap_values(X_reduced)\n        shap.summary_plot(shap_values, X_reduced, show=False)\n        plt.savefig(f\"shap_summary_{tag}.png\", bbox_inches=\"tight\", dpi=150)\n        plt.close()\n        print(f\"Saved SHAP summary to shap_summary_{tag}.png\")\n    except ImportError:\n        print(\"shap not installed - skipping interpretability plot (pip install shap)\")\n\n\ndef reduce_features(X_train_scaled: np.ndarray, X_val_scaled: np.ndarray, pca_components: \"int | None\"):\n    \"\"\"\n    pca_components=None skips PCA entirely and returns the scaled features\n    as-is. At n=57 training samples per LOOCV fold, PCA on a 1000+\n    dimensional handcrafted vector is a very unstable fit (more dimensions\n    than samples) - CatBoost/AdaBoost can both take high-dimensional input\n    directly, so skipping PCA is a legitimate and often better default for\n    this sample size, not just a fallback.\n    \"\"\"\n    if pca_components is None:\n        return X_train_scaled, X_val_scaled\n\n    n_components = min(pca_components, X_train_scaled.shape[0] - 1, X_train_scaled.shape[1])\n    pca = PCA(n_components=n_components, random_state=42)\n    return pca.fit_transform(X_train_scaled), pca.transform(X_val_scaled)\n\n\ndef bootstrap_auc_ci(y_true: np.ndarray, y_prob: np.ndarray, n_boot: int = 2000, seed: int = 42) -> \"tuple[float, float]\":\n    \"\"\"95% CI on AUC via case resampling - important at n=58, where a couple\n    of flipped predictions can move the point estimate a lot. Skips\n    resamples that happen to contain only one class (AUC undefined).\"\"\"\n    rng = np.random.default_rng(seed)\n    n = len(y_true)\n    boot_aucs = []\n    for _ in range(n_boot):\n        idx = rng.integers(0, n, size=n)\n        y_t, y_p = y_true[idx], y_prob[idx]\n        if len(set(y_t)) < 2:\n            continue\n        boot_aucs.append(roc_auc_score(y_t, y_p))\n    return float(np.percentile(boot_aucs, 2.5)), float(np.percentile(boot_aucs, 97.5))\n\n\ndef train_and_evaluate(\n    X: np.ndarray, y: np.ndarray, classifier: str = \"catboost\",\n    use_loocv: bool = True, n_folds: int = 5, pca_components: \"int | None\" = 20,\n    run_shap: bool = True, tag: str = \"\",\n):\n    \"\"\"\n    LOOCV (default, recommended for n<~100): trains n_samples models, each\n    leaving one exam out, and aggregates all out-of-fold predictions into a\n    single set of metrics at the end - this is the standard evaluation\n    protocol at small sample sizes, since per-fold metrics with 1 validation\n    sample are meaningless.\n\n    K-fold (use_loocv=False): still aggregates out-of-fold predictions the\n    same way, plus prints per-fold AUC when both classes are present in a\n    fold's validation split.\n\n    pca_components=None skips PCA (see reduce_features docstring) - worth\n    trying alongside a PCA'd run, since which one wins is itself a finding.\n    \"\"\"\n    splitter = LeaveOneOut() if use_loocv else StratifiedKFold(n_splits=n_folds, shuffle=True, random_state=42)\n\n    y_true_all, y_prob_all = [], []\n    fit_times = []\n\n    for fold_idx, (train_idx, val_idx) in enumerate(splitter.split(X, y)):\n        X_train, X_val = X[train_idx], X[val_idx]\n        y_train, y_val = y[train_idx], y[val_idx]\n\n        scaler = StandardScaler()\n        X_train_scaled = scaler.fit_transform(X_train)\n        X_val_scaled = scaler.transform(X_val)\n\n        X_train_reduced, X_val_reduced = reduce_features(X_train_scaled, X_val_scaled, pca_components)\n\n        start = time.time()\n        model = fit_classifier(classifier, X_train_reduced, y_train)\n        fit_times.append(time.time() - start)\n\n        y_prob = model.predict_proba(X_val_reduced)[:, 1]\n        y_true_all.extend(y_val.tolist())\n        y_prob_all.extend(y_prob.tolist())\n\n        if not use_loocv and len(set(y_val)) > 1:\n            y_pred = (y_prob >= 0.5).astype(int)\n            fold_m = compute_metrics(y_val, y_pred, y_prob)\n            print(f\"  [{tag} fold {fold_idx}] AUC={fold_m.auc:.3f} F1={fold_m.f1:.3f} \"\n                  f\"sens={fold_m.sensitivity:.3f} spec={fold_m.specificity:.3f}\")\n\n    y_true_all = np.array(y_true_all)\n    y_prob_all = np.array(y_prob_all)\n    y_pred_all = (y_prob_all >= 0.5).astype(int)\n    overall = compute_metrics(y_true_all, y_pred_all, y_prob_all)\n\n    ci_lo, ci_hi = bootstrap_auc_ci(y_true_all, y_prob_all)\n\n    protocol = \"LOOCV\" if use_loocv else f\"{n_folds}-fold\"\n    pca_desc = f\"PCA={pca_components}\" if pca_components is not None else \"no PCA\"\n    print(f\"[{tag}] Aggregated {protocol} results over {len(y_true_all)} out-of-fold predictions ({pca_desc}):\")\n    print(f\"  AUC={overall.auc:.3f} (95% CI [{ci_lo:.3f}, {ci_hi:.3f}])  F1={overall.f1:.3f}  \"\n          f\"Acc={overall.accuracy:.3f}  Sens={overall.sensitivity:.3f}  Spec={overall.specificity:.3f}\")\n    print(f\"  Mean per-fit time: {np.mean(fit_times):.3f}s  |  Total fits: {len(fit_times)}\\n\")\n\n    if run_shap and classifier == \"catboost\":\n        scaler_full = StandardScaler()\n        X_scaled_full = scaler_full.fit_transform(X)\n        X_reduced_full, _ = reduce_features(X_scaled_full, X_scaled_full, pca_components)\n        full_model = fit_classifier(\"catboost\", X_reduced_full, y)\n        run_shap_summary(full_model, X_reduced_full, tag)\n\n    return overall, fit_times, (ci_lo, ci_hi)\n\n\ndef train_and_evaluate_late_fusion(\n    X_handcrafted: np.ndarray, X_deep: np.ndarray, y: np.ndarray,\n    classifier: str = \"catboost\", use_loocv: bool = True, n_folds: int = 5,\n    pca_components: \"int | None\" = None, tag: str = \"late_fusion\",\n):\n    \"\"\"\n    Alternative to concatenate-then-PCA fusion: train a separate model per\n    modality within each fold, then average their predicted probabilities.\n    This avoids the failure mode where one modality's noise dilutes the\n    other's signal inside a shared PCA - each model gets to extract its own\n    modality's signal independently, and only the final probabilities mix.\n\n    X_handcrafted and X_deep must have rows aligned to the same exams/labels.\n    \"\"\"\n    splitter = LeaveOneOut() if use_loocv else StratifiedKFold(n_splits=n_folds, shuffle=True, random_state=42)\n\n    y_true_all, y_prob_all = [], []\n\n    for train_idx, val_idx in splitter.split(X_handcrafted, y):\n        y_train, y_val = y[train_idx], y[val_idx]\n        fold_probs = []\n\n        for X_mod in (X_handcrafted, X_deep):\n            X_train, X_val = X_mod[train_idx], X_mod[val_idx]\n            scaler = StandardScaler()\n            X_train_scaled = scaler.fit_transform(X_train)\n            X_val_scaled = scaler.transform(X_val)\n            X_train_reduced, X_val_reduced = reduce_features(X_train_scaled, X_val_scaled, pca_components)\n\n            model = fit_classifier(classifier, X_train_reduced, y_train)\n            fold_probs.append(model.predict_proba(X_val_reduced)[:, 1])\n\n        averaged_prob = np.mean(fold_probs, axis=0)\n        y_true_all.extend(y_val.tolist())\n        y_prob_all.extend(averaged_prob.tolist())\n\n    y_true_all = np.array(y_true_all)\n    y_prob_all = np.array(y_prob_all)\n    y_pred_all = (y_prob_all >= 0.5).astype(int)\n    overall = compute_metrics(y_true_all, y_pred_all, y_prob_all)\n    ci_lo, ci_hi = bootstrap_auc_ci(y_true_all, y_prob_all)\n\n    protocol = \"LOOCV\" if use_loocv else f\"{n_folds}-fold\"\n    print(f\"[{tag}] Aggregated {protocol} results over {len(y_true_all)} out-of-fold predictions:\")\n    print(f\"  AUC={overall.auc:.3f} (95% CI [{ci_lo:.3f}, {ci_hi:.3f}])  F1={overall.f1:.3f}  \"\n          f\"Acc={overall.accuracy:.3f}  Sens={overall.sensitivity:.3f}  Spec={overall.specificity:.3f}\\n\")\n\n    return overall, (ci_lo, ci_hi)\n\n\n# ==========================================================================\n# MAIN\n# ==========================================================================\n\nif __name__ == \"__main__\":\n    manifest = load_gold_manifest(TARGET_FINDING)\n\n    # Build each feature set once, reuse across all the ablations below.\n    X_handcrafted, y, kept_ids = build_feature_matrix(manifest, feature_mode=\"handcrafted\", n_key_slices=N_KEY_SLICES)\n    X_deep, _, _ = build_feature_matrix(manifest, feature_mode=\"deep\", n_key_slices=N_KEY_SLICES)\n    X_fused, _, _ = build_feature_matrix(manifest, feature_mode=\"fused\", n_key_slices=N_KEY_SLICES)\n    print(f\"Feature matrices built: handcrafted={X_handcrafted.shape}, deep={X_deep.shape}, \"\n          f\"fused={X_fused.shape}, kept {len(kept_ids)}/{len(manifest)} exams\\n\")\n\n    summary = {}\n\n    configs = [\n        # (tag,                 X,              pca_components)\n        (\"handcrafted_pca20\",   X_handcrafted,  PCA_COMPONENTS),\n        (\"handcrafted_no_pca\",  X_handcrafted,  None),\n        (\"deep_pca20\",          X_deep,         PCA_COMPONENTS),\n        (\"deep_no_pca\",         X_deep,         None),\n        (\"fused_pca20\",         X_fused,        PCA_COMPONENTS),\n        (\"fused_no_pca\",        X_fused,        None),\n    ]\n\n    for tag, X, pca_comp in configs:\n        print(f\"=== {tag} ===\")\n        overall, fit_times, (ci_lo, ci_hi) = train_and_evaluate(\n            X, y, classifier=\"catboost\", use_loocv=USE_LOOCV, n_folds=N_FOLDS,\n            pca_components=pca_comp, run_shap=True, tag=tag,\n        )\n        summary[tag] = {\"auc\": overall.auc, \"ci\": (ci_lo, ci_hi), \"f1\": overall.f1,\n                         \"sensitivity\": overall.sensitivity, \"specificity\": overall.specificity}\n\n    # Late fusion: separate models per modality, combined by averaging\n    # probabilities, instead of concatenating raw features through one PCA.\n    print(\"=== late_fusion (handcrafted + deep, probability-averaged) ===\")\n    lf_overall, lf_ci = train_and_evaluate_late_fusion(\n        X_handcrafted, X_deep, y, classifier=\"catboost\", use_loocv=USE_LOOCV,\n        n_folds=N_FOLDS, pca_components=None, tag=\"late_fusion\",\n    )\n    summary[\"late_fusion\"] = {\"auc\": lf_overall.auc, \"ci\": lf_ci, \"f1\": lf_overall.f1,\n                               \"sensitivity\": lf_overall.sensitivity, \"specificity\": lf_overall.specificity}\n\n    print(f\"=== Summary (CatBoost, target={TARGET_FINDING}, n={len(y)}) ===\")\n    for tag, stats in summary.items():\n        ci_lo, ci_hi = stats[\"ci\"]\n        print(f\"{tag:20s}  AUC={stats['auc']:.3f} [{ci_lo:.3f}, {ci_hi:.3f}]  \"\n              f\"F1={stats['f1']:.3f}  Sens={stats['sensitivity']:.3f}  Spec={stats['specificity']:.3f}\")\n\n    # Optional: AdaBoost as a second boosting variant on whichever config\n    # above looked strongest, for robustness of the finding.\n    # train_and_evaluate(X_deep, y, classifier=\"adaboost\", use_loocv=USE_LOOCV,\n    #                     pca_components=None, run_shap=False, tag=\"deep_adaboost\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-12T08:13:18.234477Z","iopub.execute_input":"2026-09-12T08:13:18.234843Z","iopub.status.idle":"2026-09-12T08:45:21.249141Z","shell.execute_reply.started":"2026-09-12T08:13:18.234811Z","shell.execute_reply":"2026-09-12T08:45:21.248044Z"}},"outputs":[{"name":"stdout","text":"Gold-labeled studies (any finding annotated): 58\nTarget finding 'ACL' balance: 24 positive / 58 annotated\nFinal manifest: 58 studies with usable series + label\n\nFeature matrices built: handcrafted=(58, 1362), deep=(58, 2048), fused=(58, 3410), kept 58/58 exams\n\n=== handcrafted_pca20 ===\n[handcrafted_pca20] Aggregated LOOCV results over 58 out-of-fold predictions (PCA=20):\n  AUC=0.444 (95% CI [0.287, 0.603])  F1=0.426  Acc=0.534  Sens=0.417  Spec=0.618\n  Mean per-fit time: 0.252s  |  Total fits: 58\n\nSaved SHAP summary to shap_summary_handcrafted_pca20.png\n=== handcrafted_no_pca ===\n[handcrafted_no_pca] Aggregated LOOCV results over 58 out-of-fold predictions (no PCA):\n  AUC=0.642 (95% CI [0.493, 0.773])  F1=0.531  Acc=0.603  Sens=0.542  Spec=0.647\n  Mean per-fit time: 4.068s  |  Total fits: 58\n\nSaved SHAP summary to shap_summary_handcrafted_no_pca.png\n=== deep_pca20 ===\n[deep_pca20] Aggregated LOOCV results over 58 out-of-fold predictions (PCA=20):\n  AUC=0.673 (95% CI [0.526, 0.810])  F1=0.462  Acc=0.638  Sens=0.375  Spec=0.824\n  Mean per-fit time: 0.256s  |  Total fits: 58\n\nSaved SHAP summary to shap_summary_deep_pca20.png\n=== deep_no_pca ===\n[deep_no_pca] Aggregated LOOCV results over 58 out-of-fold predictions (no PCA):\n  AUC=0.341 (95% CI [0.210, 0.473])  F1=0.143  Acc=0.379  Sens=0.125  Spec=0.559\n  Mean per-fit time: 5.665s  |  Total fits: 58\n\nSaved SHAP summary to shap_summary_deep_no_pca.png\n=== fused_pca20 ===\n[fused_pca20] Aggregated LOOCV results over 58 out-of-fold predictions (PCA=20):\n  AUC=0.556 (95% CI [0.392, 0.706])  F1=0.421  Acc=0.621  Sens=0.333  Spec=0.824\n  Mean per-fit time: 0.254s  |  Total fits: 58\n\nSaved SHAP summary to shap_summary_fused_pca20.png\n=== fused_no_pca ===\n[fused_no_pca] Aggregated LOOCV results over 58 out-of-fold predictions (no PCA):\n  AUC=0.570 (95% CI [0.416, 0.714])  F1=0.435  Acc=0.552  Sens=0.417  Spec=0.647\n  Mean per-fit time: 9.831s  |  Total fits: 58\n\nSaved SHAP summary to shap_summary_fused_no_pca.png\n=== late_fusion (handcrafted + deep, probability-averaged) ===\n[late_fusion] Aggregated LOOCV results over 58 out-of-fold predictions:\n  AUC=0.502 (95% CI [0.360, 0.646])  F1=0.375  Acc=0.483  Sens=0.375  Spec=0.559\n\n=== Summary (CatBoost, target=ACL, n=58) ===\nhandcrafted_pca20     AUC=0.444 [0.287, 0.603]  F1=0.426  Sens=0.417  Spec=0.618\nhandcrafted_no_pca    AUC=0.642 [0.493, 0.773]  F1=0.531  Sens=0.542  Spec=0.647\ndeep_pca20            AUC=0.673 [0.526, 0.810]  F1=0.462  Sens=0.375  Spec=0.824\ndeep_no_pca           AUC=0.341 [0.210, 0.473]  F1=0.143  Sens=0.125  Spec=0.559\nfused_pca20           AUC=0.556 [0.392, 0.706]  F1=0.421  Sens=0.333  Spec=0.824\nfused_no_pca          AUC=0.570 [0.416, 0.714]  F1=0.435  Sens=0.417  Spec=0.647\nlate_fusion           AUC=0.502 [0.360, 0.646]  F1=0.375  Sens=0.375  Spec=0.559\n","output_type":"stream"}],"execution_count":10}]}