{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os, glob, random, warnings\nimport numpy as np\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nfrom torch.utils.data import Dataset, DataLoader\nimport pydicom\nimport cv2\nimport timm\nfrom sklearn.metrics import roc_auc_score\n\nwarnings.filterwarnings(\"ignore\")\n\n# --------------------------- CONFIG ----------------------------------\ndef find_base():\n    \"\"\"Recursively locate the competition folder under /kaggle/input (the one with train.csv).\"\"\"\n    for root, _dirs, files in os.walk(\"/kaggle/input\"):\n        if \"train.csv\" in files:\n            return root\n    raise FileNotFoundError(\"Could not find train.csv anywhere under /kaggle/input\")\n\nBASE = find_base()\nprint(\"Using data folder:\", BASE)\nIMG_SIZE   = 256          # #TUNE 288/320/384 -> usually better, slower\nSLICES     = 16           # #TUNE slices sampled per study\nBACKBONE   = \"tf_efficientnet_b0_ns\"   # #TUNE convnext_tiny, tf_efficientnet_b3_ns\nBATCH      = 4            # studies per batch (each expands to BATCH*SLICES images)\nEPOCHS     = 4            # #TUNE more epochs (watch the 9h limit on inference only)\nLR         = 3e-4\nNUM_WORKERS= 2\nSEED       = 42\nDEVICE     = \"cuda\" if torch.cuda.is_available() else \"cpu\"\n\nLABELS = [\"ACL\",\"MCL\",\"Medial Meniscus\",\"Lateral Meniscus\",\"Medial OA\",\n          \"Lateral OA\",\"PF OA\",\"Effusion\",\"Synovitis\",\"Baker's\",\"Contusion\",\"Fracture\"]\n\ndef set_seed(s=42):\n    random.seed(s); np.random.seed(s)\n    torch.manual_seed(s); torch.cuda.manual_seed_all(s)\nset_seed(SEED)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-08-11T06:27:08.32573Z","iopub.execute_input":"2026-08-11T06:27:08.325989Z","iopub.status.idle":"2026-08-11T06:27:23.522345Z","shell.execute_reply.started":"2026-08-11T06:27:08.325964Z","shell.execute_reply":"2026-08-11T06:27:23.521447Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =====================================================================\n# RSNA Knee — REPORT-BASED WEAK LABELING (multilingual, negation-aware)\n# ---------------------------------------------------------------------\n# WHY: only ~58 studies are labeled, but all 4407 have a radiology Report.\n# This script reads each Report and guesses the 12 labels from the text,\n# handling several languages (EN/ES/FR/NL/DE/IT/PT) and negations\n# (\"no tear\", \"sin rotura\", \"Aucune\", \"Normal\", \"intact\", ...).\n#\n# HOW TO USE:\n#   1. Run this AFTER you have `BASE` defined (the find_base() cell).\n#   2. It prints how well the text-labels agree with your 58 gold labels.\n#   3. It saves /kaggle/working/weak_labels.csv  (all 4407 studies labeled).\n#   4. Then train the image model on these weak labels (see chat notes).\n#\n# It is a WEAK labeler: not perfect, but the validation print tells you\n# the accuracy per condition so you can trust/tune it.\n# =====================================================================\n\nimport re, unicodedata\nimport numpy as np\nimport pandas as pd\n\nLABELS = [\"ACL\",\"MCL\",\"Medial Meniscus\",\"Lateral Meniscus\",\"Medial OA\",\n          \"Lateral OA\",\"PF OA\",\"Effusion\",\"Synovitis\",\"Baker's\",\"Contusion\",\"Fracture\"]\n\ndef norm(text):\n    \"\"\"lowercase + strip accents so 'rétropatellaire' -> 'retropatellaire'.\"\"\"\n    if not isinstance(text, str):\n        return \"\"\n    text = unicodedata.normalize(\"NFKD\", text)\n    text = \"\".join(c for c in text if not unicodedata.combining(c))\n    return text.lower()\n\n# ------- vocabulary (all UNACCENTED lowercase) -----------------------\nNEG = [ \"no \", \"not \", \"without\", \"sin \", \"sans \", \"geen \", \"niet \", \"kein\",\n        \"keine\", \"nicht\", \"ohne\", \"aucun\", \"aucune\", \"senza\", \"sem \", \"nao \",\n        \"normal\", \"intact\", \"integ\", \"conservad\", \"unremarkable\", \"no evidence\",\n        \"geen aanwijzing\", \"negativ\", \"sans particularite\", \"geen teken\" ]\n\nLAT_MED = [\"medial\",\"interno\",\"interna\",\"mediaal\",\"innen\",\"mediale\",\"interne\",\"medio\"]\nLAT_LAT = [\"lateral\",\"externo\",\"externa\",\"external\",\"laterale\",\"lateraal\",\"aussen\",\"buiten\",\"externe\"]\n\nMENISCUS = [\"meniscus\",\"menisco\",\"menisque\",\"meniskus\",\"menisci\"]\nTEAR     = [\"tear\",\"torn\",\"rupture\",\"ruptur\",\"rotura\",\"roto\",\"rota\",\"dechirure\",\n            \"riss\",\"gerissen\",\"scheur\",\"ruptuur\",\"lesion\",\"rottura\",\"desgarro\",\"fisura\"]\n\nOA_TERMS = [\"osteoarthritis\",\"osteoarthrosis\",\"arthrosis\",\"arthrose\",\"artrosis\",\"gonartrosis\",\n            \"osteoartrosis\",\"degenerative\",\"degenerativ\",\"degenerativa\",\"chondropath\",\"chondrosis\",\n            \"chondral loss\",\"cartilage loss\",\"chondromalac\",\"kraakbeen\",\"artrose\"]\nPATELLOFEM = [\"patellofemoral\",\"femoropatel\",\"patelofemoral\",\"retropatellar\",\"rotulien\",\n              \"retropatelar\",\"patellofemoraal\",\"femoropatellar\",\"femoro-patellar\",\"patellofemorale\"]\n\nEFFUSION  = [\"effusion\",\"derrame\",\"epanchement\",\"erguss\",\"gewrichtsvocht\",\"uitstorting\",\n             \"joint fluid\",\"versamento\",\"hydrops\",\"hidrartrosis\",\"articular fluid\",\n             \"liquido articular\",\"gelenkerguss\",\"liquide articulaire\",\"joint fluid\"]\nSYNOVITIS = [\"synovitis\",\"sinovitis\",\"synovite\",\"synovitida\",\"synoviaal\",\"synovial thickening\",\n             \"synovial hypertrophy\",\"synovial proliferation\",\"engrosamiento sinovial\",\"pannus\",\n             \"hypertrophie synoviale\",\"synoviale verdikking\"]\nBAKER     = [\"baker\",\"bakerse\",\"bakercyste\",\"bakerzyste\"]\nCYST      = [\"cyst\",\"quiste\",\"kyste\",\"cyste\",\"zyste\",\"cisti\",\"cisto\",\"ciste\"]\nPOPLITEAL = [\"poplite\",\"poplitea\",\"poplitee\",\"popliteal\",\"poplitealer\"]\nCONTUSION = [\"bone contusion\",\"bone bruise\",\"bone marrow edema\",\"bone marrow oedema\",\n             \"contusion osseuse\",\"edema oseo\",\"edema de medula\",\"knochenodem\",\"knochenmarkodem\",\n             \"botcontusie\",\"beenmergoedeem\",\"oedeme osseux\",\"contusion osea\",\"medular edema\",\n             \"bone marrow lesion\"]\nFRACTURE  = [\"fracture\",\"fractura\",\"frattura\",\"fraktur\",\"fraktuur\",\"breuk\",\"broken bone\"]\n\nACL = [\"anterior cruciate\",\"ligamento cruzado anterior\",\"ligament croise anterieur\",\n       \"voorste kruisband\",\"vorderes kreuzband\",\"legamento crociato anteriore\",\" acl \",\" lca \",\" vkb \"]\nMCL = [\"medial collateral\",\"ligamento colateral medial\",\"ligament collateral medial\",\n       \"mediale collaterale\",\"innenband\",\"mediaal collateraal\",\"legamento collaterale mediale\",\" mcl \"]\n\n# word-boundary negation cues. PRE = negation before the finding (\"no tear\",\n# \"sin rotura\"). POST = \"<finding>: normal / intact\" style negation after it.\nNEG_PRE_RE  = re.compile(r\"\\b(no|not|without|sin|sans|geen|niet|kein|keine|nicht|ohne|\"\n                         r\"aucun|aucune|senza|sem|nao|non|zonder|neither)\\b\")\nNEG_POST_RE = re.compile(r\"\\b(normal|normale|normales|intact|integr[oa]|conservad[oa]|\"\n                         r\"conserve[e]?s?|unremarkable|preserved|behouden|erhalten)\\b\")\n\ndef any_in(terms, window):\n    return any(t in window for t in terms)\n\ndef _terms_re(terms):\n    return re.compile(\"|\".join(re.escape(t) for t in terms))\n\ndef neg_before(text, s, W=22):\n    return NEG_PRE_RE.search(text[max(0, s-W):s]) is not None\n\ndef neg_after(text, e, W=16):\n    return NEG_POST_RE.search(text[e:e+W]) is not None\n\ndef struct_abn_positive(text, struct_terms, abn_terms, lat_terms=None, W=60):\n    \"\"\"Positive if an abnormality term sits near the structure (with laterality) and is not negated.\"\"\"\n    abn_re = _terms_re(abn_terms)\n    for m in _terms_re(struct_terms).finditer(text):\n        s, e = m.start(), m.end()\n        ws, we = max(0, s-W), e+W\n        window = text[ws:we]\n        if lat_terms and not any_in(lat_terms, window):\n            continue\n        for am in abn_re.finditer(window):\n            a_abs_s, a_abs_e = ws + am.start(), ws + am.end()\n            if not neg_before(text, a_abs_s) and not neg_after(text, a_abs_e):\n                return 1\n    return 0\n\ndef direct_positive(text, terms, lat_terms=None, W=55):\n    \"\"\"Positive if a condition term appears (with laterality if needed) and is not negated.\"\"\"\n    for m in _terms_re(terms).finditer(text):\n        s, e = m.start(), m.end()\n        if lat_terms and not any_in(lat_terms, text[max(0, s-W):e+W]):\n            continue\n        if not neg_before(text, s) and not neg_after(text, e):\n            return 1\n    return 0\n\ndef label_report(report):\n    t = norm(report)\n    out = {}\n    out[\"ACL\"] = struct_abn_positive(t, ACL, TEAR)\n    out[\"MCL\"] = struct_abn_positive(t, MCL, TEAR + [\"sprain\",\"esguince\",\"entorse\",\"zerrung\"])\n    out[\"Medial Meniscus\"]  = struct_abn_positive(t, MENISCUS, TEAR, LAT_MED)\n    out[\"Lateral Meniscus\"] = struct_abn_positive(t, MENISCUS, TEAR, LAT_LAT)\n    out[\"Medial OA\"]  = direct_positive(t, OA_TERMS, LAT_MED)\n    out[\"Lateral OA\"] = direct_positive(t, OA_TERMS, LAT_LAT)\n    # PF OA: a patellofemoral-region mention that is not negated, and any degenerative/chondral term present\n    out[\"PF OA\"] = 1 if (direct_positive(t, PATELLOFEM) and any_in(OA_TERMS, t)) else 0\n    out[\"Effusion\"]  = direct_positive(t, EFFUSION)\n    out[\"Synovitis\"] = direct_positive(t, SYNOVITIS)\n    # Baker's: explicit \"baker\" OR a popliteal + cyst combo (not just \"popliteal\")\n    out[\"Baker's\"]   = 1 if (direct_positive(t, BAKER) or\n                             struct_abn_positive(t, POPLITEAL, CYST, None, W=30)) else 0\n    out[\"Contusion\"] = direct_positive(t, CONTUSION)\n    out[\"Fracture\"]  = direct_positive(t, FRACTURE)\n    return out\n\n# --------------------------- RUN -------------------------------------\ntrain = pd.read_csv(f\"{BASE}/train.csv\")\nweak = train[[\"StudyInstanceUID\"]].copy()\npred = train[\"Report\"].apply(label_report).apply(pd.Series)\nfor c in LABELS:\n    weak[c] = pred[c].astype(int)\n\n# ---- validate against the 58 gold labels ----\ngold = train.dropna(subset=LABELS).copy()\nif len(gold) > 0:\n    print(f\"Validating text-labels against {len(gold)} gold studies:\\n\")\n    idx = gold.index\n    print(f\"{'Label':<18} {'Acc':>6} {'Recall':>7} {'Prec':>6}  gold_pos  pred_pos\")\n    accs = []\n    for c in LABELS:\n        g = gold[c].astype(int).values\n        p = weak.loc[idx, c].astype(int).values\n        acc = (g == p).mean()\n        tp = ((g == 1) & (p == 1)).sum()\n        rec = tp / max(1, (g == 1).sum())\n        prec = tp / max(1, (p == 1).sum())\n        accs.append(acc)\n        print(f\"{c:<18} {acc:6.2f} {rec:7.2f} {prec:6.2f}  {int((g==1).sum()):>7}  {int((p==1).sum()):>7}\")\n    print(f\"\\nMean accuracy across labels: {np.mean(accs):.3f}\")\n\n# overwrite gold rows with the TRUE labels (best of both worlds)\nfor c in LABELS:\n    weak.loc[gold.index, c] = gold[c].astype(int).values\n\nweak.to_csv(\"/kaggle/working/weak_labels.csv\", index=False)\nprint(\"\\nSaved /kaggle/working/weak_labels.csv  shape:\", weak.shape)\nprint(\"Positive counts per label:\\n\", weak[LABELS].sum())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-11T06:32:34.920994Z","iopub.execute_input":"2026-08-11T06:32:34.921691Z","iopub.status.idle":"2026-08-11T06:32:38.329533Z","shell.execute_reply.started":"2026-08-11T06:32:34.921662Z","shell.execute_reply":"2026-08-11T06:32:38.32888Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =====================================================================\n# RSNA Knee — TRAINING ON WEAK (REPORT-DERIVED) LABELS\n# ---------------------------------------------------------------------\n# Trains the image model on ALL ~4,400 studies using the labels from the\n# reports (weak_labels.csv), and VALIDATES on the 58 real gold studies —\n# so the printed macro-AUC is a trustworthy signal.\n#\n# RUN ORDER in the SAME notebook session:\n#   1. find_base() cell  (defines BASE)\n#   2. knee_report_labels.py cell  (creates /kaggle/working/weak_labels.csv)\n#   3. THIS cell.\n#\n# Settings: GPU on, Internet ON (downloads the pretrained backbone).\n# Output: /kaggle/working/knee_model.pt  (used by the inference notebook).\n# =====================================================================\n\nimport os, glob, random, warnings\nimport numpy as np\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nfrom torch.utils.data import Dataset, DataLoader\nimport pydicom\nimport cv2\nimport timm\nfrom sklearn.metrics import roc_auc_score\n\nwarnings.filterwarnings(\"ignore\")\n\ndef find_base():\n    for root, _dirs, files in os.walk(\"/kaggle/input\"):\n        if \"train.csv\" in files:\n            return root\n    raise FileNotFoundError(\"Could not find train.csv under /kaggle/input\")\nBASE = find_base()\nprint(\"Using data folder:\", BASE)\n\n# --------------------------- CONFIG ----------------------------------\nIMG_SIZE    = 224         # #TUNE 256/288 -> better, slower\nSLICES      = 12          # #TUNE slices per study (more = slower I/O)\nBACKBONE    = \"tf_efficientnet_b0_ns\"   # #TUNE convnext_tiny, tf_efficientnet_b3_ns\nBATCH       = 8           # studies per batch\nEPOCHS      = 6          # #TUNE (each epoch reads ~4400*SLICES DICOMs)\nLR          = 3e-4\nNUM_WORKERS = 4\nMAX_TRAIN   = None        # set e.g. 400 for a quick pipeline test, None = all\nSEED        = 42\nDEVICE      = \"cuda\" if torch.cuda.is_available() else \"cpu\"\n\nLABELS = [\"ACL\",\"MCL\",\"Medial Meniscus\",\"Lateral Meniscus\",\"Medial OA\",\n          \"Lateral OA\",\"PF OA\",\"Effusion\",\"Synovitis\",\"Baker's\",\"Contusion\",\"Fracture\"]\n\nrandom.seed(SEED); np.random.seed(SEED)\ntorch.manual_seed(SEED); torch.cuda.manual_seed_all(SEED)\n\n# --------------------------- DICOM I/O -------------------------------\ndef load_dicom(path, size=IMG_SIZE):\n    try:\n        ds = pydicom.dcmread(path)\n        arr = ds.pixel_array.astype(np.float32)\n        slope = float(getattr(ds, \"RescaleSlope\", 1.0) or 1.0)\n        inter = float(getattr(ds, \"RescaleIntercept\", 0.0) or 0.0)\n        arr = arr * slope + inter\n        lo, hi = np.percentile(arr, 1), np.percentile(arr, 99)\n        if hi <= lo:\n            lo, hi = arr.min(), arr.max()\n        arr = np.clip((arr - lo) / (hi - lo + 1e-6), 0, 1)\n        arr = cv2.resize(arr, (size, size), interpolation=cv2.INTER_AREA)\n        return arr.astype(np.float32)\n    except Exception:\n        return np.zeros((size, size), dtype=np.float32)\n\ndef list_study_slices(study_uid, split=\"train_series\"):\n    return sorted(glob.glob(os.path.join(BASE, split, study_uid, \"*\", \"*.dcm\")))\n\ndef sample_slices(paths, n=SLICES, train=True):\n    if len(paths) == 0:\n        return []\n    if len(paths) <= n:\n        return paths + [paths[-1]] * (n - len(paths))\n    idx = np.linspace(0, len(paths) - 1, n)\n    if train:\n        idx = idx + np.random.uniform(-0.5, 0.5, size=n)\n    idx = np.clip(idx.round().astype(int), 0, len(paths) - 1)\n    return [paths[i] for i in idx]\n\n# --------------------------- DATASET ---------------------------------\nclass KneeStudyDataset(Dataset):\n    def __init__(self, df, train=True):\n        self.df = df.reset_index(drop=True)\n        self.train = train\n    def __len__(self):\n        return len(self.df)\n    def __getitem__(self, i):\n        row = self.df.iloc[i]\n        chosen = sample_slices(list_study_slices(row[\"StudyInstanceUID\"]), SLICES, self.train)\n        imgs = []\n        for p in chosen:\n            g = load_dicom(p)\n            if self.train and random.random() < 0.5:\n                g = g[:, ::-1].copy()\n            imgs.append(g)\n        if len(imgs) == 0:\n            imgs = [np.zeros((IMG_SIZE, IMG_SIZE), np.float32)] * SLICES\n        x = np.stack(imgs)[:, None, :, :]\n        x = np.repeat(x, 3, axis=1)\n        y = torch.tensor([row[c] for c in LABELS], dtype=torch.float32)\n        return torch.from_numpy(x), y\n\n# --------------------------- MODEL -----------------------------------\nclass KneeModel(nn.Module):\n    def __init__(self, backbone=BACKBONE, n_out=12, pretrained=True):\n        super().__init__()\n        self.net = timm.create_model(backbone, pretrained=pretrained,\n                                     num_classes=n_out, in_chans=3)\n    def forward(self, x):\n        b, s, c, h, w = x.shape\n        x = x.view(b * s, c, h, w)\n        return self.net(x).view(b, s, -1).mean(1)\n\n# --------------------------- DATA PREP -------------------------------\nweak = pd.read_csv(\"/kaggle/working/weak_labels.csv\")     # from the labeler cell\nraw  = pd.read_csv(f\"{BASE}/train.csv\")\ngold_ids = set(raw.dropna(subset=LABELS)[\"StudyInstanceUID\"])\n\nfor c in LABELS:\n    weak[c] = weak[c].astype(np.float32)\n\nval_df   = weak[weak[\"StudyInstanceUID\"].isin(gold_ids)].reset_index(drop=True)   # 58 gold\ntrain_df = weak[~weak[\"StudyInstanceUID\"].isin(gold_ids)].reset_index(drop=True)  # ~4349 weak\nif MAX_TRAIN:\n    train_df = train_df.sample(MAX_TRAIN, random_state=SEED).reset_index(drop=True)\nprint(f\"train (weak): {len(train_df)}   val (gold): {len(val_df)}\")\n\ntrain_dl = DataLoader(KneeStudyDataset(train_df, True), batch_size=BATCH, shuffle=True,\n                      num_workers=NUM_WORKERS, pin_memory=True, drop_last=True)\nval_dl   = DataLoader(KneeStudyDataset(val_df, False), batch_size=BATCH, shuffle=False,\n                      num_workers=NUM_WORKERS, pin_memory=True)\n\n# --------------------------- TRAIN LOOP ------------------------------\nmodel  = KneeModel(pretrained=True).to(DEVICE)\nopt    = torch.optim.AdamW(model.parameters(), lr=LR, weight_decay=1e-5)\nsched  = torch.optim.lr_scheduler.CosineAnnealingLR(opt, T_max=EPOCHS * max(1, len(train_dl)))\nscaler = torch.cuda.amp.GradScaler()\nlossfn = nn.BCEWithLogitsLoss()\n\ndef evaluate():\n    model.eval()\n    preds, gts = [], []\n    with torch.no_grad():\n        for x, y in val_dl:\n            x = x.to(DEVICE)\n            with torch.cuda.amp.autocast():\n                p = torch.sigmoid(model(x)).float().cpu().numpy()\n            preds.append(p); gts.append(y.numpy())\n    preds, gts = np.concatenate(preds), np.concatenate(gts)\n    aucs = [roc_auc_score(gts[:, j], preds[:, j]) for j in range(12)\n            if len(np.unique(gts[:, j])) > 1]\n    return float(np.mean(aucs)) if aucs else float(\"nan\")\n\nbest = -1\nfor epoch in range(EPOCHS):\n    model.train()\n    running = 0.0\n    for step, (x, y) in enumerate(train_dl):\n        x, y = x.to(DEVICE), y.to(DEVICE)\n        opt.zero_grad()\n        with torch.cuda.amp.autocast():\n            loss = lossfn(model(x), y)\n        scaler.scale(loss).backward()\n        scaler.step(opt); scaler.update(); sched.step()\n        running += loss.item()\n        if step % 50 == 0:\n            print(f\"epoch {epoch} step {step}/{len(train_dl)} loss {running/(step+1):.4f}\")\n    auc = evaluate()\n    print(f\"==> epoch {epoch} GOLD val macro-AUC: {auc:.4f}\")\n    if auc > best:\n        best = auc\n        torch.save({\"state_dict\": model.state_dict(), \"backbone\": BACKBONE,\n                    \"img_size\": IMG_SIZE, \"slices\": SLICES, \"labels\": LABELS},\n                   \"/kaggle/working/knee_model.pt\")\n        print(\"   saved best model.\")\n\nprint(\"Best GOLD val macro-AUC:\", best)\nprint(\"Done -> /kaggle/working/knee_model.pt\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-11T06:46:51.405827Z","iopub.execute_input":"2026-08-11T06:46:51.406464Z"}},"outputs":[],"execution_count":null}]}