{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import torch\nprint(f\"torch: {torch.__version__}\")\nprint(f\"CUDA available: {torch.cuda.is_available()}\")\nprint(f\"CUDA device count: {torch.cuda.device_count()}\")\nfor i in range(torch.cuda.device_count()):\n    print(f\"  cuda:{i} = {torch.cuda.get_device_name(i)}\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-09-15T08:46:57.674164Z","iopub.execute_input":"2026-09-15T08:46:57.674672Z","iopub.status.idle":"2026-09-15T08:47:01.947911Z","shell.execute_reply.started":"2026-09-15T08:46:57.674636Z","shell.execute_reply":"2026-09-15T08:47:01.946697Z"}},"outputs":[{"name":"stdout","text":"torch: 2.10.0+cu128\nCUDA available: True\nCUDA device count: 2\n  cuda:0 = Tesla T4\n  cuda:1 = Tesla T4\n","output_type":"stream"}],"execution_count":1},{"cell_type":"code","source":"!pip install -q --no-input bitsandbytes accelerate\nimport bitsandbytes as bnb\nprint(f\"bitsandbytes: {bnb.__version__}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-15T08:48:06.891827Z","iopub.execute_input":"2026-09-15T08:48:06.892314Z","iopub.status.idle":"2026-09-15T08:48:19.77555Z","shell.execute_reply.started":"2026-09-15T08:48:06.892281Z","shell.execute_reply":"2026-09-15T08:48:19.774709Z"}},"outputs":[{"name":"stdout","text":"\u001b[2K   \u001b[90m━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\u001b[0m \u001b[32m43.1/43.1 MB\u001b[0m \u001b[31m41.2 MB/s\u001b[0m eta \u001b[36m0:00:00\u001b[0m:00:01\u001b[0m00:01\u001b[0m\n\u001b[?25hbitsandbytes: 0.50.2\n","output_type":"stream"}],"execution_count":2},{"cell_type":"code","source":"from pathlib import Path\n\nMODEL_PATH = None\nfor p in Path(\"/kaggle/input/models\").rglob(\"config.json\"):\n    s = str(p).lower()\n    if \"qwen3\" in s and \"8b\" in s and \"gptq\" not in s:\n        MODEL_PATH = p.parent\n        break\nassert MODEL_PATH is not None, \"Attach a non-GPTQ Qwen3-8B model\"\nprint(\"Model:\", MODEL_PATH)\n\nINPUT = Path(\"/kaggle/input\")\nsample_path = next(INPUT.rglob(\"sample_submission.csv\"))\nDATA = sample_path.parent\nprint(\"Competition data:\", DATA)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-15T08:49:19.272083Z","iopub.execute_input":"2026-09-15T08:49:19.273516Z","iopub.status.idle":"2026-09-15T08:49:19.287321Z","shell.execute_reply.started":"2026-09-15T08:49:19.273475Z","shell.execute_reply":"2026-09-15T08:49:19.286363Z"}},"outputs":[{"name":"stdout","text":"Model: /kaggle/input/models/manojkumarcs28/qwen3-8b/pytorch/qwen3-8b/1/Qwen 3-8B\nCompetition data: /kaggle/input/competitions/rsna-knee-abnormality-detection\n","output_type":"stream"}],"execution_count":3},{"cell_type":"code","source":"import torch\nfrom transformers import AutoModelForCausalLM, AutoTokenizer, BitsAndBytesConfig\n\nbnb_config = BitsAndBytesConfig(\n    load_in_4bit=True,\n    bnb_4bit_quant_type=\"nf4\",\n    bnb_4bit_compute_dtype=torch.float16,\n    bnb_4bit_use_double_quant=True,\n)\n\ntokenizer = AutoTokenizer.from_pretrained(str(MODEL_PATH), trust_remote_code=True)\nif tokenizer.pad_token is None:\n    tokenizer.pad_token = tokenizer.eos_token\ntokenizer.padding_side = \"left\"\n\nmodel = AutoModelForCausalLM.from_pretrained(\n    str(MODEL_PATH),\n    quantization_config=bnb_config,\n    device_map=\"auto\",\n    trust_remote_code=True,\n)\nmodel.eval()\n\n# Verify\nprint(f\"CUDA available: {torch.cuda.is_available()}\")\nprint(f\"Model device: {next(model.parameters()).device}\")\nprint(\"Model loaded.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-15T08:49:31.665236Z","iopub.execute_input":"2026-09-15T08:49:31.665688Z","iopub.status.idle":"2026-09-15T08:52:13.196034Z","shell.execute_reply.started":"2026-09-15T08:49:31.665657Z","shell.execute_reply":"2026-09-15T08:52:13.195141Z"}},"outputs":[{"output_type":"display_data","data":{"text/plain":"Loading weights:   0%|          | 0/399 [00:00<?, ?it/s]","application/vnd.jupyter.widget-view+json":{"version_major":2,"version_minor":0,"model_id":"2cb53418f21e488b8ef47a8110e8e510"}},"metadata":{}},{"name":"stderr","text":"The module name Qwen 3_hyphen_8B (originally Qwen 3-8B) is not a valid Python identifier. Please rename the original module to avoid import issues.\n","output_type":"stream"},{"name":"stdout","text":"CUDA available: True\nModel device: cuda:0\nModel loaded.\n","output_type":"stream"}],"execution_count":4},{"cell_type":"code","source":"import json, re, time, torch\nimport pandas as pd\nfrom pathlib import Path\n\n# ---------- data ----------\nINPUT = Path(\"/kaggle/input\")\nDATA = next(INPUT.rglob(\"sample_submission.csv\")).parent\ntrain = pd.read_csv(DATA / \"train.csv\")\ntrain[\"Report\"] = train[\"Report\"].fillna(\"\").astype(str)\n\nLABELS = [\"ACL\",\"MCL\",\"Medial Meniscus\",\"Lateral Meniscus\",\"Medial OA\",\"Lateral OA\",\n          \"PF OA\",\"Effusion\",\"Synovitis\",\"Baker's\",\"Contusion\",\"Fracture\"]\n\ngold = train[train[LABELS].notna().all(axis=1)].head(5)\nother = train[train[LABELS].isna().all(axis=1)].sample(5, random_state=42)\nsample = pd.concat([gold, other]).reset_index(drop=True)\nreports = sample[\"Report\"].tolist()\n\n# ---------- prompt (identical for all 4 tests) ----------\nSYSTEM_PROMPT = (\n    \"You are a musculoskeletal radiologist extracting findings from a knee MRI report. \"\n    \"You read the report in its original language. \"\n    \"You return strict JSON only, no prose, no markdown, no reasoning.\"\n)\n\nUSER_TEMPLATE = \"\"\"Extract the 12 findings from this knee MRI report.\n\nCOMPETITION CRITERIA:\n- ACL: label=1 only for high-grade (full-thickness) tear. Partial, sprain, scar -> 0.\n- MCL: label=1 only for acute high-grade tear. Chronic or mild -> 0.\n- Medial/Lateral Meniscus: label=1 for any meniscal tear.\n- Medial/Lateral/PF OA: label=1 only for substantial high-grade cartilage loss.\n- Effusion: label=1 only for moderate or large effusion. Small/trace -> 0.\n- Synovitis: label=1 if synovitis or synovial thickening described.\n- Baker's: label=1 if a Baker's cyst described, any size.\n- Contusion: label=1 if bone marrow edema or contusion described.\n- Fracture: label=1 if acute fracture described.\n\nFor each finding return:\n  \"label\": 1 (present per criteria), 0 (absent per criteria), -1 (not mentioned)\n  \"confidence\": 0.0 to 1.0\n  \"severity\": one of \"none\",\"mild\",\"moderate\",\"severe\",\"full\",\"partial\",\"degenerative\",\"unknown\"\n  \"evidence\": the exact sentence in the original language. Empty if label=-1.\n\nReturn ONLY a JSON object with these 12 keys:\n[\"ACL\",\"MCL\",\"Medial Meniscus\",\"Lateral Meniscus\",\"Medial OA\",\"Lateral OA\",\n \"PF OA\",\"Effusion\",\"Synovitis\",\"Baker's\",\"Contusion\",\"Fracture\"]\n\nREPORT:\n---\n<<<REPORT>>>\n---\nJSON:\"\"\"\n\ndef build_prompt(report, thinking):\n    messages = [\n        {\"role\": \"system\", \"content\": SYSTEM_PROMPT},\n        {\"role\": \"user\", \"content\": USER_TEMPLATE.replace(\"<<<REPORT>>>\", report[:6000])},\n    ]\n    kw = {\"tokenize\": False, \"add_generation_prompt\": True}\n    try:\n        return tokenizer.apply_chat_template(messages, enable_thinking=thinking, **kw)\n    except TypeError:\n        if not thinking:\n            messages[-1][\"content\"] += \"\\n/no_think\"\n        return tokenizer.apply_chat_template(messages, **kw)\n\n# ---------- parsers ----------\ndef strip_thinking(text):\n    # Qwen3 closes the thinking block with the </think> token\n    for tok in (\"</think>\", \"</think >\", \"</think\\n>\"):\n        if tok in text:\n            return text.split(tok, 1)[1].strip()\n    return text.strip()\n\ndef extract_json(text):\n    text = re.sub(r\"^```(?:json)?\\s*\", \"\", text.strip())\n    text = re.sub(r\"\\s*```$\", \"\", text)\n    s = text.find(\"{\")\n    if s == -1:\n        return None\n    d = 0\n    for i, ch in enumerate(text[s:], s):\n        if ch == \"{\": d += 1\n        elif ch == \"}\":\n            d -= 1\n            if d == 0:\n                try: return json.loads(text[s:i+1])\n                except json.JSONDecodeError: return None\n    return None\n\ndef normalize(obj):\n    if not isinstance(obj, dict): return None\n    out = {}\n    for lab in LABELS:\n        e = obj.get(lab)\n        if not isinstance(e, dict):\n            out[lab] = {\"label\": -1, \"confidence\": 0.0, \"evidence\": \"\"}\n            continue\n        try: lv = int(e.get(\"label\", -1))\n        except (TypeError, ValueError): lv = -1\n        if lv not in (-1, 0, 1): lv = -1\n        try: c = float(e.get(\"confidence\", 0.0))\n        except (TypeError, ValueError): c = 0.0\n        c = max(0.0, min(1.0, c))\n        out[lab] = {\"label\": lv, \"confidence\": c, \"evidence\": str(e.get(\"evidence\",\"\"))[:300]}\n    return out\n\n# ---------- runners ----------\ndef run_single(reports, thinking, max_new):\n    gens = []\n    for r in reports:\n        prompt = build_prompt(r, thinking)\n        enc = tokenizer(prompt, return_tensors=\"pt\").to(model.device)\n        with torch.no_grad():\n            out = model.generate(**enc, max_new_tokens=max_new, do_sample=False,\n                                 pad_token_id=tokenizer.pad_token_id)\n        gens.append(tokenizer.decode(out[0, enc[\"input_ids\"].shape[1]:],\n                                     skip_special_tokens=True))\n    return gens\n\ndef run_batch(reports, thinking, max_new, batch_size=4):\n    gens = []\n    for i in range(0, len(reports), batch_size):\n        batch = reports[i:i+batch_size]\n        prompts = [build_prompt(r, thinking) for r in batch]\n        enc = tokenizer(prompts, return_tensors=\"pt\", padding=True,\n                        truncation=True, max_length=8192).to(model.device)\n        with torch.no_grad():\n            out = model.generate(**enc, max_new_tokens=max_new, do_sample=False,\n                                 pad_token_id=tokenizer.pad_token_id)\n        gens.extend(tokenizer.batch_decode(out[:, enc[\"input_ids\"].shape[1]:],\n                                            skip_special_tokens=True))\n    return gens\n\n# ---------- 4-way test ----------\nconfigs = [\n    (\"single + thinking OFF\", run_single, False, 800),\n    (\"single + thinking ON\",  run_single, True,  2000),\n    (\"batch  + thinking OFF\", run_batch,  False, 800),\n    (\"batch  + thinking ON\",  run_batch,  True,  2000),\n]\n\nsummary = []\nraw_examples = {}\n\nfor name, fn, thinking, max_new in configs:\n    t0 = time.time()\n    gens = fn(reports, thinking, max_new)\n    dt = time.time() - t0\n\n    ok = 0\n    pos_total = 0\n    for g in gens:\n        cleaned = strip_thinking(g) if thinking else g\n        parsed = normalize(extract_json(cleaned))\n        if parsed is not None:\n            ok += 1\n            pos_total += sum(1 for lab in LABELS if parsed[lab][\"label\"] == 1)\n\n    pr = dt / len(reports)\n    proj = pr * 4349 / 3600\n    summary.append((name, ok, pr, dt, proj, pos_total / len(reports)))\n    raw_examples[name] = gens[4][:400]  # report 4 has a gold ACL tear\n    print(f\"{name}: parse {ok}/{len(reports)}  {pr:.1f}s/report  \"\n          f\"total {dt:.1f}s  proj {proj:.1f}h  avg_pos {pos_total/len(reports):.1f}\")\n\nprint(\"\\n\" + \"=\" * 78)\nprint(f\"{'config':24s} {'parse':>7s} {'s/rep':>8s} {'total':>8s} {'4349h':>8s} {'avg_pos':>8s}\")\nprint(\"-\" * 78)\nfor name, ok, pr, dt, proj, ap in summary:\n    print(f\"{name:24s} {ok:>3d}/{len(reports):<3d} {pr:>8.1f} {dt:>8.1f} {proj:>8.1f} {ap:>8.1f}\")\n\nprint(\"\\n--- Raw generation on report 4 (gold ACL tear) ---\")\nfor name in raw_examples:\n    print(f\"\\n[{name}]\\n{raw_examples[name]!r}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-15T08:56:26.828304Z","iopub.execute_input":"2026-09-15T08:56:26.828766Z","iopub.status.idle":"2026-09-15T09:07:14.595164Z","shell.execute_reply.started":"2026-09-15T08:56:26.828724Z","shell.execute_reply":"2026-09-15T09:07:14.594344Z"}},"outputs":[{"name":"stderr","text":"The following generation flags are not valid and may be ignored: ['temperature', 'top_p', 'top_k']. Set `TRANSFORMERS_VERBOSITY=info` for more details.\n","output_type":"stream"},{"name":"stdout","text":"Testing on 10 reports\n\n--- Report 0 (GOLD) ---\nRaw first 200 chars: '{\\n  \"ACL\": {\"label\": 0, \"confidence\": 1.0, \"severity\": \"none\", \"evidence\": \"Ligamentos cruzados y colaterales dentro de límites normales.\"},\\n  \"MCL\": {\"label\": 0, \"confidence\": 1.0, \"severity\": \"none\"'\n  Parsed OK. +3  -4  unmentioned 5\n  Positives: ['Lateral Meniscus', 'Medial OA', 'Effusion']\n\n--- Report 1 (GOLD) ---\nRaw first 200 chars: '{\\n  \"ACL\": {\"label\": 0, \"confidence\": 1.0, \"severity\": \"none\", \"evidence\": \"ACL is intact.\"},\\n  \"MCL\": {\"label\": 0, \"confidence\": 1.0, \"severity\": \"none\", \"evidence\": \"The MCL is intact.\"},\\n  \"Medial '\n  Parsed OK. +5  -3  unmentioned 4\n  Positives: ['Lateral Meniscus', 'Effusion', 'Synovitis', 'Contusion', 'Fracture']\n\n--- Report 2 (GOLD) ---\nRaw first 200 chars: '{\\n  \"ACL\": {\"label\": 0, \"confidence\": 1.0, \"severity\": \"none\", \"evidence\": \"Anterior cruciate ligament (ACL): Normal with both bundles intact.\"},\\n  \"MCL\": {\"label\": 0, \"confidence\": 1.0, \"severity\": \"'\n  Parsed OK. +3  -9  unmentioned 0\n  Positives: ['Medial Meniscus', 'Medial OA', 'PF OA']\n\n--- Report 3 (GOLD) ---\nRaw first 200 chars: '{\\n  \"ACL\": {\"label\": 0, \"confidence\": 1.0, \"severity\": \"none\", \"evidence\": \"There is no increased intensity in the ACL, MCL and LCL, it appears to be normal\"},\\n  \"MCL\": {\"label\": 0, \"confidence\": 1.0,'\n  Parsed OK. +2  -6  unmentioned 4\n  Positives: ['Effusion', 'Synovitis']\n\n--- Report 4 (GOLD) ---\nRaw first 200 chars: '{\\n  \"ACL\": {\"label\": 1, \"confidence\": 1.0, \"severity\": \"full\", \"evidence\": \"complete tear of the anterior cruciate ligament, at mid substance.\"},\\n  \"MCL\": {\"label\": -1, \"confidence\": 0.0, \"severity\": '\n  Parsed OK. +2  -4  unmentioned 6\n  Positives: ['ACL', 'Medial Meniscus']\n\n--- Report 5 (report-only) ---\nRaw first 200 chars: '{\\n  \"ACL\": {\"label\": -1, \"confidence\": 0.0, \"severity\": \"unknown\", \"evidence\": \"\"},\\n  \"MCL\": {\"label\": -1, \"confidence\": 0.0, \"severity\": \"unknown\", \"evidence\": \"\"},\\n  \"Medial Meniscus\": {\"label\": -1,'\n  Parsed OK. +0  -1  unmentioned 11\n  Positives: []\n\n--- Report 6 (report-only) ---\nRaw first 200 chars: '{\\n  \"ACL\": {\"label\": 0, \"confidence\": 1.0, \"severity\": \"none\", \"evidence\": \"\"},\\n  \"MCL\": {\"label\": 0, \"confidence\": 1.0, \"severity\": \"none\", \"evidence\": \"\"},\\n  \"Medial Meniscus\": {\"label\": 1, \"confide'\n  Parsed OK. +6  -2  unmentioned 4\n  Positives: ['Medial Meniscus', 'Lateral Meniscus', 'Lateral OA', 'PF OA', 'Effusion', 'Synovitis']\n\n--- Report 7 (report-only) ---\nRaw first 200 chars: '{\\n  \"ACL\": {\"label\": -1, \"confidence\": 0.0, \"severity\": \"unknown\", \"evidence\": \"\"},\\n  \"MCL\": {\"label\": -1, \"confidence\": 0.0, \"severity\": \"unknown\", \"evidence\": \"\"},\\n  \"Medial Meniscus\": {\"label\": 1, '\n  Parsed OK. +2  -0  unmentioned 10\n  Positives: ['Medial Meniscus', 'Lateral Meniscus']\n\n--- Report 8 (report-only) ---\nRaw first 200 chars: '{\\n  \"ACL\": {\"label\": 0, \"confidence\": 1.0, \"severity\": \"none\", \"evidence\": \"Anterior cruciate ligament (ACL): Normal with both bundles intact.\"},\\n  \"MCL\": {\"label\": 0, \"confidence\": 1.0, \"severity\": \"'\n  Parsed OK. +3  -9  unmentioned 0\n  Positives: ['PF OA', 'Effusion', 'Synovitis']\n\n--- Report 9 (report-only) ---\nRaw first 200 chars: '{\\n  \"ACL\": {\"label\": 1, \"confidence\": 1.0, \"severity\": \"full\", \"evidence\": \"Ο πρόσθιος χιαστός σύνδεσµος παρουσιάζει ολική ρήξη στην µηριαία πρόσφυση αυτού.\"},\\n  \"MCL\": {\"label\": -1, \"confidence\": 0.0'\n  Parsed OK. +4  -0  unmentioned 8\n  Positives: ['ACL', 'Medial Meniscus', 'Effusion', 'Contusion']\n\nParse success: 10/10  (100%)\nTotal time: 647.6s   Per report: 64.8s\n","output_type":"stream"}],"execution_count":8},{"cell_type":"code","source":"import json, re, time, torch\nimport pandas as pd\nfrom pathlib import Path\n\n# ---------- data ----------\nINPUT = Path(\"/kaggle/input\")\nDATA = next(INPUT.rglob(\"sample_submission.csv\")).parent\ntrain = pd.read_csv(DATA / \"train.csv\")\ntrain[\"Report\"] = train[\"Report\"].fillna(\"\").astype(str)\n\nLABELS = [\"ACL\",\"MCL\",\"Medial Meniscus\",\"Lateral Meniscus\",\"Medial OA\",\"Lateral OA\",\n          \"PF OA\",\"Effusion\",\"Synovitis\",\"Baker's\",\"Contusion\",\"Fracture\"]\n\ngold = train[train[LABELS].notna().all(axis=1)].head(5)\nother = train[train[LABELS].isna().all(axis=1)].sample(5, random_state=42)\nsample = pd.concat([gold, other]).reset_index(drop=True)\nreports = sample[\"Report\"].tolist()\n\n# ---------- prompt (identical for all 4 tests) ----------\nSYSTEM_PROMPT = (\n    \"You are a musculoskeletal radiologist extracting findings from a knee MRI report. \"\n    \"You read the report in its original language. \"\n    \"You return strict JSON only, no prose, no markdown, no reasoning.\"\n)\n\nUSER_TEMPLATE = \"\"\"Extract the 12 findings from this knee MRI report.\n\nCOMPETITION CRITERIA:\n- ACL: label=1 only for high-grade (full-thickness) tear. Partial, sprain, scar -> 0.\n- MCL: label=1 only for acute high-grade tear. Chronic or mild -> 0.\n- Medial/Lateral Meniscus: label=1 for any meniscal tear.\n- Medial/Lateral/PF OA: label=1 only for substantial high-grade cartilage loss.\n- Effusion: label=1 only for moderate or large effusion. Small/trace -> 0.\n- Synovitis: label=1 if synovitis or synovial thickening described.\n- Baker's: label=1 if a Baker's cyst described, any size.\n- Contusion: label=1 if bone marrow edema or contusion described.\n- Fracture: label=1 if acute fracture described.\n\nFor each finding return:\n  \"label\": 1 (present per criteria), 0 (absent per criteria), -1 (not mentioned)\n  \"confidence\": 0.0 to 1.0\n  \"severity\": one of \"none\",\"mild\",\"moderate\",\"severe\",\"full\",\"partial\",\"degenerative\",\"unknown\"\n  \"evidence\": the exact sentence in the original language. Empty if label=-1.\n\nReturn ONLY a JSON object with these 12 keys:\n[\"ACL\",\"MCL\",\"Medial Meniscus\",\"Lateral Meniscus\",\"Medial OA\",\"Lateral OA\",\n \"PF OA\",\"Effusion\",\"Synovitis\",\"Baker's\",\"Contusion\",\"Fracture\"]\n\nREPORT:\n---\n<<<REPORT>>>\n---\nJSON:\"\"\"\n\ndef build_prompt(report, thinking):\n    messages = [\n        {\"role\": \"system\", \"content\": SYSTEM_PROMPT},\n        {\"role\": \"user\", \"content\": USER_TEMPLATE.replace(\"<<<REPORT>>>\", report[:6000])},\n    ]\n    kw = {\"tokenize\": False, \"add_generation_prompt\": True}\n    try:\n        return tokenizer.apply_chat_template(messages, enable_thinking=thinking, **kw)\n    except TypeError:\n        if not thinking:\n            messages[-1][\"content\"] += \"\\n/no_think\"\n        return tokenizer.apply_chat_template(messages, **kw)\n\n# ---------- parsers ----------\ndef strip_thinking(text):\n    # Qwen3 closes the thinking block with the </think> token\n    for tok in (\"</think>\", \"</think >\", \"</think\\n>\"):\n        if tok in text:\n            return text.split(tok, 1)[1].strip()\n    return text.strip()\n\ndef extract_json(text):\n    text = re.sub(r\"^```(?:json)?\\s*\", \"\", text.strip())\n    text = re.sub(r\"\\s*```$\", \"\", text)\n    s = text.find(\"{\")\n    if s == -1:\n        return None\n    d = 0\n    for i, ch in enumerate(text[s:], s):\n        if ch == \"{\": d += 1\n        elif ch == \"}\":\n            d -= 1\n            if d == 0:\n                try: return json.loads(text[s:i+1])\n                except json.JSONDecodeError: return None\n    return None\n\ndef normalize(obj):\n    if not isinstance(obj, dict): return None\n    out = {}\n    for lab in LABELS:\n        e = obj.get(lab)\n        if not isinstance(e, dict):\n            out[lab] = {\"label\": -1, \"confidence\": 0.0, \"evidence\": \"\"}\n            continue\n        try: lv = int(e.get(\"label\", -1))\n        except (TypeError, ValueError): lv = -1\n        if lv not in (-1, 0, 1): lv = -1\n        try: c = float(e.get(\"confidence\", 0.0))\n        except (TypeError, ValueError): c = 0.0\n        c = max(0.0, min(1.0, c))\n        out[lab] = {\"label\": lv, \"confidence\": c, \"evidence\": str(e.get(\"evidence\",\"\"))[:300]}\n    return out\n\n# ---------- runners ----------\ndef run_single(reports, thinking, max_new):\n    gens = []\n    for r in reports:\n        prompt = build_prompt(r, thinking)\n        enc = tokenizer(prompt, return_tensors=\"pt\").to(model.device)\n        with torch.no_grad():\n            out = model.generate(**enc, max_new_tokens=max_new, do_sample=False,\n                                 pad_token_id=tokenizer.pad_token_id)\n        gens.append(tokenizer.decode(out[0, enc[\"input_ids\"].shape[1]:],\n                                     skip_special_tokens=True))\n    return gens\n\ndef run_batch(reports, thinking, max_new, batch_size=4):\n    gens = []\n    for i in range(0, len(reports), batch_size):\n        batch = reports[i:i+batch_size]\n        prompts = [build_prompt(r, thinking) for r in batch]\n        enc = tokenizer(prompts, return_tensors=\"pt\", padding=True,\n                        truncation=True, max_length=8192).to(model.device)\n        with torch.no_grad():\n            out = model.generate(**enc, max_new_tokens=max_new, do_sample=False,\n                                 pad_token_id=tokenizer.pad_token_id)\n        gens.extend(tokenizer.batch_decode(out[:, enc[\"input_ids\"].shape[1]:],\n                                            skip_special_tokens=True))\n    return gens\n\n# ---------- 4-way test ----------\nconfigs = [\n    (\"single + thinking OFF\", run_single, False, 800),\n    (\"single + thinking ON\",  run_single, True,  2000),\n    (\"batch  + thinking OFF\", run_batch,  False, 800),\n    (\"batch  + thinking ON\",  run_batch,  True,  2000),\n]\n\nsummary = []\nraw_examples = {}\n\nfor name, fn, thinking, max_new in configs:\n    t0 = time.time()\n    gens = fn(reports, thinking, max_new)\n    dt = time.time() - t0\n\n    ok = 0\n    pos_total = 0\n    for g in gens:\n        cleaned = strip_thinking(g) if thinking else g\n        parsed = normalize(extract_json(cleaned))\n        if parsed is not None:\n            ok += 1\n            pos_total += sum(1 for lab in LABELS if parsed[lab][\"label\"] == 1)\n\n    pr = dt / len(reports)\n    proj = pr * 4349 / 3600\n    summary.append((name, ok, pr, dt, proj, pos_total / len(reports)))\n    raw_examples[name] = gens[4][:400]  # report 4 has a gold ACL tear\n    print(f\"{name}: parse {ok}/{len(reports)}  {pr:.1f}s/report  \"\n          f\"total {dt:.1f}s  proj {proj:.1f}h  avg_pos {pos_total/len(reports):.1f}\")\n\nprint(\"\\n\" + \"=\" * 78)\nprint(f\"{'config':24s} {'parse':>7s} {'s/rep':>8s} {'total':>8s} {'4349h':>8s} {'avg_pos':>8s}\")\nprint(\"-\" * 78)\nfor name, ok, pr, dt, proj, ap in summary:\n    print(f\"{name:24s} {ok:>3d}/{len(reports):<3d} {pr:>8.1f} {dt:>8.1f} {proj:>8.1f} {ap:>8.1f}\")\n\nprint(\"\\n--- Raw generation on report 4 (gold ACL tear) ---\")\nfor name in raw_examples:\n    print(f\"\\n[{name}]\\n{raw_examples[name]!r}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-15T09:19:43.338256Z","iopub.execute_input":"2026-09-15T09:19:43.339288Z"}},"outputs":[{"name":"stdout","text":"single + thinking ON: parse 10/10  94.2s/report  total 941.6s  proj 113.8h  avg_pos 2.8\nbatch  + thinking OFF: parse 10/10  40.5s/report  total 404.9s  proj 48.9h  avg_pos 3.0\n","output_type":"stream"}],"execution_count":null}]}