{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os, json, warnings\nfrom pathlib import Path\nfrom collections import Counter\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nwarnings.filterwarnings(\"ignore\")\npd.set_option(\"display.max_columns\", 100)\n\nBASE = Path(\"/kaggle/input\")\nprint(\"input dirs:\", [p.name for p in BASE.iterdir()])\n\nfor p in BASE.iterdir():\n    print(\"\\n##\", p)\n    for root, dirs, files in os.walk(p):\n        rel = Path(root).relative_to(p).parts\n        if len(rel) <= 2:\n            print(\"  \" * len(rel), Path(root).name, \"| dirs:\", len(dirs), \"| files:\", len(files))\n        if len(rel) > 2:\n            dirs[:] = []","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-14T18:47:11.599032Z","iopub.execute_input":"2026-08-14T18:47:11.599436Z","iopub.status.idle":"2026-08-14T18:47:24.31873Z","shell.execute_reply.started":"2026-08-14T18:47:11.599405Z","shell.execute_reply":"2026-08-14T18:47:24.318005Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"all_files = []\nfor root, dirs, files in os.walk(BASE):\n    for f in files:\n        all_files.append(Path(root) / f)\n\nprint(\"total files:\", len(all_files))\nprint(\"extensions:\", Counter([p.suffix.lower() for p in all_files]).most_common(20))\n\ncsvs = [p for p in all_files if p.suffix.lower() == \".csv\"]\ndcms = [p for p in all_files if p.suffix.lower() == \".dcm\"]\ntexts = [p for p in all_files if p.suffix.lower() in [\".txt\", \".json\", \".md\"]]\n\nprint(\"\\nCSVs:\")\nfor p in csvs[:30]:\n    print(\" \", p)\n\nprint(\"\\nDICOMs:\", len(dcms))\nfor p in dcms[:5]:\n    print(\" \", p)\n\nprint(\"\\nText/json/md:\")\nfor p in texts[:20]:\n    print(\" \", p)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-14T18:47:24.320035Z","iopub.execute_input":"2026-08-14T18:47:24.320393Z","iopub.status.idle":"2026-08-14T18:51:10.029159Z","shell.execute_reply.started":"2026-08-14T18:47:24.320372Z","shell.execute_reply":"2026-08-14T18:51:10.028527Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tables = {}\n\nfor p in csvs:\n    try:\n        df = pd.read_csv(p)\n        tables[p.name] = df\n        print(\"\\n###\", p.name, df.shape)\n        print(df.columns.tolist())\n        display(df.head(3))\n    except Exception as e:\n        print(\"ERR\", p, repr(e))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-14T18:51:10.030052Z","iopub.execute_input":"2026-08-14T18:51:10.030341Z","iopub.status.idle":"2026-08-14T18:51:10.323517Z","shell.execute_reply.started":"2026-08-14T18:51:10.030318Z","shell.execute_reply":"2026-08-14T18:51:10.32288Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"try:\n    import pydicom\n    print(\"pydicom:\", pydicom.__version__)\nexcept Exception as e:\n    print(\"pydicom unavailable:\", repr(e))\n    raise\n\nsample = dcms[:8000] if len(dcms) > 8000 else dcms\nrows = []\n\nfor p in sample:\n    try:\n        ds = pydicom.dcmread(str(p), stop_before_pixels=True, force=True)\n        rows.append({\n            \"path\": str(p),\n            \"study\": getattr(ds, \"StudyInstanceUID\", None),\n            \"series\": getattr(ds, \"SeriesInstanceUID\", None),\n            \"sop\": getattr(ds, \"SOPInstanceUID\", None),\n            \"instance\": getattr(ds, \"InstanceNumber\", np.nan),\n            \"rows\": getattr(ds, \"Rows\", np.nan),\n            \"cols\": getattr(ds, \"Columns\", np.nan),\n            \"modality\": getattr(ds, \"Modality\", None),\n            \"series_desc\": getattr(ds, \"SeriesDescription\", None),\n            \"seq\": getattr(ds, \"SequenceName\", None),\n        })\n    except Exception:\n        pass\n\nmeta = pd.DataFrame(rows)\nprint(\"meta shape:\", meta.shape)\ndisplay(meta.head())\n\nprint(\"\\nTop series descriptions:\")\nprint(meta[\"series_desc\"].astype(str).value_counts().head(40))\n\nprint(\"\\nunique studies:\", meta[\"study\"].nunique())\nprint(\"unique series:\", meta[\"series\"].nunique())\n\nmeta.to_csv(\"/kaggle/working/dicom_meta_sample.csv\", index=False)\nprint(\"saved: /kaggle/working/dicom_meta_sample.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-14T18:51:10.324368Z","iopub.execute_input":"2026-08-14T18:51:10.324657Z","iopub.status.idle":"2026-08-14T18:52:00.851094Z","shell.execute_reply.started":"2026-08-14T18:51:10.324636Z","shell.execute_reply":"2026-08-14T18:52:00.850261Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_pixel(path):\n    ds = pydicom.dcmread(str(path))\n    arr = ds.pixel_array.astype(np.float32)\n\n    if arr.ndim == 3:\n        if arr.shape[-1] in [3, 4]:\n            arr = arr[..., 0]\n        else:\n            arr = arr[0]\n\n    arr = (arr - arr.min()) / (arr.max() - arr.min() + 1e-6)\n    return arr, ds\n\nsel = meta.dropna(subset=[\"series\"]).sample(min(12, len(meta)), random_state=0)\n\nplt.figure(figsize=(16, 12))\nfor i, (_, r) in enumerate(sel.iterrows()):\n    ax = plt.subplot(3, 4, i + 1)\n    try:\n        img, ds = load_pixel(r.path)\n        ax.imshow(img, cmap=\"gray\")\n        ax.set_title(f\"{str(r.series_desc)[:20]}\\n{int(r.rows)}x{int(r.cols)}\", fontsize=8)\n    except Exception:\n        ax.set_title(\"ERR\")\n    ax.axis(\"off\")\n\nplt.tight_layout()\nplt.savefig(\"/kaggle/working/sample_slices.png\", dpi=150)\nplt.show()\nprint(\"saved: /kaggle/working/sample_slices.png\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-14T18:52:00.852741Z","iopub.execute_input":"2026-08-14T18:52:00.853048Z","iopub.status.idle":"2026-08-14T18:52:05.621815Z","shell.execute_reply.started":"2026-08-14T18:52:00.853025Z","shell.execute_reply":"2026-08-14T18:52:05.620712Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"report_cols = []\n\nfor name, df in tables.items():\n    for c in df.columns:\n        cl = str(c).lower()\n        if any(k in cl for k in [\"report\", \"impression\", \"finding\", \"conclusion\", \"text\", \"заключ\"]):\n            try:\n                desc = df[c].astype(str).str.len().describe().to_dict()\n            except Exception:\n                desc = {}\n            report_cols.append((name, c, desc))\n\nprint(\"report-like columns:\")\nfor name, c, desc in report_cols:\n    print(name, \"|\", c, \"|\", desc)\n\nshown = 0\nfor name, c, desc in report_cols:\n    df = tables[name]\n    for v in df[c].dropna().astype(str).head(3):\n        print(\"\\n---\", name, \"|\", c, \"---\")\n        print(v[:1200])\n        shown += 1\n        if shown >= 6:\n            break\n    if shown >= 6:\n        break","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-14T18:52:05.622853Z","iopub.execute_input":"2026-08-14T18:52:05.623209Z","iopub.status.idle":"2026-08-14T18:52:05.640713Z","shell.execute_reply.started":"2026-08-14T18:52:05.623185Z","shell.execute_reply":"2026-08-14T18:52:05.639863Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"LABELS = [\n    \"ACL\",\n    \"MCL\",\n    \"Medial Meniscus\",\n    \"Lateral Meniscus\",\n    \"Medial OA\",\n    \"Lateral OA\",\n    \"PF OA\",\n    \"Effusion\",\n    \"Synovitis\",\n    \"Baker's\",\n    \"Contusion\",\n    \"Fracture\",\n]\n\nprint(json.dumps(LABELS, ensure_ascii=False, indent=2))\nprint(\"\\nСледующий шаг после EDA: сделать extractor, который из текста заключения возвращает JSON с этими 12 ключами: 0/1.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-14T18:52:05.642003Z","iopub.execute_input":"2026-08-14T18:52:05.642315Z","iopub.status.idle":"2026-08-14T18:52:05.650838Z","shell.execute_reply.started":"2026-08-14T18:52:05.642293Z","shell.execute_reply":"2026-08-14T18:52:05.649826Z"}},"outputs":[],"execution_count":null}]}