{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# --- Optional installs -------------------------------------------------\n# pydicom alone can decode \"Explicit/Implicit VR Little Endian\" (uncompressed).\n# The dataset notes mention JPEG Lossless and JPEG 2000 transfer syntaxes too,\n# which need extra decoder plugins. Uncomment if you hit a\n# \"Unable to decode pixel data\" / NotImplementedError when reading pixels.\n\n# %pip install -q pydicom pylibjpeg pylibjpeg-libjpeg pylibjpeg-openjpeg python-gdcm\n# %pip install -q ipywidgets\n\nimport os\nimport re\nimport glob\nimport warnings\nfrom pathlib import Path\nfrom collections import Counter\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut  # safe no-op for MR without VOI LUT\n\nwarnings.filterwarnings(\"ignore\")\nsns.set_theme(style=\"whitegrid\")\npd.set_option(\"display.max_columns\", 50)","metadata":{"_uuid":"e90a8825-454d-4669-befa-4d57dbf22d8d","_cell_guid":"92fe0d42-1f0e-4d8d-a521-9214d5535f33","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2026-08-19T08:34:58.29574Z","iopub.execute_input":"2026-08-19T08:34:58.295974Z","iopub.status.idle":"2026-08-19T08:34:58.301383Z","shell.execute_reply.started":"2026-08-19T08:34:58.295957Z","shell.execute_reply":"2026-08-19T08:34:58.300867Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Known path for this run (confirmed working directory on Kaggle)\nDATA_DIR = Path(\"/kaggle/input/competitions/rsna-knee-abnormality-detection\")\n\nif not DATA_DIR.exists():\n    # Fallback: depth-limited search under /kaggle/input in case the mount path\n    # changes on a re-run. Bounded depth so it never walks into train_series/,\n    # which can hold tens of thousands of DICOM files.\n    DATA_DIR = None\n    kaggle_input = Path(\"/kaggle/input\")\n    if kaggle_input.exists():\n        hits = []\n        for depth in range(4):\n            pattern = \"/\".join([\"*\"] * depth + [\"train.csv\"]) if depth else \"train.csv\"\n            hits = sorted(kaggle_input.glob(pattern))\n            if hits:\n                break\n        if hits:\n            DATA_DIR = hits[0].parent\n            if len(hits) > 1:\n                print(f\"Note: found {len(hits)} train.csv files, using {DATA_DIR}.\")\n                for h in hits:\n                    print(\" -\", h.parent)\n\n    if DATA_DIR is None:\n        if Path(\"train.csv\").exists():\n            DATA_DIR = Path(\".\")\n        else:\n            # <-- EDIT THIS if nothing above found it -->\n            DATA_DIR = Path(\"./rsna-knee-abnormality-detection\")\n\nDATA_DIR = Path(DATA_DIR)\nTRAIN_SERIES_DIR = DATA_DIR / \"train_series\"\nTEST_SERIES_DIR = DATA_DIR / \"test_series\"\n\nprint(\"DATA_DIR:\", DATA_DIR.resolve())\nprint(\"Exists:\", DATA_DIR.exists())\n","metadata":{"_uuid":"5f4eba09-9842-4447-bc6a-d5e2c5d22557","_cell_guid":"80874e90-ab37-4aa6-90e8-d4a3bd1e5b5b","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2026-08-19T08:46:30.05285Z","iopub.execute_input":"2026-08-19T08:46:30.053122Z","iopub.status.idle":"2026-08-19T08:46:30.061433Z","shell.execute_reply.started":"2026-08-19T08:46:30.053073Z","shell.execute_reply":"2026-08-19T08:46:30.060668Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv(DATA_DIR / \"train.csv\")\ntrain_series = pd.read_csv(DATA_DIR / \"train_series.csv\")\ntest = pd.read_csv(DATA_DIR / \"test.csv\")\ntest_series = pd.read_csv(DATA_DIR / \"test_series.csv\")\nsample_sub = pd.read_csv(DATA_DIR / \"sample_submission.csv\")\n\nLABEL_COLS = [\"ACL\", \"MCL\", \"Medial Meniscus\", \"Lateral Meniscus\",\n              \"Medial OA\", \"Lateral OA\", \"PF OA\", \"Effusion\",\n              \"Synovitis\", \"Baker's\", \"Contusion\", \"Fracture\"]\n# Keep only labels that actually exist in the csv (names can vary slightly)\nLABEL_COLS = [c for c in LABEL_COLS if c in train.columns]\n\nprint(\"train:\", train.shape)\nprint(\"train_series:\", train_series.shape)\nprint(\"test:\", test.shape, \" | test_series:\", test_series.shape)\nprint(\"\\nLabel columns found:\", LABEL_COLS)\ntrain.head()\n","metadata":{"_uuid":"ecdb5063-c25b-4daa-b2a2-aa43de011f87","_cell_guid":"81124256-afad-4af0-a061-01e58d6096ef","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2026-08-19T08:46:48.458336Z","iopub.execute_input":"2026-08-19T08:46:48.458652Z","iopub.status.idle":"2026-08-19T08:46:48.685879Z","shell.execute_reply.started":"2026-08-19T08:46:48.458632Z","shell.execute_reply":"2026-08-19T08:46:48.685208Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# A study counts as \"labeled\" if all label columns are non-null for that row\n# (adjust this rule if partial labeling turns out to be more common)\nis_labeled = train[LABEL_COLS].notna().all(axis=1)\n\nprint(f\"Studies with full label rows: {is_labeled.sum()} / {len(train)} \"\n      f\"({is_labeled.mean():.1%})\")\nprint(f\"Studies with NO labels at all: {train[LABEL_COLS].isna().all(axis=1).sum()}\")\nprint(f\"Studies with PARTIAL labels: {(~is_labeled & ~train[LABEL_COLS].isna().all(axis=1)).sum()}\")\n\nlabeled = train[is_labeled].copy()\nlabeled[LABEL_COLS] = labeled[LABEL_COLS].astype(int)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-19T08:48:11.427682Z","iopub.execute_input":"2026-08-19T08:48:11.427927Z","iopub.status.idle":"2026-08-19T08:48:11.443576Z","shell.execute_reply.started":"2026-08-19T08:48:11.427908Z","shell.execute_reply":"2026-08-19T08:48:11.442674Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Prevalence of each finding among labeled studies\nprevalence = labeled[LABEL_COLS].mean().sort_values(ascending=False)\n\nfig, ax = plt.subplots(figsize=(9, 5))\nprevalence.plot(kind=\"barh\", ax=ax, color=\"#4C72B0\")\nax.set_xlabel(\"Fraction of labeled studies positive\")\nax.set_title(\"Label prevalence\")\nfor i, v in enumerate(prevalence.values):\n    ax.text(v, i, f\" {v:.1%}\", va=\"center\")\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-19T08:48:33.961494Z","iopub.execute_input":"2026-08-19T08:48:33.961739Z","iopub.status.idle":"2026-08-19T08:48:34.237047Z","shell.execute_reply.started":"2026-08-19T08:48:33.96172Z","shell.execute_reply":"2026-08-19T08:48:34.236412Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Co-occurrence between findings (phi coefficient / correlation on binary labels)\ncorr = labeled[LABEL_COLS].corr()\n\nfig, ax = plt.subplots(figsize=(9, 7))\nsns.heatmap(corr, annot=True, fmt=\".2f\", cmap=\"coolwarm\", center=0, ax=ax,\n            square=True, cbar_kws={\"shrink\": 0.8})\nax.set_title(\"Label co-occurrence (correlation)\")\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-19T08:48:44.801083Z","iopub.execute_input":"2026-08-19T08:48:44.801354Z","iopub.status.idle":"2026-08-19T08:48:45.196595Z","shell.execute_reply.started":"2026-08-19T08:48:44.80133Z","shell.execute_reply":"2026-08-19T08:48:45.195888Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# How many positive findings does a single study typically have?\nn_positive = labeled[LABEL_COLS].sum(axis=1)\n\nfig, ax = plt.subplots(figsize=(7, 4))\nsns.countplot(x=n_positive, ax=ax, color=\"#55A868\")\nax.set_xlabel(\"Number of positive findings in a study\")\nax.set_ylabel(\"Number of studies\")\nax.set_title(\"Findings per study\")\nplt.tight_layout()\nplt.show()\n\nprint(n_positive.describe())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-19T08:48:57.743182Z","iopub.execute_input":"2026-08-19T08:48:57.743442Z","iopub.status.idle":"2026-08-19T08:48:57.882882Z","shell.execute_reply.started":"2026-08-19T08:48:57.743417Z","shell.execute_reply":"2026-08-19T08:48:57.881998Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"series_per_study = train_series.groupby(\"StudyInstanceUID\").size()\n\nfig, ax = plt.subplots(figsize=(7, 4))\nsns.histplot(series_per_study, bins=range(1, series_per_study.max() + 2), ax=ax)\nax.set_xlabel(\"Series per study\")\nax.set_title(\"How many series make up each study\")\nplt.tight_layout()\nplt.show()\n\nprint(series_per_study.describe())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-19T08:49:27.647227Z","iopub.execute_input":"2026-08-19T08:49:27.647466Z","iopub.status.idle":"2026-08-19T08:49:27.782418Z","shell.execute_reply.started":"2026-08-19T08:49:27.647442Z","shell.execute_reply":"2026-08-19T08:49:27.781835Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, axes = plt.subplots(1, 3, figsize=(15, 4))\n\ntrain_series[\"Anatomical_Plane\"].value_counts().plot(\n    kind=\"bar\", ax=axes[0], color=\"#C44E52\")\naxes[0].set_title(\"Anatomical plane\")\naxes[0].tick_params(axis=\"x\", rotation=0)\n\ntrain_series[\"Fluid_Sensitive\"].value_counts().sort_index().plot(\n    kind=\"bar\", ax=axes[1], color=\"#8172B2\")\naxes[1].set_title(\"Fluid sensitive (0/1)\")\naxes[1].tick_params(axis=\"x\", rotation=0)\n\ntrain_series[\"Fat_Suppression\"].value_counts().sort_index().plot(\n    kind=\"bar\", ax=axes[2], color=\"#CCB974\")\naxes[2].set_title(\"Fat suppression (0/1)\")\naxes[2].tick_params(axis=\"x\", rotation=0)\n\nplt.tight_layout()\nplt.show()\n\n# Cross-tab: are Fluid_Sensitive and Fat_Suppression really correlated as the notes suggest?\npd.crosstab(train_series[\"Fluid_Sensitive\"], train_series[\"Fat_Suppression\"],\n            margins=True)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-19T08:49:56.386405Z","iopub.execute_input":"2026-08-19T08:49:56.386658Z","iopub.status.idle":"2026-08-19T08:49:56.673238Z","shell.execute_reply.started":"2026-08-19T08:49:56.386639Z","shell.execute_reply":"2026-08-19T08:49:56.672342Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Sequence \"recipe\" per plane: e.g. Sagittal PD-fat-sat, Coronal T1, etc.\nrecipe = (train_series.groupby([\"Anatomical_Plane\", \"Fluid_Sensitive\", \"Fat_Suppression\"])\n          .size().rename(\"count\").reset_index()\n          .sort_values(\"count\", ascending=False))\nrecipe\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-19T08:50:07.972405Z","iopub.execute_input":"2026-08-19T08:50:07.972639Z","iopub.status.idle":"2026-08-19T08:50:07.985574Z","shell.execute_reply.started":"2026-08-19T08:50:07.97262Z","shell.execute_reply":"2026-08-19T08:50:07.984909Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 4. Exploring the DICOM folder structure & slice counts\n\nLayout: `train_series/<StudyInstanceUID>/<SeriesInstanceUID>/<SOPInstanceUID>.dcm`\n\nCounting every slice in every series over the whole training set can be slow — this samples\na subset by default. Set `SAMPLE_N = None` to scan everything.\n","metadata":{}},{"cell_type":"code","source":"def series_dir(study_uid, series_uid, split=\"train\"):\n    base = TRAIN_SERIES_DIR if split == \"train\" else TEST_SERIES_DIR\n    return base / study_uid / series_uid\n\ndef list_dicom_files(sdir):\n    return sorted(Path(sdir).glob(\"*.dcm\"))\n\nSAMPLE_N = 300  # set to None to scan the full train_series.csv\n\nrows = train_series if SAMPLE_N is None else train_series.sample(\n    n=min(SAMPLE_N, len(train_series)), random_state=0)\n\nslice_counts = []\nmissing = 0\nfor _, r in rows.iterrows():\n    d = series_dir(r[\"StudyInstanceUID\"], r[\"SeriesInstanceUID\"])\n    if not d.exists():\n        missing += 1\n        continue\n    slice_counts.append(len(list_dicom_files(d)))\n\nprint(f\"Sampled {len(rows)} series | missing on disk: {missing}\")\nslice_counts = pd.Series(slice_counts)\nprint(slice_counts.describe())\n\nfig, ax = plt.subplots(figsize=(8, 4))\nsns.histplot(slice_counts, bins=40, ax=ax)\nax.set_xlabel(\"Slices per series\")\nax.set_title(f\"Slice count distribution (n={len(slice_counts)} sampled series)\")\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-19T08:50:47.541231Z","iopub.execute_input":"2026-08-19T08:50:47.541462Z","iopub.status.idle":"2026-08-19T08:50:48.665748Z","shell.execute_reply.started":"2026-08-19T08:50:47.54144Z","shell.execute_reply":"2026-08-19T08:50:48.665116Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def pick_example_series(anatomical_plane=None, split=\"train\"):\n    df = train_series if split == \"train\" else test_series\n    if anatomical_plane:\n        df = df[df[\"Anatomical_Plane\"] == anatomical_plane]\n    row = df.iloc[0]\n    return row[\"StudyInstanceUID\"], row[\"SeriesInstanceUID\"]\n\nstudy_uid, sid = pick_example_series()\nsdir = series_dir(study_uid, sid)\nfiles = list_dicom_files(sdir)\nprint(f\"Study {study_uid}\\nSeries {sid}\\n{len(files)} slices at {sdir}\")\n\nds = pydicom.dcmread(files[len(files) // 2], force=True)\nprint(\"\\n--- Header (allowlisted tags) ---\")\nprint(ds)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-19T08:51:02.097861Z","iopub.execute_input":"2026-08-19T08:51:02.098199Z","iopub.status.idle":"2026-08-19T08:51:02.233944Z","shell.execute_reply.started":"2026-08-19T08:51:02.098179Z","shell.execute_reply":"2026-08-19T08:51:02.233192Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_pixels(ds):\n    # Return a display-ready float array, robust to missing VOI LUT on MR.\n    arr = ds.pixel_array.astype(np.float32)\n    try:\n        arr = apply_voi_lut(arr, ds)\n    except Exception:\n        pass\n    return arr\n\ndef normalize_for_display(arr, low=1, high=99):\n    lo, hi = np.percentile(arr, [low, high])\n    if hi <= lo:\n        return np.zeros_like(arr)\n    arr = np.clip((arr - lo) / (hi - lo), 0, 1)\n    return arr\n\npixels = load_pixels(ds)\nprint(\"pixel array shape:\", pixels.shape, \"| dtype:\", pixels.dtype)\n\nfig, ax = plt.subplots(figsize=(5, 5))\nax.imshow(normalize_for_display(pixels), cmap=\"gray\")\nax.set_title(f\"Plane: {ds.get('ImageOrientationPatient', 'n/a')}\")\nax.axis(\"off\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-19T08:51:12.579596Z","iopub.execute_input":"2026-08-19T08:51:12.579825Z","iopub.status.idle":"2026-08-19T08:51:12.712034Z","shell.execute_reply.started":"2026-08-19T08:51:12.579808Z","shell.execute_reply":"2026-08-19T08:51:12.711445Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_series_volume(sdir):\n    files = list_dicom_files(sdir)\n    slices = []\n    for f in files:\n        try:\n            d = pydicom.dcmread(f, force=True)\n            slices.append(d)\n        except Exception as e:\n            print(f\"skip {f.name}: {e}\")\n    def sort_key(d):\n        return getattr(d, \"InstanceNumber\", 0)\n    slices.sort(key=sort_key)\n    arrays = [normalize_for_display(load_pixels(d)) for d in slices]\n    return arrays, slices\n\ndef show_montage(arrays, cols=6, title=None):\n    n = len(arrays)\n    rows = int(np.ceil(n / cols))\n    fig, axes = plt.subplots(rows, cols, figsize=(cols * 2, rows * 2))\n    axes = np.atleast_2d(axes)\n    for i, ax in enumerate(axes.flat):\n        if i < n:\n            ax.imshow(arrays[i], cmap=\"gray\")\n            ax.set_title(str(i), fontsize=7)\n        ax.axis(\"off\")\n    if title:\n        fig.suptitle(title)\n    plt.tight_layout()\n    plt.show()\n\narrays, slices = load_series_volume(sdir)\nshow_montage(arrays, cols=6, title=f\"{study_uid[:8]}.../{sid[:8]}...  ({len(arrays)} slices)\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-19T08:51:47.130844Z","iopub.execute_input":"2026-08-19T08:51:47.131118Z","iopub.status.idle":"2026-08-19T08:51:48.880905Z","shell.execute_reply.started":"2026-08-19T08:51:47.131075Z","shell.execute_reply":"2026-08-19T08:51:48.878596Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def show_planes_for_study(study_uid):\n    sub = train_series[train_series[\"StudyInstanceUID\"] == study_uid]\n    fig, axes = plt.subplots(1, 3, figsize=(12, 4.5))\n    for ax, plane in zip(axes, [\"Sagittal\", \"Coronal\", \"Axial\"]):\n        cand = sub[sub[\"Anatomical_Plane\"] == plane]\n        ax.axis(\"off\")\n        if cand.empty:\n            ax.set_title(f\"{plane}: none\")\n            continue\n        r = cand.iloc[0]\n        d = series_dir(study_uid, r[\"SeriesInstanceUID\"])\n        files = list_dicom_files(d)\n        if not files:\n            ax.set_title(f\"{plane}: missing on disk\")\n            continue\n        mid = pydicom.dcmread(files[len(files) // 2], force=True)\n        ax.imshow(normalize_for_display(load_pixels(mid)), cmap=\"gray\")\n        fs = \"fat-sat\" if r[\"Fat_Suppression\"] else \"no fat-sat\"\n        fl = \"fluid-sens\" if r[\"Fluid_Sensitive\"] else \"not fluid-sens\"\n        ax.set_title(f\"{plane}\\n{fl}, {fs}\", fontsize=9)\n    plt.tight_layout()\n    plt.show()\n\nshow_planes_for_study(study_uid)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-19T08:53:38.361473Z","iopub.execute_input":"2026-08-19T08:53:38.361711Z","iopub.status.idle":"2026-08-19T08:53:38.753764Z","shell.execute_reply.started":"2026-08-19T08:53:38.361693Z","shell.execute_reply":"2026-08-19T08:53:38.753083Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"try:\n    from ipywidgets import interact, IntSlider\n\n    def _view(i=0):\n        fig, ax = plt.subplots(figsize=(5, 5))\n        ax.imshow(arrays[i], cmap=\"gray\")\n        ax.set_title(f\"slice {i}/{len(arrays)-1}\")\n        ax.axis(\"off\")\n        plt.show()\n\n    interact(_view, i=IntSlider(min=0, max=len(arrays) - 1, step=1, value=len(arrays)//2))\nexcept ImportError:\n    print(\"ipywidgets not installed — run: %pip install -q ipywidgets\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-19T08:54:31.800158Z","iopub.execute_input":"2026-08-19T08:54:31.80041Z","iopub.status.idle":"2026-08-19T08:54:31.931714Z","shell.execute_reply.started":"2026-08-19T08:54:31.800392Z","shell.execute_reply":"2026-08-19T08:54:31.93122Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"report_lens = train[\"Report\"].dropna().str.len()\n\nfig, ax = plt.subplots(figsize=(8, 4))\nsns.histplot(report_lens, bins=50, ax=ax)\nax.set_xlabel(\"Report length (characters)\")\nax.set_title(\"Radiology report length distribution\")\nplt.tight_layout()\nplt.show()\n\nprint(report_lens.describe())\nprint(f\"\\nReports missing: {train['Report'].isna().sum()} / {len(train)}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-19T08:55:13.533367Z","iopub.execute_input":"2026-08-19T08:55:13.5336Z","iopub.status.idle":"2026-08-19T08:55:13.711586Z","shell.execute_reply.started":"2026-08-19T08:55:13.53358Z","shell.execute_reply":"2026-08-19T08:55:13.711044Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Rough language mix (best-effort; skips gracefully if langdetect isn't installed)\ntry:\n    from langdetect import detect, DetectorFactory\n    DetectorFactory.seed = 0\n\n    sample_reports = train[\"Report\"].dropna().sample(\n        n=min(300, train[\"Report\"].notna().sum()), random_state=0)\n\n    def safe_detect(t):\n        try:\n            return detect(t)\n        except Exception:\n            return \"unknown\"\n\n    langs = sample_reports.apply(safe_detect)\n    print(langs.value_counts())\nexcept ImportError:\n    print(\"langdetect not installed — run: %pip install -q langdetect\")\n    # Cheap heuristic fallback: fraction of non-ASCII characters per report\n    non_ascii_frac = train[\"Report\"].dropna().apply(\n        lambda t: sum(1 for c in t if ord(c) > 127) / max(len(t), 1))\n    print(\"Non-ASCII character fraction (0 = likely English):\")\n    print(non_ascii_frac.describe())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-19T08:55:22.13468Z","iopub.execute_input":"2026-08-19T08:55:22.134961Z","iopub.status.idle":"2026-08-19T08:55:22.289161Z","shell.execute_reply.started":"2026-08-19T08:55:22.134938Z","shell.execute_reply":"2026-08-19T08:55:22.288385Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Most common tokens, very lightweight stopword filter (no external downloads needed)\nSTOPWORDS = set((\n    \"the a an of and or is are was were to in on at for with without by as this that \"\n    \"these those it its no not normal history there be been being has have had\"\n).split())\n\ndef tokenize(text):\n    return re.findall(r\"[a-zA-Z']+\", text.lower())\n\ncounter = Counter()\nfor t in train[\"Report\"].dropna():\n    counter.update(w for w in tokenize(t) if w not in STOPWORDS and len(w) > 2)\n\ntop_words = pd.DataFrame(counter.most_common(30), columns=[\"word\", \"count\"])\n\nfig, ax = plt.subplots(figsize=(8, 8))\nsns.barplot(data=top_words, y=\"word\", x=\"count\", ax=ax, color=\"#4C72B0\")\nax.set_title(\"Most frequent report tokens (English-only heuristic)\")\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-19T08:55:29.575942Z","iopub.execute_input":"2026-08-19T08:55:29.576219Z","iopub.status.idle":"2026-08-19T08:55:30.012925Z","shell.execute_reply.started":"2026-08-19T08:55:29.576202Z","shell.execute_reply":"2026-08-19T08:55:30.012179Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"LABEL_PLANE_HINTS = {\n    \"ACL\": [\"Sagittal\"],\n    \"MCL\": [\"Coronal\"],\n    \"Medial Meniscus\": [\"Sagittal\", \"Coronal\"],\n    \"Lateral Meniscus\": [\"Sagittal\", \"Coronal\"],\n    \"Medial OA\": [\"Coronal\"],\n    \"Lateral OA\": [\"Coronal\"],\n    \"PF OA\": [\"Axial\"],\n    \"Effusion\": [\"Sagittal\", \"Axial\"],\n    \"Synovitis\": [\"Sagittal\", \"Axial\"],\n    \"Baker's\": [\"Axial\", \"Sagittal\"],\n    \"Contusion\": [\"Sagittal\", \"Coronal\"],\n    \"Fracture\": [\"Sagittal\", \"Coronal\"],\n}\n\ndef pick_representative_series(study_uid, preferred_planes, prefer_fluid=True):\n    # Pick one series for a study, walking down a list of preferred planes,\n    # preferring a fluid-sensitive sequence within the chosen plane when available.\n    sub = train_series[train_series[\"StudyInstanceUID\"] == study_uid]\n    if sub.empty:\n        return None\n    for plane in preferred_planes:\n        cand = sub[sub[\"Anatomical_Plane\"] == plane]\n        if cand.empty:\n            continue\n        if prefer_fluid:\n            fl = cand[cand[\"Fluid_Sensitive\"] == 1]\n            if not fl.empty:\n                return fl.iloc[0]\n        return cand.iloc[0]\n    return sub.iloc[0]  # last-resort fallback: whatever series exists\n\ndef load_middle_slice(study_uid, series_row, split=\"train\"):\n    # Load and normalize the middle slice of a series. Returns None on any failure\n    # (missing files on disk, or a transfer syntax pydicom can't decode without extra\n    # plugins) rather than raising, so one bad series doesn't kill a whole grid.\n    if series_row is None:\n        return None\n    sdir = series_dir(study_uid, series_row[\"SeriesInstanceUID\"], split=split)\n    files = list_dicom_files(sdir)\n    if not files:\n        return None\n    try:\n        ds = pydicom.dcmread(files[len(files) // 2], force=True)\n        return normalize_for_display(load_pixels(ds))\n    except Exception:\n        return None\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-19T09:49:53.363219Z","iopub.execute_input":"2026-08-19T09:49:53.363481Z","iopub.status.idle":"2026-08-19T09:49:53.370062Z","shell.execute_reply.started":"2026-08-19T09:49:53.363463Z","shell.execute_reply":"2026-08-19T09:49:53.369356Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def compare_label_slices(label, n_examples=4, seed=0):\n    preferred_planes = LABEL_PLANE_HINTS.get(label, [\"Sagittal\", \"Coronal\", \"Axial\"])\n\n    pos_studies = labeled.loc[labeled[label] == 1, \"StudyInstanceUID\"]\n    neg_studies = labeled.loc[labeled[label] == 0, \"StudyInstanceUID\"]\n\n    n_pos = min(n_examples, len(pos_studies))\n    n_neg = min(n_examples, len(neg_studies))\n    pos_sample = pos_studies.sample(n=n_pos, random_state=seed).tolist() if n_pos else []\n    neg_sample = neg_studies.sample(n=n_neg, random_state=seed).tolist() if n_neg else []\n\n    ncols = max(len(pos_sample), len(neg_sample), 1)\n    fig, axes = plt.subplots(2, ncols, figsize=(ncols * 2.6, 5.6))\n    axes = np.atleast_2d(axes)\n\n    for row_idx, (tag, studies) in enumerate([(\"POS\", pos_sample), (\"NEG\", neg_sample)]):\n        for col in range(ncols):\n            ax = axes[row_idx, col]\n            ax.axis(\"off\")\n            if col >= len(studies):\n                continue\n            study_uid = studies[col]\n            srow = pick_representative_series(study_uid, preferred_planes)\n            img = load_middle_slice(study_uid, srow)\n            if img is None:\n                ax.set_title(f\"{tag}\\n(no image)\", fontsize=8)\n                continue\n            ax.imshow(img, cmap=\"gray\")\n            plane = srow[\"Anatomical_Plane\"] if srow is not None else \"?\"\n            ax.set_title(f\"{tag} | {plane}\\n{study_uid[:10]}...\", fontsize=7)\n\n    fig.suptitle(\n        f\"{label}: positive (top row) vs negative (bottom row)  \"\n        f\"[{len(pos_studies)} pos / {len(neg_studies)} neg in labeled set]\",\n        fontsize=11,\n    )\n    plt.tight_layout()\n    plt.show()\n\nN_EXAMPLES = 4  # studies shown per row, per finding — raise/lower as you like\n\nfor _label in LABEL_COLS:\n    compare_label_slices(_label, n_examples=N_EXAMPLES)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-19T09:50:11.07085Z","iopub.execute_input":"2026-08-19T09:50:11.071117Z","iopub.status.idle":"2026-08-19T09:50:20.351986Z","shell.execute_reply.started":"2026-08-19T09:50:11.071074Z","shell.execute_reply":"2026-08-19T09:50:20.351321Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"try:\n    import ipywidgets as widgets\n    from ipywidgets import IntSlider, Dropdown, Label, VBox, interact_manual\n    from IPython.display import display\n    _WIDGETS_OK = True\nexcept ImportError:\n    _WIDGETS_OK = False\n    print(\"ipywidgets not installed — run: %pip install -q ipywidgets\")\n\nSERIES_VOLUME_CACHE = {}  # (study_uid, series_uid, split) -> list of normalized slice arrays\n\ndef load_series_arrays_cached(study_uid, series_uid, split=\"train\"):\n    key = (study_uid, series_uid, split)\n    if key not in SERIES_VOLUME_CACHE:\n        sdir = series_dir(study_uid, series_uid, split=split)\n        arrays, _slices = load_series_volume(sdir)\n        SERIES_VOLUME_CACHE[key] = arrays\n    return SERIES_VOLUME_CACHE[key]\n\ndef positive_series_for(label, plane, n=5, fluid_only=False, seed=0):\n    # Up to n (study_uid, series_row) pairs: positive studies for `label` that have\n    # a series in the requested `plane`. One series picked per study.\n    pos_studies = labeled.loc[labeled[label] == 1, \"StudyInstanceUID\"].tolist()\n    shuffled = pd.Series(pos_studies).sample(frac=1, random_state=seed).tolist()\n    results = []\n    for study_uid in shuffled:\n        sub = train_series[(train_series[\"StudyInstanceUID\"] == study_uid) &\n                            (train_series[\"Anatomical_Plane\"] == plane)]\n        if fluid_only:\n            sub = sub[sub[\"Fluid_Sensitive\"] == 1]\n        if sub.empty:\n            continue\n        results.append((study_uid, sub.iloc[0]))\n        if len(results) >= n:\n            break\n    return results\n\ndef slider_for_series(study_uid, series_row, label_text=\"\"):\n    # One independent scrub-able slider for a single series.\n    arrays = load_series_arrays_cached(study_uid, series_row[\"SeriesInstanceUID\"])\n    if not arrays:\n        print(f\"{label_text}: no slices found on disk for study {study_uid[:10]}...\")\n        return\n\n    slider = IntSlider(min=0, max=len(arrays) - 1, step=1, value=len(arrays) // 2,\n                        description=\"slice\", continuous_update=False)\n\n    def _show(i):\n        fig, ax = plt.subplots(figsize=(4, 4))\n        ax.imshow(arrays[i], cmap=\"gray\")\n        ax.set_title(f\"{label_text}\\nslice {i + 1}/{len(arrays)}\", fontsize=9)\n        ax.axis(\"off\")\n        plt.show()\n\n    out = widgets.interactive_output(_show, {\"i\": slider})\n    display(VBox([Label(label_text), slider, out]))\n\ndef show_positive_sliders(label, plane, n=5, fluid_only=False):\n    # For one deformity + plane, render up to n independent sliders, one per\n    # positive study, so you can scrub each looking for where the finding shows up.\n    if not _WIDGETS_OK:\n        print(\"ipywidgets not available.\")\n        return\n    matches = positive_series_for(label, plane, n=n, fluid_only=fluid_only)\n    if not matches:\n        print(f\"No positive studies with a {plane} series found for '{label}'.\")\n        return\n    print(f\"{label} — {plane}: {len(matches)} positive example(s)\")\n    for study_uid, series_row in matches:\n        tag = f\"{label} | {plane} | {study_uid[:10]}...\"\n        slider_for_series(study_uid, series_row, label_text=tag)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-19T10:18:56.504449Z","iopub.execute_input":"2026-08-19T10:18:56.504687Z","iopub.status.idle":"2026-08-19T10:18:56.51526Z","shell.execute_reply.started":"2026-08-19T10:18:56.50467Z","shell.execute_reply":"2026-08-19T10:18:56.51451Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if _WIDGETS_OK:\n    _label_dd = Dropdown(options=LABEL_COLS, description=\"Finding:\")\n    _plane_dd = Dropdown(options=[\"Sagittal\", \"Coronal\", \"Axial\"], description=\"Plane:\")\n    _n_slider = IntSlider(min=1, max=8, value=5, description=\"Count:\")\n\n    @interact_manual(label=_label_dd, plane=_plane_dd, n=_n_slider)\n    def _browse_positive_examples(label, plane, n):\n        show_positive_sliders(label, plane, n=n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-19T10:19:15.22545Z","iopub.execute_input":"2026-08-19T10:19:15.225692Z","iopub.status.idle":"2026-08-19T10:19:15.240073Z","shell.execute_reply.started":"2026-08-19T10:19:15.225672Z","shell.execute_reply":"2026-08-19T10:19:15.239287Z"}},"outputs":[],"execution_count":null}]}