{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":13451,"datasetId":654585,"databundleVersionId":1188070}],"dockerImageVersionId":31239,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os, random, shutil, subprocess\nimport pandas as pd\nfrom pathlib import Path\nfrom tqdm import tqdm\n\nSEED = 42\nrandom.seed(SEED)\n\nRSNA_BASE = Path(\"/kaggle/input/rsna-intracranial-hemorrhage-detection/rsna-intracranial-hemorrhage-detection\")\n\nTRAIN_CSV = RSNA_BASE / \"stage_2_train.csv\"\nSAMPLE_SUB = RSNA_BASE / \"stage_2_sample_submission.csv\"\n\n# Your real folders:\nTRAIN_DICOM_DIR = RSNA_BASE / \"stage_2_train\"\nTEST_DICOM_DIR  = RSNA_BASE / \"stage_2_test\"\n\n# Output\nOUT_DIR = Path(\"/kaggle/working/rsna_subset_15k\")\n\n# Subset settings\nN_TRAIN_IMAGES = 15000\nN_TEST_IMAGES  = 0      # optional demo-only test subset; set e.g. 3000 if needed\n\n# Optional adjustment (set to 0 to disable)\nMIN_EPIDURAL = 100      # keep natural prevalence but ensure at least this many epidural positives\n\nprint(\"RSNA_BASE exists:\", RSNA_BASE.exists())\nprint(\"TRAIN_CSV exists:\", TRAIN_CSV.exists())\nprint(\"TRAIN_DICOM_DIR exists:\", TRAIN_DICOM_DIR.exists())\nprint(\"SAMPLE_SUB exists:\", SAMPLE_SUB.exists())\nprint(\"TEST_DICOM_DIR exists:\", TEST_DICOM_DIR.exists())\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-01-02T05:34:44.670366Z","iopub.execute_input":"2026-01-02T05:34:44.670752Z","iopub.status.idle":"2026-01-02T05:34:44.684663Z","shell.execute_reply.started":"2026-01-02T05:34:44.670728Z","shell.execute_reply":"2026-01-02T05:34:44.683548Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_long = pd.read_csv(TRAIN_CSV)  # columns: ID, Label\n\n# Split \"ID\" into base image_id + label_name\ntmp = df_long[\"ID\"].str.rsplit(\"_\", n=1, expand=True)\ndf_long[\"image_id\"] = tmp[0]       # e.g. ID_000012eaf\ndf_long[\"label_name\"] = tmp[1]     # any / epidural / ...\n\n# Pivot long → wide\ndf_wide = (\n    df_long.pivot_table(index=\"image_id\", columns=\"label_name\", values=\"Label\", aggfunc=\"max\")\n    .reset_index()\n)\n\nlabels = [\"any\",\"epidural\",\"intraparenchymal\",\"intraventricular\",\"subarachnoid\",\"subdural\"]\nfor c in labels:\n    if c not in df_wide.columns:\n        df_wide[c] = 0.0\n    df_wide[c] = (df_wide[c].fillna(0.0) >= 0.5).astype(int)\n\nprint(\"Total train images:\", len(df_wide))\nprint(\"Positive counts (full train):\")\nprint(df_wide[labels].sum().sort_values(ascending=False))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-02T05:35:16.536646Z","iopub.execute_input":"2026-01-02T05:35:16.536943Z","iopub.status.idle":"2026-01-02T05:35:46.25622Z","shell.execute_reply.started":"2026-01-02T05:35:16.536926Z","shell.execute_reply":"2026-01-02T05:35:46.255482Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"all_ids = df_wide[\"image_id\"].tolist()\n\ndef sample_once():\n    sel = random.sample(all_ids, N_TRAIN_IMAGES)\n    counts = df_wide[df_wide[\"image_id\"].isin(sel)][labels].sum()\n    return sel, counts\n\nselected_train, subset_counts = None, None\n\n# Try a few times to avoid the rare case of subtype count being 0\nfor attempt in range(10):\n    sel, counts = sample_once()\n    if (counts[1:] > 0).all():  # all subtypes non-zero\n        selected_train, subset_counts = sel, counts\n        print(f\"✅ Non-zero subtype coverage on attempt {attempt+1}\")\n        break\n\nif selected_train is None:\n    selected_train, subset_counts = sel, counts\n    print(\"⚠️ Using last sample (rare).\")\n\nprint(\"Selected train images:\", len(selected_train))\nprint(\"Subset positives BEFORE epidural adjustment:\")\nprint(subset_counts.sort_values(ascending=False))\nprint(\"\\nSubset prevalence BEFORE adjustment:\")\nprint((subset_counts / len(selected_train)).round(4))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-02T05:35:56.188997Z","iopub.execute_input":"2026-01-02T05:35:56.189243Z","iopub.status.idle":"2026-01-02T05:35:56.506107Z","shell.execute_reply.started":"2026-01-02T05:35:56.189227Z","shell.execute_reply":"2026-01-02T05:35:56.505412Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if MIN_EPIDURAL and MIN_EPIDURAL > 0:\n    sel_df = df_wide[df_wide[\"image_id\"].isin(selected_train)]\n    epi_now = int(sel_df[\"epidural\"].sum())\n    print(\"Epidural in subset BEFORE adjustment:\", epi_now)\n\n    if epi_now < MIN_EPIDURAL:\n        need = MIN_EPIDURAL - epi_now\n\n        # epidural positives not already selected\n        epi_pool = df_wide[(df_wide[\"epidural\"] == 1) & (~df_wide[\"image_id\"].isin(selected_train))][\"image_id\"].tolist()\n        random.shuffle(epi_pool)\n        add_ids = epi_pool[:need]\n\n        if len(add_ids) < need:\n            print(f\"⚠️ Only found {len(add_ids)} additional epidural cases (needed {need}). Will add what we can.\")\n\n        need = len(add_ids)\n\n        # remove same number from current selection (remove negatives first to minimize distortion)\n        removable_df = df_wide[df_wide[\"image_id\"].isin(selected_train)].copy()\n        removable = removable_df[removable_df[\"any\"] == 0][\"image_id\"].tolist()  # negatives\n        random.shuffle(removable)\n\n        if len(removable) < need:\n            # if not enough negatives, remove from remaining selection\n            extra = [x for x in selected_train if x not in set(removable)]\n            random.shuffle(extra)\n            removable = removable + extra\n\n        remove_ids = removable[:need]\n\n        selected_set = set(selected_train)\n        selected_set.difference_update(remove_ids)\n        selected_set.update(add_ids)\n        selected_train = list(selected_set)\n\n        print(f\"✅ Adjusted subset: +{len(add_ids)} epidural, -{len(remove_ids)} others\")\n    else:\n        print(\"✅ No epidural adjustment needed.\")\n\n    # Re-check distribution after adjustment\n    sel_df2 = df_wide[df_wide[\"image_id\"].isin(selected_train)]\n    counts2 = sel_df2[labels].sum().sort_values(ascending=False)\n    print(\"\\nSubset positives AFTER adjustment:\")\n    print(counts2)\n    print(\"\\nSubset prevalence AFTER adjustment:\")\n    print((counts2 / len(selected_train)).round(4))\n    print(\"\\nFinal subset size:\", len(selected_train))\nelse:\n    print(\"Skipping epidural adjustment (MIN_EPIDURAL=0).\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-02T05:36:23.330491Z","iopub.execute_input":"2026-01-02T05:36:23.331117Z","iopub.status.idle":"2026-01-02T05:36:24.066861Z","shell.execute_reply.started":"2026-01-02T05:36:23.331093Z","shell.execute_reply":"2026-01-02T05:36:24.066137Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"OUT_DIR.mkdir(parents=True, exist_ok=True)\nTRAIN_OUT = OUT_DIR / \"train_images\"\nTRAIN_OUT.mkdir(parents=True, exist_ok=True)\n\n# Save selected IDs\npd.DataFrame({\"image_id\": selected_train}).to_csv(OUT_DIR / \"train_selected_image_ids.csv\", index=False)\n\n# Filter stage_2_train.csv to only selected images (keep Kaggle format ID,Label)\nselected_set = set(selected_train)\ndf_subset_long = df_long[df_long[\"image_id\"].isin(selected_set)][[\"ID\",\"Label\"]].copy()\ndf_subset_long.to_csv(OUT_DIR / \"stage_2_train_subset.csv\", index=False)\n\nprint(\"Wrote:\", OUT_DIR / \"train_selected_image_ids.csv\")\nprint(\"Wrote:\", OUT_DIR / \"stage_2_train_subset.csv\")\nprint(\"Subset CSV rows:\", len(df_subset_long), \"Expected ~\", 6 * len(selected_train))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-02T05:37:00.701469Z","iopub.execute_input":"2026-01-02T05:37:00.701729Z","iopub.status.idle":"2026-01-02T05:37:01.533404Z","shell.execute_reply.started":"2026-01-02T05:37:00.701715Z","shell.execute_reply":"2026-01-02T05:37:01.532612Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"missing = 0\ncopied = 0\n\nfor iid in tqdm(selected_train, desc=\"Copying TRAIN DICOMs\"):\n    src = TRAIN_DICOM_DIR / f\"{iid}.dcm\"\n    dst = TRAIN_OUT / f\"{iid}.dcm\"\n    if src.exists():\n        shutil.copy2(src, dst)\n        copied += 1\n    else:\n        missing += 1\n\nprint(\"Copied:\", copied)\nprint(\"Missing:\", missing)\nprint(\"Files in train_images:\", len(list(TRAIN_OUT.glob(\"*.dcm\"))))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-02T05:37:25.834159Z","iopub.execute_input":"2026-01-02T05:37:25.834417Z","iopub.status.idle":"2026-01-02T05:43:01.042224Z","shell.execute_reply.started":"2026-01-02T05:37:25.834403Z","shell.execute_reply":"2026-01-02T05:43:01.040973Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if N_TEST_IMAGES and N_TEST_IMAGES > 0:\n    assert SAMPLE_SUB.exists(), \"stage_2_sample_submission.csv not found\"\n    assert TEST_DICOM_DIR.exists(), \"stage_2_test folder not found\"\n\n    TEST_OUT = OUT_DIR / \"test_images\"\n    TEST_OUT.mkdir(parents=True, exist_ok=True)\n\n    sub = pd.read_csv(SAMPLE_SUB)  # columns: ID, Label (placeholder)\n    tmp2 = sub[\"ID\"].str.rsplit(\"_\", n=1, expand=True)\n    test_ids = tmp2[0].unique().tolist()\n\n    random.shuffle(test_ids)\n    selected_test = test_ids[:N_TEST_IMAGES]\n\n    pd.DataFrame({\"image_id\": selected_test}).to_csv(OUT_DIR / \"test_selected_image_ids.csv\", index=False)\n\n    missing_t = 0\n    copied_t = 0\n    for iid in tqdm(selected_test, desc=\"Copying TEST DICOMs\"):\n        src = TEST_DICOM_DIR / f\"{iid}.dcm\"\n        dst = TEST_OUT / f\"{iid}.dcm\"\n        if src.exists():\n            shutil.copy2(src, dst)\n            copied_t += 1\n        else:\n            missing_t += 1\n\n    print(\"TEST copied:\", copied_t, \"missing:\", missing_t)\n    print(\"Files in test_images:\", len(list(TEST_OUT.glob(\"*.dcm\"))))\nelse:\n    print(\"Skipping test subset (N_TEST_IMAGES=0).\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-02T06:05:47.321159Z","iopub.execute_input":"2026-01-02T06:05:47.322468Z","iopub.status.idle":"2026-01-02T06:05:47.334382Z","shell.execute_reply.started":"2026-01-02T06:05:47.322439Z","shell.execute_reply":"2026-01-02T06:05:47.333104Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Subset folder sizes:\")\n!du -sh /kaggle/working/rsna_subset_15k\n!du -sh /kaggle/working/rsna_subset_15k/train_images\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-02T06:06:14.540148Z","iopub.execute_input":"2026-01-02T06:06:14.540406Z","iopub.status.idle":"2026-01-02T06:06:15.010215Z","shell.execute_reply.started":"2026-01-02T06:06:14.540392Z","shell.execute_reply":"2026-01-02T06:06:15.008857Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"archive = Path(\"/kaggle/working/rsna_subset_15k.tar.gz\")\nif archive.exists():\n    archive.unlink()\n\n# Create tar.gz archive\nsubprocess.check_call([\"tar\", \"-czf\", str(archive), \"-C\", \"/kaggle/working\", \"rsna_subset_15k\"])\nprint(\"Archive created:\", archive, \"GB:\", round(archive.stat().st_size / 1e9, 2))\n\n# Remove old parts\nfor f in Path(\"/kaggle/working\").glob(\"rsna_subset_15k.tar.gz.part_*\"):\n    f.unlink()\n\n# Split into 4.5GB chunks\nsubprocess.check_call([\"split\", \"-b\", \"4500m\", str(archive), \"/kaggle/working/rsna_subset_15k.tar.gz.part_\"])\n\nprint(\"Parts created:\")\n!ls -lh /kaggle/working | grep \"rsna_subset_15k.tar.gz.part_\" || true\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-02T06:06:55.029712Z","iopub.execute_input":"2026-01-02T06:06:55.030063Z","iopub.status.idle":"2026-01-02T06:15:15.518129Z","shell.execute_reply.started":"2026-01-02T06:06:55.030043Z","shell.execute_reply":"2026-01-02T06:15:15.516179Z"}},"outputs":[],"execution_count":null}]}