{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"\"\"\"\nRSNA Intracranial Hemorrhage Detection — Preprocessing (CAPPED VERSION)\n==========================================================================\nSame as before, but processes a CAPPED, stratified subset instead of\nall 752,803 images — sized to safely fit under Kaggle's 20GB output\nquota, WITH room left over for training outputs / model checkpoints.\n\nWhy capped: the uncapped run hit the 20GB output wall at ~95%\ncompletion, causing a memory-pressure crash near the end, and only\n~448,789 images actually made it to disk before the wall was hit\n(the rest silently failed to write once the disk was full). Capping\nupfront avoids all of that — you decide the size, not the platform.\n\nSubset composition:\n- ALL positive (\"any\"=1) patients — roughly ~108k, based on the\n  dataset's ~14.4% overall positive rate. Never subsample the rare\n  positive class.\n- Random negatives to fill the remainder up to TARGET_TOTAL_IMAGES.\n\nAt an observed average of ~35-45KB per processed image (varies with\nhead size / crop), 300,000 images should land around 10-14GB —\ncomfortably under the 20GB limit with meaningful headroom for\ncheckpoints (~100-300MB for ResNet-50/ViT-B/16) and CSVs.\n\nSame fixes as before, carried forward:\n- Consistent windowing (clip and rescale use the same width/level).\n- 3-channel brain/subdural/bone stack.\n- Head cropping.\n- Multiprocessing with maxtasksperchild (memory-leak fix).\n- Atomic writes with explicit format=\"PNG\".\n- Resumable — skips files that already exist.\n\"\"\"\n\nimport os\nimport multiprocessing as mp\nfrom functools import partial\n\nimport numpy as np\nimport pandas as pd\nimport pydicom\nfrom PIL import Image\nfrom scipy import ndimage\nfrom tqdm import tqdm\n\n# ------------------------------------------------------------------\n# CONFIG\n# ------------------------------------------------------------------\nDATA_ROOT = \"/kaggle/input/competitions/rsna-intracranial-hemorrhage-detection/rsna-intracranial-hemorrhage-detection\"\nTRAIN_CSV = os.path.join(DATA_ROOT, \"stage_2_train.csv\")\nTRAIN_IMG_DIR = os.path.join(DATA_ROOT, \"stage_2_train\")\nOUTPUT_DIR = \"/kaggle/working/png/train_brain_subdural_bone\"\n\nNUM_WORKERS = max(mp.cpu_count() - 1, 1)\nTARGET_TOTAL_IMAGES = 300_000\nRANDOM_SEED = 42\n\nWINDOWS = {\n    \"brain\": (80, 40),\n    \"subdural\": (200, 80),\n    \"bone\": (2000, 600),\n}\n\n\ndef linear_windowing(img_hu, width, level):\n    lower = level - width / 2\n    upper = level + width / 2\n    img = np.clip(img_hu, lower, upper)\n    img = (img - lower) / (upper - lower)\n    return (img * 255).astype(np.uint8)\n\n\ndef dicom_to_hu(dcm):\n    pixels = dcm.pixel_array.astype(np.float32)\n    pixels = pixels * dcm.RescaleSlope + dcm.RescaleIntercept\n    return pixels\n\n\nclass CropHead:\n    def __init__(self, offset=10):\n        self.offset = offset\n\n    def crop_extents(self, img_bool):\n        try:\n            labeled, n = ndimage.label(img_bool)\n            if n == 0:\n                return 0, img_bool.shape[0], 0, img_bool.shape[1]\n            sizes = np.bincount(labeled.flatten())\n            sizes[0] = 0\n            head_label = np.argmax(sizes)\n            head_mask = labeled == head_label\n            rows = np.flatnonzero(head_mask.any(axis=1))\n            cols = np.flatnonzero(head_mask.any(axis=0))\n            x_min = max(rows.min() - self.offset, 0)\n            x_max = min(rows.max() + self.offset + 1, img_bool.shape[0])\n            y_min = max(cols.min() - self.offset, 0)\n            y_max = min(cols.max() + self.offset + 1, img_bool.shape[1])\n            return x_min, x_max, y_min, y_max\n        except ValueError:\n            return 0, img_bool.shape[0], 0, img_bool.shape[1]\n\n    def __call__(self, img):\n        x_min, x_max, y_min, y_max = self.crop_extents(img > 0)\n        return img[x_min:x_max, y_min:y_max]\n\n\ndef process_one_dicom(path, crop_head):\n    dcm = pydicom.dcmread(path)\n    hu = dicom_to_hu(dcm)\n    channels = []\n    for name in (\"brain\", \"subdural\", \"bone\"):\n        width, level = WINDOWS[name]\n        channels.append(linear_windowing(hu, width, level))\n    img = np.stack(channels, axis=-1)\n    img = crop_head(img)\n    if img.shape[0] == 0 or img.shape[1] == 0:\n        img = np.zeros((512, 512, 3), dtype=np.uint8)\n    return img\n\n\ndef _worker(fname, img_dir, output_dir):\n    out_name = fname.replace(\".dcm\", \".png\")\n    out_path = os.path.join(output_dir, out_name)\n    if os.path.exists(out_path):\n        return \"skipped\"\n    tmp_path = out_path + f\".tmp{os.getpid()}\"\n    try:\n        crop_head = CropHead()\n        img = process_one_dicom(os.path.join(img_dir, fname), crop_head)\n        Image.fromarray(img).save(tmp_path, format=\"PNG\")\n        os.replace(tmp_path, out_path)\n        return None\n    except Exception as e:\n        if os.path.exists(tmp_path):\n            try:\n                os.remove(tmp_path)\n            except OSError:\n                pass\n        return (fname, str(e))\n\n\ndef select_capped_subset(train_csv, target_total, seed=RANDOM_SEED):\n    \"\"\"\n    Selects ALL positive patients + a random sample of negatives,\n    up to target_total images total. Returns a set of image IDs\n    (without extension) to process.\n    \"\"\"\n    y = pd.read_csv(train_csv)\n    id_split = y.ID.str.rsplit(\"_\", n=1, expand=True)\n    y = pd.concat([id_split, y.Label], axis=1)\n    y.columns = [\"id\", \"sub_type\", \"label\"]\n    y = y.drop_duplicates(subset=[\"id\", \"sub_type\"])\n\n    df = y.pivot(index=\"id\", columns=\"sub_type\", values=\"label\")\n\n    positives = df[df[\"any\"] == 1]\n    negatives = df[df[\"any\"] == 0]\n\n    n_pos = len(positives)\n    n_neg_needed = max(target_total - n_pos, 0)\n\n    print(f\"Full dataset: {len(df)} patients, {n_pos} positive \"\n          f\"({n_pos/len(df):.1%})\")\n\n    if n_neg_needed > len(negatives):\n        print(f\"WARNING: target_total ({target_total}) with all \"\n              f\"positives ({n_pos}) needs {n_neg_needed} negatives, \"\n              f\"but only {len(negatives)} exist. Using all negatives instead.\")\n        n_neg_needed = len(negatives)\n\n    sampled_negatives = negatives.sample(n=n_neg_needed, random_state=seed)\n    subset_ids = set(positives.index) | set(sampled_negatives.index)\n\n    print(f\"Capped subset: {len(subset_ids)} patients \"\n          f\"({n_pos} positive + {n_neg_needed} negative)\")\n\n    return subset_ids\n\n\ndef main():\n    os.makedirs(OUTPUT_DIR, exist_ok=True)\n\n    print(\"Selecting capped, stratified subset...\")\n    subset_ids = select_capped_subset(TRAIN_CSV, TARGET_TOTAL_IMAGES)\n\n    all_filenames = os.listdir(TRAIN_IMG_DIR)\n    # Filter to only the files in our selected subset\n    filenames = [f for f in all_filenames if f.replace(\".dcm\", \"\") in subset_ids]\n\n    print(f\"Found {len(all_filenames)} total DICOM files; \"\n          f\"processing {len(filenames)} in the capped subset.\")\n    print(f\"Estimated output size at ~40KB/image: \"\n          f\"~{len(filenames) * 40 / 1e6:.1f} GB\")\n    print(f\"Using {NUM_WORKERS} worker processes.\")\n    print(f\"Output: {OUTPUT_DIR}\")\n\n    worker_fn = partial(_worker, img_dir=TRAIN_IMG_DIR, output_dir=OUTPUT_DIR)\n\n    failed = []\n    skipped = 0\n    processed = 0\n\n    # maxtasksperchild recycles workers periodically to prevent the\n    # memory leak that crashed the previous uncapped run near 95%.\n    with mp.Pool(NUM_WORKERS, maxtasksperchild=2000) as pool:\n        for result in tqdm(pool.imap_unordered(worker_fn, filenames, chunksize=64),\n                            total=len(filenames)):\n            if result == \"skipped\":\n                skipped += 1\n            elif result is None:\n                processed += 1\n            else:\n                failed.append(result)\n\n    print(f\"\\nNewly processed: {processed}\")\n    print(f\"Skipped (already done): {skipped}\")\n    if failed:\n        print(f\"{len(failed)} files failed. First few:\")\n        for f, e in failed[:10]:\n            print(f\"  {f}: {e}\")\n\n    print(\"Done.\")\n\n\nif __name__ == \"__main__\":\n    pass  # main() commented out — run the smoke test below first, then call main() separately\n    # main()\n\n# ============================================================\n# SMOKE TEST — 50 files from the capped subset, runs immediately\n# ============================================================\n_subset_ids_test = select_capped_subset(TRAIN_CSV, TARGET_TOTAL_IMAGES)\n_all_filenames_test = os.listdir(TRAIN_IMG_DIR)\ntest_files = [f for f in _all_filenames_test if f.replace(\".dcm\", \"\") in _subset_ids_test][:50]\n\ntest_output_dir = \"/kaggle/working/smoke_test_png\"\nos.makedirs(test_output_dir, exist_ok=True)\n\ncrop_head = CropHead()\nsuccess, failed = 0, []\n\nfor fname in test_files:\n    try:\n        img = process_one_dicom(os.path.join(TRAIN_IMG_DIR, fname), crop_head)\n        out_path = os.path.join(test_output_dir, fname.replace(\".dcm\", \".png\"))\n        Image.fromarray(img).save(out_path, format=\"PNG\")\n        success += 1\n    except Exception as e:\n        failed.append((fname, str(e)))\n\nprint(f\"\\nSmoke test — Success: {success}/50\")\nprint(f\"Smoke test — Failed: {len(failed)}\")\nfor f, e in failed[:5]:\n    print(f\"  {f}: {e}\")\n\nfor f in os.listdir(test_output_dir)[:5]:\n    path = os.path.join(test_output_dir, f)\n    print(f, os.path.getsize(path), \"bytes\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-07-13T14:15:04.948972Z","iopub.execute_input":"2026-07-13T14:15:04.949399Z","iopub.status.idle":"2026-07-13T14:15:37.196291Z","shell.execute_reply.started":"2026-07-13T14:15:04.949368Z","shell.execute_reply":"2026-07-13T14:15:37.195532Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"main()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-13T14:18:54.588433Z","iopub.execute_input":"2026-07-13T14:18:54.589133Z","iopub.status.idle":"2026-07-13T16:53:36.057714Z","shell.execute_reply.started":"2026-07-13T14:18:54.589097Z","shell.execute_reply":"2026-07-13T16:53:36.056984Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nout = \"/kaggle/working/png/train_brain_subdural_bone\"\n\nfiles = os.listdir(out)\nprint(\"PNG count:\", len(files))\nprint(\"First 5:\", files[:5])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-13T16:53:48.331988Z","iopub.execute_input":"2026-07-13T16:53:48.332281Z","iopub.status.idle":"2026-07-13T16:53:48.482111Z","shell.execute_reply.started":"2026-07-13T16:53:48.332253Z","shell.execute_reply":"2026-07-13T16:53:48.481471Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from PIL import Image\nimport random\nimport matplotlib.pyplot as plt\n\nsample = random.sample(files, 5)\n\nfor f in sample:\n    img = Image.open(os.path.join(out, f))\n    plt.figure(figsize=(4,4))\n    plt.imshow(img)\n    plt.title(f)\n    plt.axis(\"off\")\n    plt.show()\n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-13T16:54:41.141679Z","iopub.execute_input":"2026-07-13T16:54:41.142421Z","iopub.status.idle":"2026-07-13T16:54:41.64886Z","shell.execute_reply.started":"2026-07-13T16:54:41.142388Z","shell.execute_reply":"2026-07-13T16:54:41.648095Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from PIL import Image\nimport numpy as np\n\nimg = Image.open(\"/kaggle/working/png/train_brain_subdural_bone/ID_866d13178.png\")\nprint(np.array(img).shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-13T16:56:11.052161Z","iopub.execute_input":"2026-07-13T16:56:11.052768Z","iopub.status.idle":"2026-07-13T16:56:11.060313Z","shell.execute_reply.started":"2026-07-13T16:56:11.052738Z","shell.execute_reply":"2026-07-13T16:56:11.05968Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import shutil\n\nusage = shutil.disk_usage(\"/kaggle/working\")\n\nprint(f\"Used: {usage.used/1e9:.2f} GB\")\nprint(f\"Free: {usage.free/1e9:.2f} GB\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-13T16:57:52.380526Z","iopub.execute_input":"2026-07-13T16:57:52.381099Z","iopub.status.idle":"2026-07-13T16:57:52.386953Z","shell.execute_reply.started":"2026-07-13T16:57:52.38107Z","shell.execute_reply":"2026-07-13T16:57:52.385472Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nzip_path = \"/kaggle/working/rsna_png.zip\"\n\nif os.path.exists(zip_path):\n    size = os.path.getsize(zip_path) / (1024**3)\n    print(f\"Current ZIP size: {size:.2f} GB\")\nelse:\n    print(\"ZIP file hasn't been created yet.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-13T17:03:40.799296Z","iopub.execute_input":"2026-07-13T17:03:40.800103Z","iopub.status.idle":"2026-07-13T17:03:40.80472Z","shell.execute_reply.started":"2026-07-13T17:03:40.800057Z","shell.execute_reply":"2026-07-13T17:03:40.803949Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import shutil\n\nusage = shutil.disk_usage(\"/kaggle/working\")\nprint(f\"Used: {usage.used/1e9:.2f} GB\")\nprint(f\"Free: {usage.free/1e9:.2f} GB\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-13T17:04:35.47933Z","iopub.execute_input":"2026-07-13T17:04:35.479829Z","iopub.status.idle":"2026-07-13T17:04:35.484189Z","shell.execute_reply.started":"2026-07-13T17:04:35.479802Z","shell.execute_reply":"2026-07-13T17:04:35.483425Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport zipfile\n\nzip_path = \"/kaggle/working/rsna_png.zip\"\n\nprint(\"Exists:\", os.path.exists(zip_path))\nprint(\"Size (GB):\", os.path.getsize(zip_path)/(1024**3))\n\ntry:\n    with zipfile.ZipFile(zip_path, \"r\") as z:\n        print(\"ZIP is valid\")\n        print(\"Files inside:\", len(z.namelist()))\nexcept Exception as e:\n    print(\"ZIP is NOT valid\")\n    print(e)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-13T17:05:41.533713Z","iopub.execute_input":"2026-07-13T17:05:41.534493Z","iopub.status.idle":"2026-07-13T17:05:42.292417Z","shell.execute_reply.started":"2026-07-13T17:05:41.534464Z","shell.execute_reply":"2026-07-13T17:05:42.291665Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\npng_dir = \"/kaggle/working/png/train_brain_subdural_bone\"\n\nprint(len(os.listdir(png_dir)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-13T17:18:44.653634Z","iopub.execute_input":"2026-07-13T17:18:44.654405Z","iopub.status.idle":"2026-07-13T17:18:44.826774Z","shell.execute_reply.started":"2026-07-13T17:18:44.654374Z","shell.execute_reply":"2026-07-13T17:18:44.82606Z"}},"outputs":[],"execution_count":null}]}