{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":13451,"datasetId":654585,"databundleVersionId":1188070,"isSourceIdPinned":false},{"sourceType":"datasetVersion","sourceId":15817985,"datasetId":10139541,"databundleVersionId":16766523}],"dockerImageVersionId":31328,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# ============================================================\n# Cell 1 — Install dependencies\n# ============================================================\n!pip -q install pydicom SimpleITK pandas numpy tqdm","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-04-19T09:42:03.483549Z","iopub.execute_input":"2026-04-19T09:42:03.484922Z","iopub.status.idle":"2026-04-19T09:42:08.341103Z","shell.execute_reply.started":"2026-04-19T09:42:03.484874Z","shell.execute_reply":"2026-04-19T09:42:08.339368Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# Cell 2 — Imports and configuration\n# ============================================================\nfrom pathlib import Path\nimport os\nimport shutil\nimport json\n\nimport numpy as np\nimport pandas as pd\nimport SimpleITK as sitk\nimport pydicom\nfrom tqdm.auto import tqdm\n\n# Hide SimpleITK / ITK warning messages\nsitk.ProcessObject.GlobalWarningDisplayOff()\n\n# ------------------------------------------------------------\n# Kaggle RSNA DICOM root\n# Competition Link:\n# https://www.kaggle.com/competitions/rsna-intracranial-hemorrhage-detection\n# ------------------------------------------------------------\nRSNA_TRAIN_DIR = Path(\n    \"/kaggle/input/competitions/rsna-intracranial-hemorrhage-detection/rsna-intracranial-hemorrhage-detection/stage_2_train\"\n)\n\n# ------------------------------------------------------------\n# CSV containing only CT-Seg1500 normal RSNA series\n# Required columns:\n#   case_id, SeriesInstanceUID, OrderKey, InstanceNumber, filepath\n# Link:\n# https://www.kaggle.com/datasets/haithamalshari/rsna-derived-normals-with-masks-ct-seg1500\n# ------------------------------------------------------------\nFILTERED_SERIES_CSV = Path(\n    \"/kaggle/input/datasets/haithamalshari/rsna-derived-normals-with-masks-ct-seg1500/rsna_series_grouped_CT-Seg1500_normals.csv\"\n)\n\n# ------------------------------------------------------------\n# Output\n# ------------------------------------------------------------\nOUTPUT_ROOT = Path(\"/kaggle/working/RSNA-Normal-NIfTI\")\nOUT_CT_DIR = OUTPUT_ROOT / \"ct_scans\"\nOUT_MASK_DIR = OUTPUT_ROOT / \"masks\"\nREPORT_DIR = OUTPUT_ROOT / \"reports\"\n\nOUT_CT_DIR.mkdir(parents=True, exist_ok=True)\nOUT_MASK_DIR.mkdir(parents=True, exist_ok=True)\nREPORT_DIR.mkdir(parents=True, exist_ok=True)\n\nCONVERSION_LOG_CSV = REPORT_DIR / \"rsna_normal_nifti_conversion_log.csv\"\nSUMMARY_JSON = REPORT_DIR / \"rsna_normal_nifti_summary.json\"\n\nprint(\"RSNA_TRAIN_DIR      =\", RSNA_TRAIN_DIR)\nprint(\"FILTERED_SERIES_CSV =\", FILTERED_SERIES_CSV)\nprint(\"OUTPUT_ROOT         =\", OUTPUT_ROOT)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T09:42:08.344602Z","iopub.execute_input":"2026-04-19T09:42:08.346148Z","iopub.status.idle":"2026-04-19T09:42:09.413855Z","shell.execute_reply.started":"2026-04-19T09:42:08.346011Z","shell.execute_reply":"2026-04-19T09:42:09.411665Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# Cell 3 — Utility functions\n# ============================================================\ndef series_uid_to_case_id(series_uid: str) -> str:\n    \"\"\"\n    Converts:\n      NONSTD_ID_000a935543 -> 000a935543\n    \"\"\"\n    s = str(series_uid).strip()\n    prefix = \"NONSTD_ID_\"\n    return s[len(prefix):] if s.startswith(prefix) else s\n\n\ndef resolve_dicom_path(filepath: str) -> Path:\n    \"\"\"\n    Resolve DICOM path from the CSV.\n    \n    The filepath may be:\n    - already an absolute Kaggle path\n    - a relative path / filename\n    \"\"\"\n    p = Path(str(filepath))\n\n    if p.exists():\n        return p\n\n    # If CSV stores only filename or relative path, resolve against RSNA_TRAIN_DIR\n    candidate = RSNA_TRAIN_DIR / p.name\n    if candidate.exists():\n        return candidate\n\n    # If CSV stores path containing stage_2_train/... try basename fallback\n    candidate = RSNA_TRAIN_DIR / Path(str(filepath)).name\n    if candidate.exists():\n        return candidate\n\n    raise FileNotFoundError(f\"Could not resolve DICOM path: {filepath}\")\n\n\ndef get_free_space_gb(path=\"/kaggle/working\") -> float:\n    usage = shutil.disk_usage(path)\n    return usage.free / (1024 ** 3)\n\n\ndef make_zero_mask_like(ct_img: sitk.Image) -> sitk.Image:\n    \"\"\"\n    Create a zero-valued uint8 mask with the exact same physical space\n    as the CT image.\n    \"\"\"\n    mask = sitk.Image(ct_img.GetSize(), sitk.sitkUInt8)\n    mask.CopyInformation(ct_img)\n    return mask\n\n\ndef write_nifti_pair(ct_img: sitk.Image, case_id: str):\n    \"\"\"\n    Save CT and zero mask as .nii.gz.\n    \"\"\"\n    ct_out = OUT_CT_DIR / f\"{case_id}.nii.gz\"\n    mask_out = OUT_MASK_DIR / f\"{case_id}.nii.gz\"\n\n    zero_mask = make_zero_mask_like(ct_img)\n\n    sitk.WriteImage(ct_img, str(ct_out), True)\n    sitk.WriteImage(zero_mask, str(mask_out), True)\n\n    return ct_out, mask_out\n\n\ndef append_log_row(row: dict, csv_path: Path):\n    df = pd.DataFrame([row])\n    header = not csv_path.exists()\n    df.to_csv(csv_path, mode=\"a\", header=header, index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T09:42:09.415913Z","iopub.execute_input":"2026-04-19T09:42:09.416929Z","iopub.status.idle":"2026-04-19T09:42:09.436615Z","shell.execute_reply.started":"2026-04-19T09:42:09.416876Z","shell.execute_reply":"2026-04-19T09:42:09.434112Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# Cell 4 — DICOM sorting helpers\n# ============================================================\nDICOM_HEADER_TAGS = [\n    (0x0008, 0x0008),  # ImageType\n    (0x0020, 0x0032),  # ImagePositionPatient\n    (0x0020, 0x0037),  # ImageOrientationPatient\n    (0x0020, 0x0013),  # InstanceNumber\n    (0x0018, 0x1120),  # Gantry/Detector Tilt\n]\n\nMIN_SLICES = 20\nMAX_Z_EXTENT_MM = 300.0\nMAX_Z_JUMP_MM = 15.0\nIOP_DOT_TOL = 0.995\nGANTRY_TILT_EPS = 0.01\n\n\ndef _as_float_list(x):\n    try:\n        return [float(v) for v in x]\n    except Exception:\n        return None\n\n\ndef _safe_get(ds, tag):\n    if tag in ds:\n        return ds.get(tag).value\n    return None\n\n\ndef read_dicom_header(path: Path):\n    return pydicom.dcmread(\n        str(path),\n        stop_before_pixels=True,\n        specific_tags=[pydicom.tag.Tag(g, e) for g, e in DICOM_HEADER_TAGS],\n        force=True,\n    )\n\n\ndef qc_and_sort_dicoms(filepaths):\n    \"\"\"\n    Sort slices using ImagePositionPatient + ImageOrientationPatient.\n    \n    Returns:\n      sorted_paths, qc_info\n    \n    If rejected:\n      None, qc_info\n    \"\"\"\n    infos = []\n\n    for fp in filepaths:\n        try:\n            ds = read_dicom_header(fp)\n        except Exception as e:\n            return None, {\"reason\": f\"header_read_fail:{repr(e)}\"}\n\n        img_type = _safe_get(ds, (0x0008, 0x0008))\n        if img_type is not None:\n            if isinstance(img_type, (list, tuple, pydicom.multival.MultiValue)):\n                img_type_vals = [str(v).upper() for v in img_type]\n            else:\n                img_type_vals = [str(img_type).upper()]\n        else:\n            img_type_vals = []\n\n        if any(\"LOCALIZER\" in v or \"SCOUT\" in v for v in img_type_vals):\n            return None, {\"reason\": \"localizer_or_scout\"}\n\n        tilt = _safe_get(ds, (0x0018, 0x1120))\n        try:\n            tilt_f = float(tilt) if tilt is not None else 0.0\n        except Exception:\n            tilt_f = 0.0\n\n        if abs(tilt_f) > GANTRY_TILT_EPS:\n            return None, {\"reason\": f\"gantry_tilt:{tilt_f:.3f}\"}\n\n        ipp = _as_float_list(_safe_get(ds, (0x0020, 0x0032)))\n        iop = _as_float_list(_safe_get(ds, (0x0020, 0x0037)))\n        inst = _safe_get(ds, (0x0020, 0x0013))\n\n        if ipp is None or iop is None or len(ipp) != 3 or len(iop) != 6:\n            return None, {\"reason\": \"missing_ipp_or_iop\"}\n\n        row = np.array(iop[:3], dtype=np.float64)\n        col = np.array(iop[3:], dtype=np.float64)\n\n        row = row / (np.linalg.norm(row) + 1e-12)\n        col = col / (np.linalg.norm(col) + 1e-12)\n\n        normal = np.cross(row, col)\n        normal = normal / (np.linalg.norm(normal) + 1e-12)\n\n        pos = np.array(ipp, dtype=np.float64)\n        z = float(np.dot(pos, normal))\n\n        infos.append({\n            \"fp\": fp,\n            \"row\": row,\n            \"col\": col,\n            \"normal\": normal,\n            \"z\": z,\n            \"inst\": int(inst) if inst is not None and str(inst).isdigit() else None,\n        })\n\n    if len(infos) < MIN_SLICES:\n        return None, {\"reason\": f\"too_few_slices:{len(infos)}\"}\n\n    ref_row = infos[0][\"row\"]\n    ref_col = infos[0][\"col\"]\n\n    for it in infos[1:]:\n        if abs(float(np.dot(ref_row, it[\"row\"]))) < IOP_DOT_TOL:\n            return None, {\"reason\": \"mixed_row_orientation\"}\n        if abs(float(np.dot(ref_col, it[\"col\"]))) < IOP_DOT_TOL:\n            return None, {\"reason\": \"mixed_col_orientation\"}\n\n    infos_sorted = sorted(infos, key=lambda d: d[\"z\"])\n    z_vals = np.array([d[\"z\"] for d in infos_sorted], dtype=np.float64)\n\n    z_extent = float(z_vals[-1] - z_vals[0])\n    if z_extent > MAX_Z_EXTENT_MM:\n        return None, {\"reason\": f\"z_extent_too_large:{z_extent:.1f}\"}\n\n    dz = np.diff(z_vals)\n    max_jump = float(np.max(np.abs(dz))) if len(dz) > 0 else 0.0\n\n    if max_jump > MAX_Z_JUMP_MM:\n        return None, {\"reason\": f\"z_jump_too_large:{max_jump:.1f}\"}\n\n    sorted_paths = [d[\"fp\"] for d in infos_sorted]\n\n    qc_info = {\n        \"reason\": \"ok\",\n        \"n_slices\": len(sorted_paths),\n        \"z_extent_mm\": z_extent,\n        \"max_z_jump_mm\": max_jump,\n    }\n\n    return sorted_paths, qc_info","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T09:42:09.439036Z","iopub.execute_input":"2026-04-19T09:42:09.440407Z","iopub.status.idle":"2026-04-19T09:42:09.481249Z","shell.execute_reply.started":"2026-04-19T09:42:09.440354Z","shell.execute_reply":"2026-04-19T09:42:09.479802Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# Cell 5 — Convert one series\n# ============================================================\ndef convert_sorted_dicom_series(sorted_paths):\n    \"\"\"\n    Convert sorted DICOM file list to SimpleITK image.\n    \"\"\"\n    reader = sitk.ImageSeriesReader()\n    reader.SetFileNames([str(p) for p in sorted_paths])\n    reader.MetaDataDictionaryArrayUpdateOn()\n    reader.LoadPrivateTagsOn()\n    img = reader.Execute()\n    return img","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T09:42:09.484826Z","iopub.execute_input":"2026-04-19T09:42:09.48528Z","iopub.status.idle":"2026-04-19T09:42:09.515107Z","shell.execute_reply.started":"2026-04-19T09:42:09.485241Z","shell.execute_reply":"2026-04-19T09:42:09.513666Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# Cell 6 — Load filtered normal CSV\n# ============================================================\nif not FILTERED_SERIES_CSV.exists():\n    raise FileNotFoundError(f\"Missing filtered CSV: {FILTERED_SERIES_CSV}\")\n\ndf = pd.read_csv(FILTERED_SERIES_CSV)\n\nrequired_cols = [\"case_id\", \"SeriesInstanceUID\", \"filepath\"]\nfor c in required_cols:\n    if c not in df.columns:\n        raise ValueError(f\"Missing required column in filtered CSV: {c}\")\n\ndf[\"case_id\"] = df[\"case_id\"].astype(str)\ndf[\"SeriesInstanceUID\"] = df[\"SeriesInstanceUID\"].astype(str)\ndf[\"filepath\"] = df[\"filepath\"].astype(str)\n\n# Ensure case_id exists even if generated CSV did not preserve it correctly\ndf[\"case_id\"] = df[\"case_id\"].fillna(df[\"SeriesInstanceUID\"].map(series_uid_to_case_id))\n\nseries_table = (\n    df.groupby([\"case_id\", \"SeriesInstanceUID\"], sort=False)\n    .agg(\n        n_slices_csv=(\"filepath\", \"count\")\n    )\n    .reset_index()\n)\n\nprint(\"Slice rows in CSV:\", len(df))\nprint(\"Series to convert:\", len(series_table))\nprint(series_table.head(10).to_string(index=False))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T09:42:09.517194Z","iopub.execute_input":"2026-04-19T09:42:09.51767Z","iopub.status.idle":"2026-04-19T09:42:09.659511Z","shell.execute_reply.started":"2026-04-19T09:42:09.517617Z","shell.execute_reply":"2026-04-19T09:42:09.65819Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# Cell 7 — Convert selected RSNA normals to NIfTI + zero masks\n# ============================================================\nMIN_DISK_FREE_GB = 5.0\n\nconversion_rows = []\n\nprint(\"Starting conversion...\")\nprint(f\"Stopping if free space < {MIN_DISK_FREE_GB:.1f} GB\")\nprint(\"Initial free space GB:\", round(get_free_space_gb(\"/kaggle/working\"), 2))\n\ngrouped = df.groupby([\"case_id\", \"SeriesInstanceUID\"], sort=False)\n\nfor (case_id, series_uid), g in tqdm(grouped, desc=\"Converting RSNA normals\", unit=\"series\"):\n    free_gb = get_free_space_gb(\"/kaggle/working\")\n    if free_gb < MIN_DISK_FREE_GB:\n        conversion_rows.append({\n            \"case_id\": case_id,\n            \"SeriesInstanceUID\": series_uid,\n            \"status\": \"stopped_low_storage\",\n            \"reason\": f\"free_space_gb={free_gb:.2f}\",\n            \"n_slices_csv\": len(g),\n            \"ct_out\": None,\n            \"mask_out\": None,\n        })\n        print(f\"\\n[STOP] Low storage: {free_gb:.2f} GB\")\n        break\n\n    ct_out = OUT_CT_DIR / f\"{case_id}.nii.gz\"\n    mask_out = OUT_MASK_DIR / f\"{case_id}.nii.gz\"\n\n    # Resume behavior: skip already converted complete pair\n    if ct_out.exists() and mask_out.exists():\n        conversion_rows.append({\n            \"case_id\": case_id,\n            \"SeriesInstanceUID\": series_uid,\n            \"status\": \"skipped_exists\",\n            \"reason\": \"ct_and_mask_exist\",\n            \"n_slices_csv\": len(g),\n            \"ct_out\": str(ct_out),\n            \"mask_out\": str(mask_out),\n        })\n        continue\n\n    try:\n        dicom_paths = [resolve_dicom_path(p) for p in g[\"filepath\"].tolist()]\n\n        sorted_paths, qc = qc_and_sort_dicoms(dicom_paths)\n        if sorted_paths is None:\n            conversion_rows.append({\n                \"case_id\": case_id,\n                \"SeriesInstanceUID\": series_uid,\n                \"status\": \"skipped_qc\",\n                \"reason\": qc.get(\"reason\", \"qc_reject\"),\n                \"n_slices_csv\": len(g),\n                \"ct_out\": None,\n                \"mask_out\": None,\n            })\n            continue\n\n        ct_img = convert_sorted_dicom_series(sorted_paths)\n        ct_out, mask_out = write_nifti_pair(ct_img, case_id)\n\n        conversion_rows.append({\n            \"case_id\": case_id,\n            \"SeriesInstanceUID\": series_uid,\n            \"status\": \"ok\",\n            \"reason\": \"converted\",\n            \"n_slices_csv\": len(g),\n            \"n_slices_used\": len(sorted_paths),\n            \"z_extent_mm\": qc.get(\"z_extent_mm\"),\n            \"max_z_jump_mm\": qc.get(\"max_z_jump_mm\"),\n            \"ct_out\": str(ct_out),\n            \"mask_out\": str(mask_out),\n        })\n\n    except Exception as e:\n        conversion_rows.append({\n            \"case_id\": case_id,\n            \"SeriesInstanceUID\": series_uid,\n            \"status\": \"failed\",\n            \"reason\": repr(e),\n            \"n_slices_csv\": len(g),\n            \"ct_out\": None,\n            \"mask_out\": None,\n        })\n\nconversion_log = pd.DataFrame(conversion_rows)\nconversion_log.to_csv(CONVERSION_LOG_CSV, index=False)\n\nprint(\"\\nSaved conversion log:\", CONVERSION_LOG_CSV)\nprint(conversion_log[\"status\"].value_counts(dropna=False).to_string())\nprint(\"Final free space GB:\", round(get_free_space_gb(\"/kaggle/working\"), 2))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T09:42:09.661312Z","iopub.execute_input":"2026-04-19T09:42:09.662305Z","iopub.status.idle":"2026-04-19T09:59:33.021786Z","shell.execute_reply.started":"2026-04-19T09:42:09.66226Z","shell.execute_reply":"2026-04-19T09:59:33.019849Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# Cell 8 — Final audit and summary\n# ============================================================\ndef count_niigz(folder: Path):\n    return len(list(folder.glob(\"*.nii.gz\")))\n\nsummary = {\n    \"input_series\": int(series_table[\"SeriesInstanceUID\"].nunique()),\n    \"conversion_status_counts\": conversion_log[\"status\"].value_counts(dropna=False).to_dict(),\n    \"ct_files\": int(count_niigz(OUT_CT_DIR)),\n    \"mask_files\": int(count_niigz(OUT_MASK_DIR)),\n    \"output_root\": str(OUTPUT_ROOT),\n    \"ct_dir\": str(OUT_CT_DIR),\n    \"mask_dir\": str(OUT_MASK_DIR),\n}\n\nwith open(SUMMARY_JSON, \"w\") as f:\n    json.dump(summary, f, indent=2)\n\nprint(json.dumps(summary, indent=2))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T09:59:33.026276Z","iopub.execute_input":"2026-04-19T09:59:33.026987Z","iopub.status.idle":"2026-04-19T09:59:33.049292Z","shell.execute_reply.started":"2026-04-19T09:59:33.026939Z","shell.execute_reply":"2026-04-19T09:59:33.047984Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# Cell 9 — Optional zip output\n# ============================================================\nMAKE_ZIP = True\n\nif MAKE_ZIP:\n    zip_path = Path(\"/kaggle/working/RSNA-Normal-NIfTI.zip\")\n    if zip_path.exists():\n        zip_path.unlink()\n\n    subprocess_cmd = f\"cd '{OUTPUT_ROOT.parent}' && zip -qr '{zip_path.name}' '{OUTPUT_ROOT.name}'\"\n    print(\"Running:\", subprocess_cmd)\n\n    import subprocess\n    subprocess.check_call([\"bash\", \"-lc\", subprocess_cmd])\n\n    print(\"Created:\", zip_path)\n    print(\"Zip size GB:\", round(zip_path.stat().st_size / (1024**3), 3))\nelse:\n    print(\"MAKE_ZIP=False; skipping zip.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T09:59:33.050764Z","iopub.execute_input":"2026-04-19T09:59:33.051172Z","iopub.status.idle":"2026-04-19T10:03:45.246277Z","shell.execute_reply.started":"2026-04-19T09:59:33.051132Z","shell.execute_reply":"2026-04-19T10:03:45.244796Z"}},"outputs":[],"execution_count":null}]}