{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":99552,"databundleVersionId":13441085,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# RSNA Intracranial Aneurysm Detection — Exploration avancée","metadata":{}},{"cell_type":"code","source":"\n# 0) Imports & setup\nimport os, math, random, ast, json, warnings\nimport numpy as np\nimport pandas as pd\nimport ast, json, matplotlib.pyplot as plt\nimport pydicom\nimport warnings\nwarnings.filterwarnings(\"ignore\", category=FutureWarning)\nimport seaborn as sns\nsns.set_context(\"notebook\")\nfrom IPython.display import display, Markdown as md\nimport textwrap\n\ndisplay(md(\"## 🛠️ Import and setup\"))\n\ndisplay(md(\"\"\"In this first part, we import all the necessary libraries:\n\n- **os, math, random, ast, json, warnings**: file management, math utilities, and warnings  \n- **numpy, pandas**: data manipulation  \n- **matplotlib, seaborn**: visualizations  \n- **pydicom**: reading medical DICOM images  \n- **IPython.display**: for Markdown-formatted display in the notebook  \n\nWe also configure **pandas** for a more readable display (number of columns, column width, etc.).  \n\nFinally, two small helper functions are added:  \n- `short_uid()`: shortens very long identifiers (SeriesInstanceUID).  \n- `hr()`: displays a Markdown separator line (`---`) to structure notebook outputs.  \n\"\"\"))\n\n\n\n# ---------- Pandas display settings ----------\npd.set_option(\"display.max_columns\", 100)\npd.set_option(\"display.width\", 160)\npd.set_option(\"display.max_colwidth\", 60)\n\n# ---------- Small helpers ----------\ndef short_uid(x, n=28):\n    \"\"\"Shorten long SeriesInstanceUID for readability\"\"\"\n    x = str(x)\n    return x if len(x) <= n else x[:n] + \"…\"\n\ndef hr():\n    display(md(\"---\"))\n\nhr()\n\n\n# ---------- 1) Dataset folder content ----------\ndisplay(md(\"## 📁 Dataset folder content\"))\n\nDATA_DIR = \"/kaggle/input/rsna-intracranial-aneurysm-detection\"\nSERIES_DIR = os.path.join(DATA_DIR, \"series\")\n\nprint(\"Dataset directory set to:\", DATA_DIR)\nprint(\"Series directory set to:\", SERIES_DIR)\n\n\nif os.path.isdir(DATA_DIR):\n    items = sorted(os.listdir(DATA_DIR))\n    display(md(\"\\n\".join(f\"- `{it}`\" for it in items)))\nelse:\n    display(md(\"**⚠️ DATA_DIR not found. Did you add the competition dataset?**\"))\n\nhr()\n\n# ---------- 2) General dataset info ----------\ndisplay(md(\"## 📊 General dataset info\"))\n\ndisplay(md(\"\"\" This section displays general information about the dataset: the size of the train.csv and train_localizers.csv files, \nas well as the complete list of columns in train.csv, allowing you to check the structure and content of the data before any analysis.\"\"\"))\n\ntrain_df = pd.read_csv(os.path.join(DATA_DIR, \"train.csv\"))\nloc_df   = pd.read_csv(os.path.join(DATA_DIR, \"train_localizers.csv\"))\n\nprint(\"train.csv shape:\", train_df.shape)\nprint(\"train_localizers.csv shape:\", loc_df.shape)\n\ndisplay(md(f\"- **Shape of `train.csv`**: `{train_df.shape}`\"))\ndisplay(md(f\"- **Columns (`train.csv`)** ({len(train_df.columns)}):\"))\ndisplay(md(\"`\" + \"`, `\".join(train_df.columns.tolist()) + \"`\"))\n\nhr()\n\n# ---------- 3) Preview of key columns ----------\ndisplay(md(\"## 👀 Preview of key columns\"))\ndisplay(md(\"\"\"This section provides a quick preview of the key columns (`PatientAge`, `PatientSex`, `Modality`, and `Aneurysm Present`) \nwith shortened series identifiers, to easily inspect a few representative rows of the dataset.\"\"\"))\n\nbase_cols = [\"SeriesInstanceUID\",\"PatientAge\",\"PatientSex\",\"Modality\",\"Aneurysm Present\"]\npreview = train_df[base_cols].head(8).copy()\npreview.insert(1, \"SeriesUID_short\", preview[\"SeriesInstanceUID\"].map(short_uid))\npreview = preview.drop(columns=[\"SeriesInstanceUID\"])\ndisplay(preview.style.set_properties(**{\"text-align\": \"left\"}))\n\nhr()\n\n# ---------- 4) DICOM quick look ----------\ndisplay(md(\"## 🖼️ DICOM quick look\"))\n\ndisplay(md(\"\"\"The dataset contains 188 DICOM series, each identified by a unique `SeriesInstanceUID` (e.g., `1.2.826.0.1.3680043.8.498.10004044428023505108375152878107656647`). Each series is composed of multiple images, with the number of slices varying from one series to another.\n\"\"\"))\nif os.path.isdir(SERIES_DIR):\n    all_series = sorted(os.listdir(SERIES_DIR))\n    if len(all_series) == 0:\n        display(md(\"**⚠️ No series found in `series/`.**\"))\n    else:\n        example_series = all_series[0]\n        series_path = os.path.join(SERIES_DIR, example_series)\n        dicom_files = [f for f in os.listdir(series_path) if f.lower().endswith(\".dcm\")]\n        if not dicom_files:\n            dicom_files = os.listdir(series_path)  # fallback\n\n        display(md(f\"- **Example series**: `{short_uid(example_series, 40)}`\"))\n        display(md(f\"- **Number of images**: `{len(dicom_files)}`\"))\n\n        # Display one DICOM image\n        dcm_path = os.path.join(series_path, dicom_files[0])\n        dcm = pydicom.dcmread(dcm_path)\n        plt.figure(figsize=(5,5))\n        plt.imshow(dcm.pixel_array, cmap=\"gray\")\n        plt.title(f\"Example DICOM — {short_uid(example_series, 40)}\")\n        plt.axis(\"off\")\n        plt.show()\nelse:\n    display(md(\"**⚠️ Folder `series/` not found.**\"))\n\nhr()\n\n# ---------- 5) Localization columns (13 arteries) ----------\ndisplay(md(\"## 🧠 Localization columns (13 arteries)\"))\n\ndisplay(md(\"\"\" This section displays one representative annotated slice for each of the 13 arterial localizations, \nhighlighting the approximate position of aneurysms directly on DICOM images.\"\"\"))\n\nloc_palette = {\n    \"Left Infraclinoid Internal Carotid Artery\": \"tab:blue\",\n    \"Right Infraclinoid Internal Carotid Artery\": \"tab:orange\",\n    \"Left Supraclinoid Internal Carotid Artery\": \"tab:green\",\n    \"Right Supraclinoid Internal Carotid Artery\": \"tab:red\",\n    \"Left Middle Cerebral Artery\": \"tab:purple\",\n    \"Right Middle Cerebral Artery\": \"tab:brown\",\n    \"Anterior Communicating Artery\": \"tab:pink\",\n    \"Left Anterior Cerebral Artery\": \"tab:orange\",\n    \"Right Anterior Cerebral Artery\": \"tab:olive\",\n    \"Left Posterior Communicating Artery\": \"tab:cyan\",\n    \"Right Posterior Communicating Artery\": \"tab:cyan\",\n    \"Basilar Tip\": \"tab:green\",\n    \"Other Posterior Circulation\": \"tab:red\",\n}\n\ndef parse_coords(s):\n    if isinstance(s, dict): return s\n    for parser in (ast.literal_eval, json.loads):\n        try:\n            out = parser(s)\n            if isinstance(out, dict) and \"x\" in out and \"y\" in out:\n                return out\n        except Exception:\n            pass\n    return None\n\ndef show_localizer_point(row):\n    sid = row[\"SeriesInstanceUID\"]; sop = row[\"SOPInstanceUID\"]\n    coords = parse_coords(row[\"coordinates\"])\n    spath = os.path.join(SERIES_DIR, sid)\n    dcm_path = os.path.join(spath, sop + \".dcm\")\n    if not os.path.exists(dcm_path):\n        matches = [f for f in os.listdir(spath) if f.startswith(str(sop))]\n        if matches: dcm_path = os.path.join(spath, matches[0])\n\n    if not os.path.exists(dcm_path) or coords is None:\n        display(md(\"**⚠️ Cannot display annotated slice (file or coordinates missing).**\"))\n        return\n\n    dcm = pydicom.dcmread(dcm_path)\n    arr = dcm.pixel_array\n\n    # --- Handle multi-frame DICOMs ---\n    if arr.ndim == 3:\n        # shape typically (nframes, H, W)\n        mid = arr.shape[0] // 2\n        img = arr[mid]\n    else:\n        img = arr\n\n    # --- MONOCHROME1 images (white=low) -> invert to look like MONOCHROME2 ---\n    try:\n        if getattr(dcm, \"PhotometricInterpretation\", \"\") == \"MONOCHROME1\":\n            # normalize then invert\n            img = img.astype(\"float32\")\n            img = (img - img.min()) / max(1e-6, (img.max() - img.min()))\n            img = 1.0 - img\n    except Exception:\n        pass\n\n    H, W = img.shape[:2]\n\n    # --- Many localizer coords are in a 512×512 reference; rescale if needed ---\n    x, y = float(coords[\"x\"]), float(coords[\"y\"])\n    if (W != 512) or (H != 512):\n        x = x * (W / 512.0)\n        y = y * (H / 512.0)\n\n    # Clamp inside image bounds (avoids warnings if slightly out of range)\n    x = max(0, min(W - 1, x))\n    y = max(0, min(H - 1, y))\n\n    # --- Display ---\n    plt.figure(figsize=(6,6))\n    plt.imshow(img, cmap=\"gray\")\n    color = loc_palette.get(row.get(\"location\",\"\"), \"red\")\n    plt.scatter(x, y, s=60, marker=\"x\", c=color)\n    plt.axis(\"off\")\n    plt.show()\n\nRANDOM_STATE = 42\nunique_locs = loc_df[\"location\"].unique()\n\nfor loc in unique_locs:\n    subset = loc_df[loc_df[\"location\"] == loc]\n    if len(subset) == 0:\n        continue\n    row = subset.sample(1, random_state=RANDOM_STATE).iloc[0]\n    display(md(f\"📍 Localization: **{loc}**\"))\n    show_localizer_point(row)\n\n\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T13:16:38.299614Z","iopub.execute_input":"2025-09-15T13:16:38.299788Z","iopub.status.idle":"2025-09-15T13:16:44.139554Z","shell.execute_reply.started":"2025-09-15T13:16:38.29977Z","shell.execute_reply":"2025-09-15T13:16:44.138814Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# --------------------------------------------\n# 🧑‍⚕️👩‍⚕️ Demographics \n# --------------------------------------------\ndisplay(md(\"## 🧑‍⚕️👩‍⚕️ Demographics\"))\n\n# --- Combined figure: Age, Sex, Target ---\nfig = plt.figure(figsize=(14,4))\n\n# Age distribution\nplt.subplot(1,3,1)\nif sns is not None:\n    sns.histplot(train_df[\"PatientAge\"], bins=20, kde=True, color=\"skyblue\")\nelse:\n    plt.hist(train_df[\"PatientAge\"], bins=20, color=\"skyblue\")\nplt.title(\"Patient Age Distribution\")\n\n# Sex distribution\nplt.subplot(1,3,2)\nsex_counts = train_df[\"PatientSex\"].value_counts(dropna=False)\nplt.bar(sex_counts.index.astype(str), sex_counts.values, color=\"orange\")\nplt.title(\"Patient Sex Distribution\")\nplt.xlabel(\"Sex\"); plt.ylabel(\"Count\")\n\n# Aneurysm Present distribution\nplt.subplot(1,3,3)\ntarget_counts = train_df[\"Aneurysm Present\"].value_counts()\nplt.bar([\"Absent (0)\",\"Present (1)\"],\n        [target_counts.get(0,0), target_counts.get(1,0)],\n        color=[\"lightgreen\",\"salmon\"])\nplt.title(\"Target: Aneurysm Present\")\nplt.tight_layout()\nplt.show()\n\n# --- Separate barplot for clarity ---\ncounts = train_df[\"Aneurysm Present\"].value_counts()\nplt.figure(figsize=(5,4))\nif sns is not None:\n    sns.barplot(x=counts.index, y=counts.values, palette=\"viridis\")\nelse:\n    plt.bar(counts.index.astype(int), counts.values, color=\"gray\")\nplt.title(\"Distribution of Aneurysm Presence\")\nplt.xticks([0,1], [\"Absent (0)\", \"Present (1)\"])\nplt.ylabel(\"Number of series\")\nplt.show()\n\nprint(f\"Number of series without aneurysm: {counts.get(0,0)}\")\nprint(f\"Number of series with aneurysm: {counts.get(1,0)}\")\n\n# --- Age distribution (again, for clarity) ---\nplt.figure(figsize=(6,4))\nif sns is not None:\n    sns.histplot(train_df[\"PatientAge\"], bins=20, kde=True, color=\"steelblue\")\nelse:\n    plt.hist(train_df[\"PatientAge\"], bins=20, color=\"steelblue\")\nplt.title(\"Age Distribution of Patients\")\nplt.show()\n\n# --- Sex vs Aneurysm Presence ---\nplt.figure(figsize=(6,4))\nif sns is not None:\n    sns.countplot(x=\"PatientSex\", hue=\"Aneurysm Present\", data=train_df, palette=\"coolwarm\")\nelse:\n    sexes = sorted(train_df[\"PatientSex\"].dropna().unique())\n    for i, sex in enumerate(sexes):\n        sub = train_df[train_df[\"PatientSex\"]==sex][\"Aneurysm Present\"].value_counts()\n        plt.bar([i-0.2, i+0.2], [sub.get(0,0), sub.get(1,0)], width=0.4)\n    plt.xticks(range(len(sexes)), sexes)\nplt.title(\"Sex vs Aneurysm Presence\")\nplt.show()\n\n# --- Extra stats ---\nprevalence = train_df[\"Aneurysm Present\"].mean()\ndisplay(md(f\"**Overall prevalence of aneurysm in dataset:** {prevalence:.2%}\"))\n\nsex_ct = pd.crosstab(train_df[\"PatientSex\"], train_df[\"Aneurysm Present\"], normalize=\"index\")\ndisplay(md(\"**Sex vs Presence (row-wise proportions):**\"))\ndisplay(sex_ct)\n\ndisplay(md(\"\"\"Descriptive analysis of the dataset reveals several interesting trends. \nThe age distribution shows that the majority of patients are between 50 and 70 years old, with a peak around 60 years old. \nIn terms of gender, there are more women than men in the sample (approximately 3,000 versus 1,300). \nThe overall prevalence of aneurysms is 42.9%, meaning that nearly one in two patients has a detected abnormality. \nComparing by gender, we see that women are proportionally more affected (46.8% of aneurysms in female patients versus 34.1% in male patients). \nFinally, the target variable confirms a relative imbalance with 2,484 series without aneurysms and 1,864 positive series.\"\"\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T13:16:44.140742Z","iopub.execute_input":"2025-09-15T13:16:44.141144Z","iopub.status.idle":"2025-09-15T13:16:45.151698Z","shell.execute_reply.started":"2025-09-15T13:16:44.141125Z","shell.execute_reply":"2025-09-15T13:16:45.151165Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# --------------------------------------------\n# 🔬 Imaging Modality \n# --------------------------------------------\ndisplay(md(\"## 🔬 Imaging Modality\"))\n\n# --- Crosstabs: counts & proportions ---\nmod_ct = pd.crosstab(train_df[\"Modality\"], train_df[\"Aneurysm Present\"])\nmod_ct_prop = pd.crosstab(train_df[\"Modality\"], train_df[\"Aneurysm Present\"], normalize=\"index\")\n\ndisplay(md(\"**Counts (by Modality × Target):**\"))\ndisplay(mod_ct)\n\ndisplay(md(\"**Row-wise proportions (per modality):**\"))\ndisplay(mod_ct_prop.round(3))\n\n# --- Visualization ---\nplt.figure(figsize=(6,4))\nif sns is not None:\n    sns.countplot(x=\"Modality\", hue=\"Aneurysm Present\", data=train_df, palette=\"Set2\")\n    plt.title(\"Imaging Modality vs Aneurysm Presence\")\n    plt.xlabel(\"Modality\"); plt.ylabel(\"Number of series\")\n    plt.show()\nelse:\n    modalities = sorted(train_df[\"Modality\"].dropna().unique())\n    for k, mod in enumerate(modalities):\n        sub = train_df[train_df[\"Modality\"]==mod][\"Aneurysm Present\"].value_counts()\n        plt.bar([k-0.2, k+0.2], [sub.get(0,0), sub.get(1,0)], width=0.4, label=mod)\n    plt.xticks(range(len(modalities)), modalities)\n    plt.title(\"Imaging Modality vs Aneurysm Presence\")\n    plt.xlabel(\"Modality\"); plt.ylabel(\"Number of series\")\n    plt.show()\n\ndisplay(md(\"\"\"Here, we can see the distribution of detected aneurysms according to the medical imaging modality:\n\n- CTA (Computed Tomography Angiography): a CT-based technique, often used in emergencies because it is fast and provides good visualization of the vessels. This modality has the highest proportion of detected aneurysms (53.9%) .\n- MRA (Magnetic Resonance Angiography): magnetic resonance imaging without the injection of iodinated contrast medium. It is non-invasive but slightly less sensitive than CTA for small aneurysms. Here, the proportion of aneurysms is 44.3%.\n- Post-contrast T1 MRI: T1 MRI sequence performed after contrast injection, useful for visualizing vascular walls and certain abnormalities. Here, it shows a lower proportion of aneurysms (25.2%).\n- T2 MRI: MRI sequence sensitive to fluid variations and cerebral hemodynamics, useful in the general evaluation of the brain. It also shows a low proportion of aneurysms (26.2%).\n\nThese results suggest that targeted vascular methods (CTA, MRA) detect proportionally more aneurysms than conventional MRI sequences (post-contrast T1, T2), which is consistent with their clinical role.\"\"\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T13:16:45.152373Z","iopub.execute_input":"2025-09-15T13:16:45.152595Z","iopub.status.idle":"2025-09-15T13:16:45.314694Z","shell.execute_reply.started":"2025-09-15T13:16:45.15257Z","shell.execute_reply":"2025-09-15T13:16:45.314016Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# --------------------------------------------\n# 🧠 Multi-aneurysm (per series) & Localization Distribution\n# --------------------------------------------\ndisplay(md(\"## 🧠 Multi-aneurysm (per series) & Localization Distribution\"))\n\n# --- 1) Identify localization columns (13 arteries) ---\nlocation_cols = [c for c in train_df.columns \n                 if c not in [\"SeriesInstanceUID\",\"PatientAge\",\"PatientSex\",\"Modality\",\"Aneurysm Present\"]]\n\nif len(location_cols) == 0:\n    display(md(\"**⚠️ No localization columns detected.**\"))\nelse:\n    display(md(f\"- **Detected localization columns:** {len(location_cols)}\"))\n\n    # --- 2) Number of positive localizations per series ---\n    train_df = train_df.copy()  # avoid potential SettingWithCopy warnings\n    train_df[\"nb_locations\"] = train_df[location_cols].sum(axis=1)\n\n    display(md(\"**Descriptive stats for the number of positive localizations per series:**\"))\n    display(train_df[\"nb_locations\"].describe().to_frame().T)\n\n    # Plot 1: distribution of the number of localizations (per series)\n    plt.figure(figsize=(6,4))\n    if sns is not None:\n        sns.countplot(x=\"nb_locations\", data=train_df, palette=\"viridis\")\n    else:\n        vals = train_df[\"nb_locations\"].value_counts().sort_index()\n        plt.bar(vals.index.astype(str), vals.values)\n    plt.title(\"Number of positive localizations per series\")\n    plt.xlabel(\"# of localizations\"); plt.ylabel(\"Number of series\")\n    plt.show()\n\n    # --- 3) Global counts by localization (across all series) ---\n    loc_counts = train_df[location_cols].sum().sort_values(ascending=False)\n    loc_perc   = 100 * loc_counts / len(train_df)\n\n    summary_loc = pd.DataFrame({\n        \"count\": loc_counts,\n        \"percentage_of_series\": loc_perc.round(2)\n    })\n\n    display(md(\"**Localization frequency across the dataset (sorted):**\"))\n    display(summary_loc)\n\n    # Plot 2: localization frequency barplot\n    plt.figure(figsize=(10,5))\n    if sns is not None:\n        sns.barplot(x=loc_counts.values, y=loc_counts.index, palette=\"magma\")\n    else:\n        plt.barh(loc_counts.index, loc_counts.values)\n        plt.gca().invert_yaxis()\n    plt.title(\"Localization frequency (series with aneurysm)\")\n    plt.xlabel(\"Number of positive series\")\n    plt.ylabel(\"Localization\")\n    plt.show()\n\ndisplay(md(\"\"\"This analysis focuses on the distribution of aneurysms by anatomical location.\nThe vast majority of patients have zero positive locations (no aneurysms), but there are also cases with 1 to 5 positive locations per series.\nIn terms of frequency, the most affected locations are:\nthe anterior communicating artery (8.35%),\nthe left supraclinoid internal carotid artery (7.61%),\nthe right middle cerebral artery (6.76%),\nthe right supraclinoid internal carotid artery (6.37%).\nThese areas therefore constitute the main sites of aneurysms in the dataset.\nConversely, certain arteries such as the right anterior cerebral artery (1.29%) or the left infraclinoid internal carotid artery (1.79%) are much less represented.\nThis distribution suggests that certain anatomical locations are more vulnerable to the development of aneurysms, which is consistent with the medical literature on arterial bifurcation areas and areas of hemodynamic turbulence.\"\"\"))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T13:16:45.316218Z","iopub.execute_input":"2025-09-15T13:16:45.316447Z","iopub.status.idle":"2025-09-15T13:16:45.757604Z","shell.execute_reply.started":"2025-09-15T13:16:45.316423Z","shell.execute_reply":"2025-09-15T13:16:45.756961Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --------------------------------------------\n# 🩻 DICOM exploration \n# --------------------------------------------\n\ndisplay(md(\"## 🩻 DICOM exploration \"))\nRANDOM_STATE = 42\nrandom.seed(RANDOM_STATE)\nnp.random.seed(RANDOM_STATE)\n\n\nFAST_MODE = True          \nMAX_SERIES_FOR_STATS = 80 \n\ndef _safe_list_dcms(series_path):\n    \"\"\"Liste les DICOM d'une série. En FAST_MODE: tri simple par nom (pas de header).\"\"\"\n    files = [f for f in os.listdir(series_path) if f.lower().endswith(\".dcm\")]\n    if not files:\n        files = os.listdir(series_path)  \n    if FAST_MODE:\n        return sorted(files)        \n    \n    def _key(fname):\n        try:\n            ds = pydicom.dcmread(os.path.join(series_path, fname), stop_before_pixels=True)\n            return getattr(ds, \"InstanceNumber\", 1e12)\n        except Exception:\n            return 1e12\n    try:\n        return sorted(files, key=_key)\n    except Exception:\n        return sorted(files)\n\ndef count_slices_in_series_fast(series_path: str) -> int:\n    \"\"\"Comptage approximatif (rapide) : nombre de fichiers .dcm\"\"\"\n    files = [f for f in os.listdir(series_path) if f.lower().endswith(\".dcm\")]\n    if not files:\n        files = os.listdir(series_path)\n    return len(files) if len(files) > 0 else 0\n\ndef count_slices_in_series_precise(series_path: str) -> int:\n    \"\"\"Comptage précis : somme NumberOfFrames de chaque fichier (lent).\"\"\"\n    files = [f for f in os.listdir(series_path) if f.lower().endswith(\".dcm\")]\n    if not files:\n        files = os.listdir(series_path)\n    total = 0\n    for f in files:\n        try:\n            ds = pydicom.dcmread(os.path.join(series_path, f), stop_before_pixels=True)\n            frames = int(getattr(ds, \"NumberOfFrames\", 1) or 1)\n            total += max(frames, 1)\n        except Exception:\n            total += 1\n    return total\n\ndef show_series_grid(series_id, n=9):\n    \"\"\"Grille de n slices réparties dans la série (FAST_MODE = pas de tri par header).\"\"\"\n    spath = os.path.join(SERIES_DIR, series_id)\n    if not os.path.isdir(spath):\n        print(\"⚠️ Series not found:\", series_id); return\n    files = _safe_list_dcms(spath)\n    if not files:\n        print(\"⚠️ No files found in series:\", series_id); return\n\n    idxs = np.linspace(0, len(files)-1, min(n, len(files))).astype(int)\n    rows = cols = int(math.ceil(math.sqrt(len(idxs))))\n    fig, axes = plt.subplots(rows, cols, figsize=(12, 12))\n    axes = np.array(axes).reshape(rows, cols)\n\n    for ax, i in zip(axes.flatten(), idxs):\n        dcm = pydicom.dcmread(os.path.join(spath, files[i]))\n        arr = dcm.pixel_array\n        if arr.ndim == 3:   # multi-frame\n            arr = arr[arr.shape[0]//2]\n        ax.imshow(arr, cmap=\"gray\")\n        ax.set_title(f\"Slice {i}\")\n        ax.axis(\"off\")\n\n    for ax in axes.flatten()[len(idxs):]:\n        ax.axis(\"off\")\n    plt.suptitle(f\"Series {series_id} — {len(files)} images\", y=0.92)\n    plt.tight_layout()\n    plt.show()\n\n# ---- Slice stats ----\nslice_counts = []\nseries_ids = sorted([d for d in os.listdir(SERIES_DIR) if os.path.isdir(os.path.join(SERIES_DIR, d))])\n\nif MAX_SERIES_FOR_STATS is not None:\n    series_ids = series_ids[:MAX_SERIES_FOR_STATS]\n\nfor sid in series_ids:\n    spath = os.path.join(SERIES_DIR, sid)\n    if FAST_MODE:\n        n = count_slices_in_series_fast(spath)\n    else:\n        n = count_slices_in_series_precise(spath)\n    slice_counts.append((sid, n))\n\nvals = [v for _, v in slice_counts] or [0]\nmin_slices = int(np.min(vals)); max_slices = int(np.max(vals))\nmean_slices = float(np.mean(vals)); median_slices = float(np.median(vals))\nsid_min = next(s for s, v in slice_counts if v == min_slices)\nsid_max = next(s for s, v in slice_counts if v == max_slices)\n\ndisplay(md(\"### 📊 Slice count per series — summary\"))\ndisplay(md(f\"- **Series analysed** : {len(vals)} (FAST_MODE={FAST_MODE})\"))\ndisplay(md(f\"- **Min** : {min_slices} (e.g., `{sid_min}`)\"))\ndisplay(md(f\"- **Max** : {max_slices} (e.g., `{sid_max}`)\"))\ndisplay(md(f\"- **Mean** : {mean_slices:.2f}\"))\ndisplay(md(f\"- **Median** : {median_slices:.0f}\"))\n\nplt.figure(figsize=(6,4))\nplt.hist(vals, bins=20)\nplt.title(\"Distribution of slices per series\")\nplt.xlabel(\"Slices\"); plt.ylabel(\"Series count\")\nplt.show()\n\ndisplay(md(\"\"\"Here, we analyze the number of slices per DICOM series. \nThere is significant variability between series: the number of slices ranges from 13 to 898, with an average of 195 and a median of 157. \nThe distribution shows that the majority of series have fewer than 200 slices, but some longer series have several hundred images.\"\"\"))\n\n# ---- Show one positive example ----\npos_ids = train_df.loc[train_df[\"Aneurysm Present\"]==1, \"SeriesInstanceUID\"]\nchosen_series = pos_ids.iloc[0] if len(pos_ids) else train_df[\"SeriesInstanceUID\"].iloc[0]\n\ndisplay(md(f\"**🖼️ Grid of slices for series:** `{chosen_series}`\"))\nshow_series_grid(chosen_series, n=9)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T13:16:45.758537Z","iopub.execute_input":"2025-09-15T13:16:45.758779Z","iopub.status.idle":"2025-09-15T13:16:50.940283Z","shell.execute_reply.started":"2025-09-15T13:16:45.758761Z","shell.execute_reply":"2025-09-15T13:16:50.939649Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# ------------------------------------------------------------\n# ✅ Preprocessing Checklist — Step 2\n# ------------------------------------------------------------\ndisplay(md(\"## ✅ Preprocessing Checklist (Step 2)\"))\n\nchecklist = [\n    \"- [ ] **Intensity normalization** (e.g., min-max or z-score)\",\n    \"- [ ] **Approach choice**: **2D (slice-based)** or **3D (full volume)** models\",\n    \"- [ ] **Class imbalance handling** (positives vs negatives, weighting or oversampling)\",\n    \"- [ ] **Stratified split** train / validation / test (preserve prevalence)\",\n    \"- [ ] **Create a PyTorch Dataset/Dataloader** to load DICOM series\",\n    \"- [ ] **Data augmentations** (rotations, flips, contrast, noise, etc.)\",\n    \"- [ ] **Save preprocessed tensors** for faster training\"\n]\n\nfor item in checklist:\n    display(md(item))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T13:16:50.941117Z","iopub.execute_input":"2025-09-15T13:16:50.941657Z","iopub.status.idle":"2025-09-15T13:16:50.954908Z","shell.execute_reply.started":"2025-09-15T13:16:50.941634Z","shell.execute_reply":"2025-09-15T13:16:50.954238Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\n\ndef preprocess_dcm(path, size=(224,224)):\n    dcm = pydicom.dcmread(path)\n    arr = dcm.pixel_array.astype(np.float32)\n    # Normalisation z-score\n    arr = (arr - arr.mean()) / (arr.std() + 1e-6)\n    # Resize en 2D\n    arr_resized = cv2.resize(arr, size, interpolation=cv2.INTER_AREA)\n    return arr_resized\n\n\n# Sélectionner une série (par exemple la première)\nexample_series = sorted(os.listdir(SERIES_DIR))[0]\nseries_path = os.path.join(SERIES_DIR, example_series)\n\n# Liste des DICOMs dans cette série\ndicom_files = [f for f in os.listdir(series_path) if f.lower().endswith(\".dcm\")]\nif not dicom_files:\n    raise ValueError(f\"Aucun fichier DICOM trouvé dans {series_path}\")\n\n# On prend le premier fichier DICOM\nfirst_file = os.path.join(series_path, dicom_files[0])\n\n# Prétraitement et affichage\nimg = preprocess_dcm(first_file)\nplt.imshow(img, cmap=\"gray\")\nplt.title(f\"Prétraitement DICOM (224x224) — {example_series}\")\nplt.axis(\"off\")\nplt.show()\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T13:36:11.214675Z","iopub.execute_input":"2025-09-15T13:36:11.215409Z","iopub.status.idle":"2025-09-15T13:36:11.393768Z","shell.execute_reply.started":"2025-09-15T13:36:11.215386Z","shell.execute_reply":"2025-09-15T13:36:11.392929Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedGroupKFold\n\nX = train_df[\"SeriesInstanceUID\"]\ny = train_df[\"Aneurysm Present\"]\n\n# Ici on utilise SeriesInstanceUID comme fallback (pas de PatientID dispo)\ngroups = train_df[\"SeriesInstanceUID\"]\n\ncv = StratifiedGroupKFold(n_splits=5, shuffle=True, random_state=42)\nfor fold, (train_idx, val_idx) in enumerate(cv.split(X, y, groups)):\n    print(f\"Fold {fold}: Train {len(train_idx)} | Val {len(val_idx)}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T13:36:33.599507Z","iopub.execute_input":"2025-09-15T13:36:33.599803Z","iopub.status.idle":"2025-09-15T13:36:35.012229Z","shell.execute_reply.started":"2025-09-15T13:36:33.599782Z","shell.execute_reply":"2025-09-15T13:36:35.011416Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torchvision.models as models\nfrom torchvision.models import resnet18, ResNet18_Weights\n\n# Charger ResNet18 pré-entraîné sur ImageNet\nmodel = resnet18(weights=ResNet18_Weights.IMAGENET1K_V1)\n\n# Modifier la dernière couche pour une sortie binaire (présence ou non d’anévrisme)\nmodel.fc = nn.Linear(model.fc.in_features, 1)  # sortie = 1 logit\n\nprint(model)\n\n!pip install torchsummary\nfrom torchsummary import summary\n\ndevice = torch.device(\"cpu\")\nmodel.to(device)\n\nfrom torchsummary import summary\nsummary(model, (3, 224, 224), device=\"cpu\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T13:28:07.955266Z","iopub.execute_input":"2025-09-15T13:28:07.955916Z","iopub.status.idle":"2025-09-15T13:28:11.359755Z","shell.execute_reply.started":"2025-09-15T13:28:07.95589Z","shell.execute_reply":"2025-09-15T13:28:11.358775Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from graphviz import Digraph\nfrom IPython.display import Image\n\n# Création du graphe\ndot = Digraph(comment=\"ResNet18 Architecture\", format=\"png\")\ndot.attr(rankdir=\"TB\", size=\"8\")\n\n# Entrée\ndot.node(\"input\", \"Entrée\\n(3, 224, 224)\", shape=\"box\", style=\"filled\", color=\"lightblue\")\n\n# Blocs principaux\ndot.node(\"conv1\", \"Conv1 + BN + ReLU\\n(64, 112, 112)\", shape=\"box\")\ndot.node(\"pool\", \"MaxPool\\n(64, 56, 56)\", shape=\"box\", style=\"filled\", color=\"lightgrey\")\n\ndot.node(\"layer1\", \"Bloc Résiduel x2\\n(64, 56, 56)\", shape=\"box\")\ndot.node(\"layer2\", \"Bloc Résiduel x2\\n(128, 28, 28)\", shape=\"box\")\ndot.node(\"layer3\", \"Bloc Résiduel x2\\n(256, 14, 14)\", shape=\"box\")\ndot.node(\"layer4\", \"Bloc Résiduel x2\\n(512, 7, 7)\", shape=\"box\")\n\n# Pooling + FC\ndot.node(\"gap\", \"Global Avg Pool\\n(512, 1, 1)\", shape=\"box\", style=\"filled\", color=\"lightgrey\")\ndot.node(\"fc\", \"Fully Connected\\n(512 → 1)\", shape=\"box\")\ndot.node(\"sigmoid\", \"Sigmoïde\\nProbabilité\", shape=\"box\", style=\"filled\", color=\"lightgreen\")\n\n# Connexions\ndot.edges([(\"input\",\"conv1\"), (\"conv1\",\"pool\"), (\"pool\",\"layer1\"),\n           (\"layer1\",\"layer2\"), (\"layer2\",\"layer3\"), (\"layer3\",\"layer4\"),\n           (\"layer4\",\"gap\"), (\"gap\",\"fc\"), (\"fc\",\"sigmoid\")])\n\n# Sauvegarde et affichage dans le notebook\ndot.render(\"resnet18_architecture\", format=\"png\")  \nImage(filename=\"resnet18_architecture.png\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T13:42:23.936024Z","iopub.execute_input":"2025-09-15T13:42:23.936306Z","iopub.status.idle":"2025-09-15T13:42:23.973828Z","shell.execute_reply.started":"2025-09-15T13:42:23.936284Z","shell.execute_reply":"2025-09-15T13:42:23.973255Z"}},"outputs":[],"execution_count":null}]}