{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":99552,"databundleVersionId":13441085}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# RSNA Intracranial Aneurysm Detection","metadata":{}},{"cell_type":"markdown","source":"## Data Exploration","metadata":{}},{"cell_type":"markdown","source":"### Import Libraries","metadata":{}},{"cell_type":"code","source":"import os, glob, json, textwrap\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport pydicom\nfrom pydicom.filereader import dcmread\n\nimport nibabel as nib","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-08-27T07:14:12.230496Z","iopub.execute_input":"2025-08-27T07:14:12.230773Z","iopub.status.idle":"2025-08-27T07:14:14.643869Z","shell.execute_reply.started":"2025-08-27T07:14:12.230747Z","shell.execute_reply":"2025-08-27T07:14:14.642742Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Path Setup","metadata":{}},{"cell_type":"code","source":"pd.set_option(\"display.max_columns\", 100)\nsns.set_theme(context=\"notebook\", style=\"whitegrid\")\nDATA_DIR = \"/kaggle/input/rsna-intracranial-aneurysm-detection\"","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-08-27T07:14:15.856225Z","iopub.execute_input":"2025-08-27T07:14:15.857099Z","iopub.status.idle":"2025-08-27T07:14:15.862102Z","shell.execute_reply.started":"2025-08-27T07:14:15.857071Z","shell.execute_reply":"2025-08-27T07:14:15.86124Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv(f\"{DATA_DIR}/train.csv\")\nlocs = pd.read_csv(f\"{DATA_DIR}/train_localizers.csv\")\n\nprint(train.shape)\nprint(locs.shape)\ntrain.head(3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-27T07:14:56.200558Z","iopub.execute_input":"2025-08-27T07:14:56.201199Z","iopub.status.idle":"2025-08-27T07:14:56.2849Z","shell.execute_reply.started":"2025-08-27T07:14:56.201175Z","shell.execute_reply":"2025-08-27T07:14:56.284147Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Identify label columns (13 locations + presence)\nlocation_cols = [\n    \"Left Infraclinoid Internal Carotid Artery\",\n    \"Right Infraclinoid Internal Carotid Artery\",\n    \"Left Supraclinoid Internal Carotid Artery\",\n    \"Right Supraclinoid Internal Carotid Artery\",\n    \"Left Middle Cerebral Artery\",\n    \"Right Middle Cerebral Artery\",\n    \"Anterior Communicating Artery\",\n    \"Left Anterior Cerebral Artery\",\n    \"Right Anterior Cerebral Artery\",\n    \"Left Posterior Communicating Artery\",\n    \"Right Posterior Communicating Artery\",\n    \"Basilar Tip\",\n    \"Other Posterior Circulation\",\n]\npresence_col = \"Aneurysm Present\"\nmeta_cols = [\"SeriesInstanceUID\", \"Modality\", \"PatientAge\", \"PatientSex\"]\nassert set(location_cols).issubset(train.columns)\nassert presence_col in train.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-27T07:15:17.170854Z","iopub.execute_input":"2025-08-27T07:15:17.17114Z","iopub.status.idle":"2025-08-27T07:15:17.177802Z","shell.execute_reply.started":"2025-08-27T07:15:17.17112Z","shell.execute_reply":"2025-08-27T07:15:17.176823Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"series_root = f\"{DATA_DIR}/series\"\nseg_root = f\"{DATA_DIR}/segmentations\"\n\nseries_dirs = set(os.listdir(series_root))\nhas_series = train[\"SeriesInstanceUID\"].isin(series_dirs)\ncoverage = has_series.mean()\n\nloc_per_series = locs.groupby(\"SeriesInstanceUID\").size().rename(\"n_localizers\")\ntrain = train.merge(loc_per_series, on=\"SeriesInstanceUID\", how=\"left\").fillna({\"n_localizers\": 0})\n\n# Check what fraction have segmentations\nseg_series = set([f.replace(\".nii.gz\",\"\").replace(\".nii\",\"\") for f in os.listdir(seg_root)])\ntrain[\"has_seg\"] = train[\"SeriesInstanceUID\"].isin(seg_series)\n\ncoverage, train[\"has_seg\"].mean(), train[\"n_localizers\"].gt(0).mean()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-27T07:15:30.365512Z","iopub.execute_input":"2025-08-27T07:15:30.365829Z","iopub.status.idle":"2025-08-27T07:15:30.648839Z","shell.execute_reply.started":"2025-08-27T07:15:30.365808Z","shell.execute_reply":"2025-08-27T07:15:30.648028Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"##### Checking for inconsisties","metadata":{}},{"cell_type":"code","source":"any_location = train[location_cols].sum(axis=1).gt(0).astype(int)\ntrain[\"presence_from_locations\"] = any_location\nconsistency = (train[presence_col].astype(int) == train[\"presence_from_locations\"]).mean()\nincons = train.loc[train[presence_col].astype(int) != train[\"presence_from_locations\"], \n                   meta_cols + [presence_col, \"presence_from_locations\"]].head(10)\nconsistency, incons","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-27T07:15:42.447054Z","iopub.execute_input":"2025-08-27T07:15:42.447352Z","iopub.status.idle":"2025-08-27T07:15:42.463584Z","shell.execute_reply.started":"2025-08-27T07:15:42.44733Z","shell.execute_reply":"2025-08-27T07:15:42.462539Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":">No inconsistes found","metadata":{}},{"cell_type":"code","source":"fig, (ax0, ax1, ax2) = plt.subplots(1, 3, figsize=(16, 4))\n\n# 1) Presence count\nsns.countplot(data=train, x=presence_col, ax=ax0)\nax0.set_title(\"Aneurysm Present\")\nax0.set_xlabel(\"\")\nax0.set_ylabel(\"Count\")\n\n# 2) Location positive rates\npos_rates = train[location_cols].mean().sort_values(ascending=True)  # ascending so lowest at bottom\npos_rates.plot(kind=\"barh\", ax=ax1)  # horizontal\nax1.set_title(\"Location positive rates\")\nax1.set_xlabel(\"Rate\")\nax1.set_ylabel(\"\")  # labels are on y as categories\nax1.set_xlim(0, 1)  # rates within [0,1]\n\n# 3) Modality distribution\norder = train[\"Modality\"].value_counts().index\nsns.countplot(data=train, x=\"Modality\", ax=ax2, order=order)\nax2.set_title(\"Modality distribution\")\nax2.set_xlabel(\"\")\nax2.set_ylabel(\"Count\")\nax2.tick_params(axis=\"x\", rotation=30)\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-27T07:19:23.104812Z","iopub.execute_input":"2025-08-27T07:19:23.105094Z","iopub.status.idle":"2025-08-27T07:19:23.902483Z","shell.execute_reply.started":"2025-08-27T07:19:23.105077Z","shell.execute_reply":"2025-08-27T07:19:23.901623Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Demographics\nfig, axes = plt.subplots(1, 2, figsize=(12, 4))\n\n# Left subplot: sex counts\nsns.countplot(data=train, x=\"PatientSex\", ax=axes[0])  # Fixed: axes[0]\naxes[0].set_title(\"Sex\")                              # Fixed: axes[0]\naxes[0].set_xlabel(\"\")                                # Fixed: axes[0]\naxes[0].set_ylabel(\"Count\")                           # Fixed: axes[0]\n\ndef parse_age(x):\n    if pd.isna(x):\n        return np.nan\n    # typical DICOM age format: '045Y'\n    try:\n        return float(str(x).strip()[:3])\n    except Exception:\n        return pd.to_numeric(x, errors=\"coerce\")\n\nage_years = train[\"PatientAge\"].apply(parse_age)\n\n# Right subplot: age histogram\nsns.histplot(age_years, bins=20, ax=axes[1])\naxes[1].set_title(\"Age (years, approx)\")\naxes[1].set_xlabel(\"Years\")\naxes[1].set_ylabel(\"Count\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-27T07:40:53.949998Z","iopub.execute_input":"2025-08-27T07:40:53.950305Z","iopub.status.idle":"2025-08-27T07:40:54.47811Z","shell.execute_reply.started":"2025-08-27T07:40:53.950275Z","shell.execute_reply":"2025-08-27T07:40:54.477163Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Localizers","metadata":{}},{"cell_type":"code","source":"locs.head(3), locs[\"SeriesInstanceUID\"].nunique(), locs.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-27T07:25:43.287541Z","iopub.execute_input":"2025-08-27T07:25:43.287893Z","iopub.status.idle":"2025-08-27T07:25:43.297052Z","shell.execute_reply.started":"2025-08-27T07:25:43.28787Z","shell.execute_reply":"2025-08-27T07:25:43.296263Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# parse coordinate strings if provided as \"x y\" or \"x,y\"\ndef parse_xy(s):\n    if pd.isna(s): \n        return (np.nan, np.nan)\n    s = str(s).replace(\",\", \" \").split()\n    if len(s) >= 2:\n        try:\n            return float(s[0]), float(s[1])  # Fixed: was float(s), float(s[1])\n        except ValueError:\n            return (np.nan, np.nan)\n    return (np.nan, np.nan)\n\nlocs[[\"x\",\"y\"]] = locs[\"coordinates\"].apply(lambda s: pd.Series(parse_xy(s)))\nsns.histplot(locs[\"SeriesInstanceUID\"].value_counts(), bins=30)\nplt.title(\"Localizer count per series\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-27T07:30:18.062255Z","iopub.execute_input":"2025-08-27T07:30:18.062978Z","iopub.status.idle":"2025-08-27T07:30:18.540146Z","shell.execute_reply.started":"2025-08-27T07:30:18.06295Z","shell.execute_reply":"2025-08-27T07:30:18.539197Z"}},"outputs":[],"execution_count":null}]}