{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":99552,"databundleVersionId":13190393,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport os\nimport numpy as np\nimport pydicom\nimport nibabel as nib\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom glob import glob             \nfrom tqdm.notebook import tqdm\nimport seaborn as sns\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-30T15:04:48.648954Z","iopub.execute_input":"2025-07-30T15:04:48.649363Z","iopub.status.idle":"2025-07-30T15:04:50.297163Z","shell.execute_reply.started":"2025-07-30T15:04:48.64933Z","shell.execute_reply":"2025-07-30T15:04:50.296128Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pd.set_option('display.max_columns', None) # we want to display all columns in this notebook\npd.set_option('display.max_rows', 100) # increase number of displayed rows\npd.set_option('max_colwidth', None) # make full cells content visible","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-30T15:04:50.298201Z","iopub.execute_input":"2025-07-30T15:04:50.298957Z","iopub.status.idle":"2025-07-30T15:04:50.304997Z","shell.execute_reply.started":"2025-07-30T15:04:50.298928Z","shell.execute_reply":"2025-07-30T15:04:50.303994Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DATA_DIR = \"/kaggle/input/rsna-intracranial-aneurysm-detection/\"\nSEGMENTATION_DIR = os.path.join(DATA_DIR, \"segmentation\")\nSERIES_DIR = os.path.join(DATA_DIR, \"series\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T15:04:50.305885Z","iopub.execute_input":"2025-07-30T15:04:50.306227Z","iopub.status.idle":"2025-07-30T15:04:50.335655Z","shell.execute_reply.started":"2025-07-30T15:04:50.306187Z","shell.execute_reply":"2025-07-30T15:04:50.334563Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CSV files\n# --- 1. Load and Inspect CSV Data ---\nprint(\"--- 1. Loading and Inspecting CSV Data ---\")\ntrain_df = pd.read_csv(os.path.join(DATA_DIR, \"train.csv\"))\nlocalizers_df = pd.read_csv(os.path.join(DATA_DIR, \"train_localizers.csv\"))\nprint(f\"Loaded {len(train_df)} rows from train.csv\")\nprint(f\"Loaded {len(localizers_df)} rows from train_localizers.csv\")\nprint(\"\\n--- train.csv head ---\")\ndisplay(train_df.head())\nprint(\"\\n--- train_localizers.csv head ---\")\ndisplay(localizers_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T15:38:44.487475Z","iopub.execute_input":"2025-07-30T15:38:44.487843Z","iopub.status.idle":"2025-07-30T15:38:44.553251Z","shell.execute_reply.started":"2025-07-30T15:38:44.487817Z","shell.execute_reply":"2025-07-30T15:38:44.551969Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_df.columns)\nprint(localizers_df.columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T15:04:50.425956Z","iopub.execute_input":"2025-07-30T15:04:50.426291Z","iopub.status.idle":"2025-07-30T15:04:50.432613Z","shell.execute_reply.started":"2025-07-30T15:04:50.426247Z","shell.execute_reply":"2025-07-30T15:04:50.431083Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 2. Statistical Analysis and Visualization of CSV Data ---\nprint(\"\\n--- 2. Statistical Analysis and Visualization of CSV Data ---\")\n\n# Modality Distribution\nplt.figure(figsize=(8, 5))\nsns.countplot(data=train_df, x='Modality', palette='viridis')\nplt.title('Distribution of Modalities in Training Data')\nplt.xlabel('Modality')\nplt.ylabel('Number of Series')\nplt.grid(axis='y', linestyle='--', alpha=0.7)\nplt.show()\n\n# Aneurysm Presence Distribution\nplt.figure(figsize=(6, 4))\nsns.countplot(data=train_df, x='Aneurysm Present', palette='coolwarm')\nplt.title('Distribution of Aneurysm Presence')\nplt.xlabel('Aneurysm Present (0: No, 1: Yes)')\nplt.ylabel('Number of Series')\nplt.xticks([0, 1], ['No Aneurysm', 'Aneurysm'])\nplt.grid(axis='y', linestyle='--', alpha=0.7)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T15:04:50.435681Z","iopub.execute_input":"2025-07-30T15:04:50.436188Z","iopub.status.idle":"2025-07-30T15:04:50.823803Z","shell.execute_reply.started":"2025-07-30T15:04:50.436148Z","shell.execute_reply":"2025-07-30T15:04:50.822536Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 6. Analyze modalities\ndef analyze_modality_distribution(train_df):\n    \"\"\"Analyze the distribution of imaging modalities\"\"\"\n    \n    print(\"\\n\" + \"=\"*50)\n    print(\"🖼️ IMAGING MODALITY ANALYSIS\")\n    print(\"=\"*50)\n    \n    if 'Modality' in train_df.columns:\n        modality_counts = train_df['Modality'].value_counts()\n        print(\"Modality Distribution:\")\n        print(modality_counts)\n        \n        # Visualize modality distribution\n        plt.figure(figsize=(12, 6))\n        \n        plt.subplot(1, 2, 1)\n        modality_counts.plot(kind='bar', color='skyblue', edgecolor='black')\n        plt.title('Distribution of Imaging Modalities')\n        plt.xlabel('Modality')\n        plt.ylabel('Number of Series')\n        plt.xticks(rotation=45)\n        \n        plt.subplot(1, 2, 2)\n        plt.pie(modality_counts.values, labels=modality_counts.index, autopct='%1.1f%%', startangle=90)\n        plt.title('Modality Distribution (%)')\n        \n        plt.tight_layout()\n        plt.show()\n        \n        return modality_counts\n    else:\n        print(\"❌ No 'Modality' column found in train.csv\")\n        return None\nmodality_counts = analyze_modality_distribution(train_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T15:13:33.153216Z","iopub.execute_input":"2025-07-30T15:13:33.15401Z","iopub.status.idle":"2025-07-30T15:13:33.572829Z","shell.execute_reply.started":"2025-07-30T15:13:33.153976Z","shell.execute_reply":"2025-07-30T15:13:33.571795Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Group age into bins\ntrain_df['AgeGroup'] = pd.cut(train_df['PatientAge'], bins=[0, 20, 40, 60, 80, 100], labels=[\"0–20\", \"21–40\", \"41–60\", \"61–80\", \"81+\"])\n\n# Sex distribution\nsns.countplot(x=\"PatientSex\", data=train_df)\nplt.title(\"Patient Sex Distribution\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T15:04:50.825209Z","iopub.execute_input":"2025-07-30T15:04:50.825648Z","iopub.status.idle":"2025-07-30T15:04:51.011327Z","shell.execute_reply.started":"2025-07-30T15:04:50.825609Z","shell.execute_reply":"2025-07-30T15:04:51.009457Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Age group by sex\nsns.countplot(data=train_df, x=\"AgeGroup\", hue=\"PatientSex\")\nplt.title(\"Age Group by Sex\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T15:04:51.013257Z","iopub.execute_input":"2025-07-30T15:04:51.013871Z","iopub.status.idle":"2025-07-30T15:04:51.247146Z","shell.execute_reply.started":"2025-07-30T15:04:51.013836Z","shell.execute_reply":"2025-07-30T15:04:51.246145Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Age Distribution by Aneurysm Status\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Convert age to numeric (handling possible string entries)\ntrain_df['PatientAge'] = pd.to_numeric(train_df['PatientAge'], errors='coerce')\n\n# Plot age distribution\nplt.figure(figsize=(10, 5))\nsns.histplot(data=train_df, x='PatientAge', hue='Aneurysm Present', \n             bins=30, kde=True, element='step', palette=['#1f77b4', '#ff7f0e'])\nplt.title('Age Distribution by Aneurysm Status')\nplt.xlabel('Age')\nplt.ylabel('Count')\nplt.show()\n\n# Statistical summary\nprint(\"Age stats for aneurysm cases:\\n\", train_df[train_df['Aneurysm Present'] == 1]['PatientAge'].describe())\nprint(\"\\nAge stats for non-aneurysm cases:\\n\", train_df[train_df['Aneurysm Present'] == 0]['PatientAge'].describe())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T15:04:51.248444Z","iopub.execute_input":"2025-07-30T15:04:51.248722Z","iopub.status.idle":"2025-07-30T15:04:51.622201Z","shell.execute_reply.started":"2025-07-30T15:04:51.248699Z","shell.execute_reply":"2025-07-30T15:04:51.621169Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Sex Distribution by Aneurysm Status\n# Plot sex distribution\nplt.figure(figsize=(6, 4))\nsns.countplot(data=train_df, x='PatientSex', hue='Aneurysm Present', \n              palette=['#1f77b4', '#ff7f0e'])\nplt.title('Sex Distribution by Aneurysm Status')\nplt.xlabel('Sex')\nplt.ylabel('Count')\nplt.show()\n\n# Cross-tabulation\nprint(pd.crosstab(train_df['PatientSex'], train_df['Aneurysm Present'], \n      margins=True, margins_name=\"Total\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T15:04:51.623263Z","iopub.execute_input":"2025-07-30T15:04:51.623647Z","iopub.status.idle":"2025-07-30T15:04:51.865169Z","shell.execute_reply.started":"2025-07-30T15:04:51.623618Z","shell.execute_reply":"2025-07-30T15:04:51.863737Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Get location columns\nlocation_cols = [col for col in train_df.columns if col not in ['SeriesInstanceUID', 'PatientAge', 'PatientSex', 'Modality', 'Aneurysm Present']]\n\n# Convert to numeric if not already\ntrain_df[location_cols] = train_df[location_cols].apply(pd.to_numeric, errors='coerce').fillna(0).astype(int)\n\n# Calculate count and percentage\nlocation_counts = train_df[location_cols].sum().sort_values(ascending=False)\nlocation_percentages = location_counts / location_counts.sum() * 100\n\n# Plotting\nfig, axes = plt.subplots(1, 2, figsize=(18, 8))\n\n# Bar chart\nsns.barplot(x=location_counts.values, y=location_counts.index, ax=axes[0], palette='mako')\naxes[0].set_title(\"Aneurysm Count per Brain Location\")\naxes[0].set_xlabel(\"Count\")\naxes[0].set_ylabel(\"Brain Location\")\n\n# Annotate with count and percentage\nfor i, (count, percent) in enumerate(zip(location_counts.values, location_percentages.values)):\n    axes[0].text(count + 1, i, f\"{int(count)} ({percent:.1f}%)\", va='center')\n\n# Pie chart\naxes[1].pie(location_counts.values,\n            labels=[f\"{loc}\\n{val} ({pct:.1f}%)\" for loc, val, pct in zip(location_counts.index, location_counts.values, location_percentages.values)],\n            autopct=None,\n            startangle=140,\n            colors=sns.color_palette(\"mako\", len(location_counts)))\naxes[1].set_title(\"Aneurysm Distribution by Location (Percentage)\")\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T15:28:44.481879Z","iopub.execute_input":"2025-07-30T15:28:44.482242Z","iopub.status.idle":"2025-07-30T15:28:45.07694Z","shell.execute_reply.started":"2025-07-30T15:28:44.482218Z","shell.execute_reply":"2025-07-30T15:28:45.0758Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### localizer.csv","metadata":{}},{"cell_type":"code","source":"import ast\n# Convert string coordinates to dictionaries\nlocalizers_df['coords'] = localizers_df['coordinates'].apply(ast.literal_eval)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T15:39:34.557035Z","iopub.execute_input":"2025-07-30T15:39:34.557449Z","iopub.status.idle":"2025-07-30T15:39:34.604997Z","shell.execute_reply.started":"2025-07-30T15:39:34.557418Z","shell.execute_reply":"2025-07-30T15:39:34.603803Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Extract x, y, z coordinates\nlocalizers_df['x'] = localizers_df['coords'].apply(lambda x: x.get('x', np.nan))\nlocalizers_df['y'] = localizers_df['coords'].apply(lambda x: x.get('y', np.nan))\nprint(localizers_df[['SeriesInstanceUID', 'SOPInstanceUID', 'location', 'x', 'y']].head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T15:39:43.588958Z","iopub.execute_input":"2025-07-30T15:39:43.589295Z","iopub.status.idle":"2025-07-30T15:39:43.605505Z","shell.execute_reply.started":"2025-07-30T15:39:43.589257Z","shell.execute_reply":"2025-07-30T15:39:43.604369Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Analyze Location Distribution\nplt.figure(figsize=(10, 5))\nsns.countplot(y=\"location\", data=localizers_df, order=localizers_df['location'].value_counts().index)\nplt.title(\"Brain Location Distribution (where aneurysm is present)\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T15:40:48.17231Z","iopub.execute_input":"2025-07-30T15:40:48.172983Z","iopub.status.idle":"2025-07-30T15:40:48.48115Z","shell.execute_reply.started":"2025-07-30T15:40:48.172949Z","shell.execute_reply":"2025-07-30T15:40:48.480134Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Heatmap of Coordinate Densities\nheatmap_data = localizers_df[['x', 'y']].dropna()\n\nplt.hist2d(heatmap_data['x'], heatmap_data['y'], bins=50, cmap='hot')\nplt.colorbar(label='Frequency')\nplt.title(\"Heatmap of Aneurysm Coordinates\")\nplt.xlabel(\"x\")\nplt.ylabel(\"y\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T15:45:51.451214Z","iopub.execute_input":"2025-07-30T15:45:51.451584Z","iopub.status.idle":"2025-07-30T15:45:51.72131Z","shell.execute_reply.started":"2025-07-30T15:45:51.451561Z","shell.execute_reply":"2025-07-30T15:45:51.719924Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Merge with aneurysm labels\n# merged_df = localizers_df.merge(train_df[['SeriesInstanceUID', 'Aneurysm Present']], \n#                                on='SeriesInstanceUID', how='left')\n\nmerged_df = pd.merge(train_df, localizers_df, on='SeriesInstanceUID', how='left')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T16:02:27.676871Z","iopub.execute_input":"2025-07-30T16:02:27.677226Z","iopub.status.idle":"2025-07-30T16:02:27.691734Z","shell.execute_reply.started":"2025-07-30T16:02:27.677201Z","shell.execute_reply":"2025-07-30T16:02:27.690662Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Age vs Location\nplt.figure(figsize=(10, 6))\nsns.boxplot(data=merged_df, x='location', y='PatientAge')\nplt.xticks(rotation=45)\nplt.title(\"Patient Age by Aneurysm Location\")\nplt.show()\n\n# Sex vs Location\nplt.figure(figsize=(10, 5))\nsns.countplot(data=merged_df, x='location', hue='PatientSex')\nplt.xticks(rotation=45)\nplt.title(\"Aneurysm Location by Patient Sex\")\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T15:47:00.352004Z","iopub.execute_input":"2025-07-30T15:47:00.352396Z","iopub.status.idle":"2025-07-30T15:47:01.07879Z","shell.execute_reply.started":"2025-07-30T15:47:00.352371Z","shell.execute_reply":"2025-07-30T15:47:01.07763Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(10, 8))\nsns.scatterplot(data=merged_df[merged_df['Aneurysm Present'] == 1], \n                x='x', y='y', hue='location', palette='viridis', s=100, alpha=0.7)\nplt.title('Aneurysm Spatial Distribution (x-y plane)')\nplt.xlabel('X Coordinate')\nplt.ylabel('Y Coordinate')\nplt.legend(bbox_to_anchor=(1.05, 1), loc='upper left')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T15:50:01.12163Z","iopub.execute_input":"2025-07-30T15:50:01.12194Z","iopub.status.idle":"2025-07-30T15:50:01.766541Z","shell.execute_reply.started":"2025-07-30T15:50:01.121919Z","shell.execute_reply":"2025-07-30T15:50:01.765357Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Brain Location Frequency\n# Count aneurysms per brain location\nlocation_counts = merged_df[merged_df['Aneurysm Present'] == 1]['location'].value_counts()\n\n# Plot\nplt.figure(figsize=(12, 6))\nsns.barplot(x=location_counts.values, y=location_counts.index, palette='rocket')\nplt.title('Aneurysm Frequency by Brain Location')\nplt.xlabel('Count')\nplt.ylabel('Location')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T15:51:00.372379Z","iopub.execute_input":"2025-07-30T15:51:00.373053Z","iopub.status.idle":"2025-07-30T15:51:00.714592Z","shell.execute_reply.started":"2025-07-30T15:51:00.373015Z","shell.execute_reply":"2025-07-30T15:51:00.713347Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"merged_df.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T15:52:00.842585Z","iopub.execute_input":"2025-07-30T15:52:00.842935Z","iopub.status.idle":"2025-07-30T15:52:00.852335Z","shell.execute_reply.started":"2025-07-30T15:52:00.842911Z","shell.execute_reply":"2025-07-30T15:52:00.851141Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Age distribution by location\nplt.figure(figsize=(12, 6))\nsns.boxplot(data=merged_df[merged_df['Aneurysm Present'] == 1], \n            x='location', y='PatientAge', palette='Set3')\nplt.title('Age Distribution by Aneurysm Location')\nplt.xticks(rotation=45)\nplt.show()\n\n# Sex distribution by location\nplt.figure(figsize=(12, 6))\nsns.countplot(data=merged_df[merged_df['Aneurysm Present'] == 1], \n              x='location', hue='PatientSex', palette='Set2')\nplt.title('Sex Distribution by Aneurysm Location')\nplt.xticks(rotation=45)\nplt.legend(title='Sex')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T16:02:36.005159Z","iopub.execute_input":"2025-07-30T16:02:36.00556Z","iopub.status.idle":"2025-07-30T16:02:36.773517Z","shell.execute_reply.started":"2025-07-30T16:02:36.005531Z","shell.execute_reply":"2025-07-30T16:02:36.772351Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"---  \n### ❤️ Enjoyed the Notebook?\n\n**If you found this notebook useful, please consider giving it an _upvote_!**  \nIt helps others discover the work and keeps me motivated. 🚀  \n---\n","metadata":{}},{"cell_type":"markdown","source":"**— Happy Kaggling!**","metadata":{}},{"cell_type":"markdown","source":"🔧 Image analysis and feature engineering sections will be added soon — stay tuned!","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}