{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":99552,"databundleVersionId":13747926,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport pydicom\nimport matplotlib.pyplot as plt\nfrom pathlib import Path\nimport glob\nfrom IPython.display import display, HTML","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-09-16T15:35:07.988357Z","iopub.execute_input":"2025-09-16T15:35:07.988858Z","iopub.status.idle":"2025-09-16T15:35:07.994446Z","shell.execute_reply.started":"2025-09-16T15:35:07.988829Z","shell.execute_reply":"2025-09-16T15:35:07.993108Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Set up paths - adjust these to match your directory structure\nBASE_PATH = Path('/kaggle/input/rsna-intracranial-aneurysm-detection')\n\n# Load the train labels CSV\ntrain_df = pd.read_csv(BASE_PATH / 'train.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T15:35:07.996007Z","iopub.execute_input":"2025-09-16T15:35:07.996327Z","iopub.status.idle":"2025-09-16T15:35:08.032849Z","shell.execute_reply.started":"2025-09-16T15:35:07.996274Z","shell.execute_reply":"2025-09-16T15:35:08.03176Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Dataset shape: {train_df.shape}\")\nprint(f\"Columns: {train_df.columns.tolist()}\")\nprint(\"\\nFirst few rows:\")\ndisplay(train_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T15:35:08.033933Z","iopub.execute_input":"2025-09-16T15:35:08.034201Z","iopub.status.idle":"2025-09-16T15:35:08.050712Z","shell.execute_reply.started":"2025-09-16T15:35:08.03418Z","shell.execute_reply":"2025-09-16T15:35:08.049547Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Verify the correct directory structure\nprint(\"Verifying RSNA dataset structure...\")\nprint(\"=\" * 60)\n\n# Check for the series directory\nseries_dir = BASE_PATH / 'series'\nif series_dir.exists():\n    series_folders = list(series_dir.iterdir())\n    print(f\"Found 'series' directory with {len(series_folders)} series folders\")\n    \n    # Show a few examples\n    print(f\"\\nExample series folders:\")\n    for folder in series_folders[:3]:\n        if folder.is_dir():\n            dcm_files = list(folder.glob('*.dcm'))\n            print(f\"  📁 {folder.name[:50]}...\")\n            print(f\"     └── Contains {len(dcm_files)} DICOM files\")\nelse:\n    print(\"'series' directory not found\")\n\nprint(\"=\" * 60)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T15:35:08.052902Z","iopub.execute_input":"2025-09-16T15:35:08.053504Z","iopub.status.idle":"2025-09-16T15:35:08.088106Z","shell.execute_reply.started":"2025-09-16T15:35:08.053477Z","shell.execute_reply":"2025-09-16T15:35:08.086866Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 5: Updated DICOM loader with correct path structure\ndef load_dicom_series(series_uid, base_path=BASE_PATH):\n    \"\"\"Load all DICOM files for a given SeriesInstanceUID.\"\"\"\n    \n    # Construct the correct path\n    series_path = base_path / 'series' / series_uid\n    \n    if not series_path.exists():\n        print(f\"Series not found at: {series_path}\")\n        return []\n    \n    # Get all DICOM files in the series directory\n    dcm_files = sorted(series_path.glob('*.dcm'))\n    \n    print(f\"Found {len(dcm_files)} DICOM files for series {series_uid[:30]}...\")\n    \n    # Load the DICOM files\n    dicoms = []\n    for dcm_file in dcm_files:\n        try:\n            ds = pydicom.dcmread(dcm_file)\n            dicoms.append(ds)\n        except Exception as e:\n            print(f\"Error reading {dcm_file.name}: {e}\")\n    \n    print(f\"Successfully loaded {len(dicoms)} DICOM files\")\n    return dicoms","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T15:35:08.08901Z","iopub.execute_input":"2025-09-16T15:35:08.089247Z","iopub.status.idle":"2025-09-16T15:35:08.096084Z","shell.execute_reply.started":"2025-09-16T15:35:08.089226Z","shell.execute_reply":"2025-09-16T15:35:08.094986Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Test with the first series from our CSV\nsample_series_uid = train_df['SeriesInstanceUID'].iloc[0]\nprint(f\"Loading series: {sample_series_uid}\")\nsample_dicoms = load_dicom_series(sample_series_uid)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T15:35:08.096999Z","iopub.execute_input":"2025-09-16T15:35:08.097253Z","iopub.status.idle":"2025-09-16T15:35:08.478493Z","shell.execute_reply.started":"2025-09-16T15:35:08.097232Z","shell.execute_reply":"2025-09-16T15:35:08.477241Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 6: Display DICOM metadata and basic info\ndef analyze_dicom_series(dicoms):\n    \"\"\"Analyze and display information about a DICOM series.\"\"\"\n    \n    if not dicoms:\n        print(\"No DICOM files to analyze\")\n        return\n    \n    # Get metadata from first DICOM\n    first_dcm = dicoms[0]\n    \n    print(\"=\" * 60)\n    print(\"DICOM SERIES INFORMATION\")\n    print(\"=\" * 60)\n    \n    # Basic metadata\n    metadata = {\n        'Number of slices': len(dicoms),\n        'Patient ID': getattr(first_dcm, 'PatientID', 'N/A'),\n        'Patient Age': getattr(first_dcm, 'PatientAge', 'N/A'),\n        'Patient Sex': getattr(first_dcm, 'PatientSex', 'N/A'),\n        'Study Date': getattr(first_dcm, 'StudyDate', 'N/A'),\n        'Modality': getattr(first_dcm, 'Modality', 'N/A'),\n        'Manufacturer': getattr(first_dcm, 'Manufacturer', 'N/A'),\n        'Slice Thickness': f\"{getattr(first_dcm, 'SliceThickness', 'N/A')} mm\",\n        'Image Size': f\"{getattr(first_dcm, 'Rows', 'N/A')} x {getattr(first_dcm, 'Columns', 'N/A')}\",\n    }\n    \n    for key, value in metadata.items():\n        print(f\"{key:20}: {value}\")\n    \n    # Window settings\n    window_center = getattr(first_dcm, 'WindowCenter', None)\n    window_width = getattr(first_dcm, 'WindowWidth', None)\n    \n    if window_center is not None and window_width is not None:\n        # Handle cases where these might be multi-valued\n        if isinstance(window_center, pydicom.multival.MultiValue):\n            window_center = window_center[0]\n        if isinstance(window_width, pydicom.multival.MultiValue):\n            window_width = window_width[0]\n        \n        print(f\"Window Center       : {window_center}\")\n        print(f\"Window Width        : {window_width}\")\n    \n    # Pixel array info\n    try:\n        pixel_array = first_dcm.pixel_array\n        print(f\"\\nPixel Array Info:\")\n        print(f\"  Shape             : {pixel_array.shape}\")\n        print(f\"  Data type         : {pixel_array.dtype}\")\n        print(f\"  Min value         : {pixel_array.min()}\")\n        print(f\"  Max value         : {pixel_array.max()}\")\n    except Exception as e:\n        print(f\"\\nCould not access pixel array: {e}\")\n    \n    return metadata","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T15:35:08.480456Z","iopub.execute_input":"2025-09-16T15:35:08.480746Z","iopub.status.idle":"2025-09-16T15:35:08.490762Z","shell.execute_reply.started":"2025-09-16T15:35:08.480723Z","shell.execute_reply":"2025-09-16T15:35:08.48954Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"metadata = analyze_dicom_series(sample_dicoms)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T15:35:08.492036Z","iopub.execute_input":"2025-09-16T15:35:08.492344Z","iopub.status.idle":"2025-09-16T15:35:08.530983Z","shell.execute_reply.started":"2025-09-16T15:35:08.492313Z","shell.execute_reply":"2025-09-16T15:35:08.529934Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 7: Visualization functions for DICOM images\ndef apply_windowing(img, window_center, window_width):\n    \"\"\"Apply windowing to improve contrast for visualization.\"\"\"\n    img_min = window_center - window_width // 2\n    img_max = window_center + window_width // 2\n    \n    img_windowed = img.copy()\n    img_windowed[img_windowed < img_min] = img_min\n    img_windowed[img_windowed > img_max] = img_max\n    \n    # Normalize to 0-255\n    img_windowed = ((img_windowed - img_min) / (img_max - img_min) * 255).astype(np.uint8)\n    \n    return img_windowed\n\ndef get_pixel_array_hu(dicom_dataset):\n    \"\"\"Convert pixel array to Hounsfield Units and apply windowing.\"\"\"\n    # Get pixel array\n    image = dicom_dataset.pixel_array.astype(float)\n    \n    # Convert to Hounsfield Units (HU)\n    rescale_slope = getattr(dicom_dataset, 'RescaleSlope', 1)\n    rescale_intercept = getattr(dicom_dataset, 'RescaleIntercept', 0)\n    \n    if rescale_slope != 1 or rescale_intercept != 0:\n        image = image * rescale_slope + rescale_intercept\n    \n    # Get window settings (use brain window as default)\n    window_center = getattr(dicom_dataset, 'WindowCenter', 40)\n    window_width = getattr(dicom_dataset, 'WindowWidth', 80)\n    \n    # Handle multi-valued window settings\n    if isinstance(window_center, pydicom.multival.MultiValue):\n        window_center = float(window_center[0])\n    if isinstance(window_width, pydicom.multival.MultiValue):\n        window_width = float(window_width[0])\n    \n    # Apply windowing\n    img_windowed = apply_windowing(image, window_center, window_width)\n    \n    return img_windowed, image  # Return both windowed and HU values","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T15:35:08.532085Z","iopub.execute_input":"2025-09-16T15:35:08.532487Z","iopub.status.idle":"2025-09-16T15:35:08.55067Z","shell.execute_reply.started":"2025-09-16T15:35:08.532388Z","shell.execute_reply":"2025-09-16T15:35:08.549398Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Test visualization with a single slice\nif sample_dicoms:\n    middle_idx = len(sample_dicoms) // 2\n    img_windowed, img_hu = get_pixel_array_hu(sample_dicoms[middle_idx])\n    \n    fig, axes = plt.subplots(1, 2, figsize=(12, 5))\n    \n    # Windowed image\n    axes[0].imshow(img_windowed, cmap='gray')\n    axes[0].set_title(f'Windowed Image (Slice {middle_idx + 1}/{len(sample_dicoms)})')\n    axes[0].axis('off')\n    \n    # Histogram of HU values\n    axes[1].hist(img_hu.flatten(), bins=50, edgecolor='black')\n    axes[1].set_xlabel('Hounsfield Units (HU)')\n    axes[1].set_ylabel('Frequency')\n    axes[1].set_title('Distribution of HU Values')\n    axes[1].grid(True, alpha=0.3)\n    \n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T15:35:08.551591Z","iopub.execute_input":"2025-09-16T15:35:08.551956Z","iopub.status.idle":"2025-09-16T15:35:09.075927Z","shell.execute_reply.started":"2025-09-16T15:35:08.551925Z","shell.execute_reply":"2025-09-16T15:35:09.074958Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 8: Display multiple slices from a series\ndef plot_series_slices(dicoms, num_slices=16, cols=4):\n    \"\"\"Display multiple slices from a DICOM series.\"\"\"\n    \n    if not dicoms:\n        print(\"No DICOM files to display\")\n        return\n    \n    num_slices = min(num_slices, len(dicoms))\n    rows = (num_slices + cols - 1) // cols\n    \n    # Sample evenly across the series\n    indices = np.linspace(0, len(dicoms) - 1, num_slices, dtype=int)\n    \n    fig, axes = plt.subplots(rows, cols, figsize=(cols * 3, rows * 3))\n    axes = axes.flatten() if num_slices > 1 else [axes]\n    \n    for i, idx in enumerate(indices):\n        if i < len(axes):\n            try:\n                img_windowed, _ = get_pixel_array_hu(dicoms[idx])\n                axes[i].imshow(img_windowed, cmap='gray')\n                axes[i].set_title(f'Slice {idx + 1}/{len(dicoms)}')\n                axes[i].axis('off')\n            except Exception as e:\n                axes[i].text(0.5, 0.5, f'Error: {str(e)[:30]}', \n                           ha='center', va='center')\n                axes[i].axis('off')\n    \n    # Hide unused subplots\n    for i in range(num_slices, len(axes)):\n        axes[i].axis('off')\n    \n    # Add series info to the title\n    series_uid = train_df['SeriesInstanceUID'].iloc[0]\n    series_info = train_df[train_df['SeriesInstanceUID'] == series_uid].iloc[0]\n    aneurysm_status = \"WITH Aneurysm\" if series_info['Aneurysm Present'] else \"NO Aneurysm\"\n    \n    plt.suptitle(f'Series: {series_uid[:30]}... - {aneurysm_status}\\n'\n                 f'Modality: {series_info[\"Modality\"]} | '\n                 f'Age: {series_info[\"PatientAge\"]} | '\n                 f'Sex: {series_info[\"PatientSex\"]}',\n                 fontsize=12)\n    \n    plt.tight_layout()\n    plt.savefig('multi_slice_analysis.png')\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T15:35:09.077104Z","iopub.execute_input":"2025-09-16T15:35:09.078197Z","iopub.status.idle":"2025-09-16T15:35:09.088599Z","shell.execute_reply.started":"2025-09-16T15:35:09.078163Z","shell.execute_reply":"2025-09-16T15:35:09.0876Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_series_slices(sample_dicoms, num_slices=16)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T15:35:09.089593Z","iopub.execute_input":"2025-09-16T15:35:09.089875Z","iopub.status.idle":"2025-09-16T15:35:12.78851Z","shell.execute_reply.started":"2025-09-16T15:35:09.089854Z","shell.execute_reply":"2025-09-16T15:35:12.787378Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 9: Compare positive and negative cases\ndef compare_aneurysm_cases(train_df, num_slices_per_series=4):\n    \"\"\"Display side-by-side comparison of cases with and without aneurysms.\"\"\"\n    \n    # Get one positive and one negative case\n    positive_series = train_df[train_df['Aneurysm Present'] == 1]['SeriesInstanceUID'].iloc[0]\n    negative_series = train_df[train_df['Aneurysm Present'] == 0]['SeriesInstanceUID'].iloc[0]\n    \n    print(f\"Loading positive case: {positive_series[:30]}...\")\n    pos_dicoms = load_dicom_series(positive_series)\n    \n    print(f\"Loading negative case: {negative_series[:30]}...\")\n    neg_dicoms = load_dicom_series(negative_series)\n    \n    if not pos_dicoms or not neg_dicoms:\n        print(\"Could not load both series for comparison\")\n        return\n    \n    # Create comparison figure\n    fig, axes = plt.subplots(2, num_slices_per_series, figsize=(num_slices_per_series * 3, 6))\n    \n    # Sample slices evenly through each series\n    pos_indices = np.linspace(0, len(pos_dicoms) - 1, num_slices_per_series, dtype=int)\n    neg_indices = np.linspace(0, len(neg_dicoms) - 1, num_slices_per_series, dtype=int)\n    \n    # Display positive case slices\n    for i, idx in enumerate(pos_indices):\n        img, _ = get_pixel_array_hu(pos_dicoms[idx])\n        axes[0, i].imshow(img, cmap='gray')\n        axes[0, i].set_title(f'Slice {idx + 1}/{len(pos_dicoms)}')\n        axes[0, i].axis('off')\n        if i == 0:\n            axes[0, i].set_ylabel('WITH\\nAneurysm', rotation=0, ha='right', va='center')\n    \n    # Display negative case slices\n    for i, idx in enumerate(neg_indices):\n        img, _ = get_pixel_array_hu(neg_dicoms[idx])\n        axes[1, i].imshow(img, cmap='gray')\n        axes[1, i].set_title(f'Slice {idx + 1}/{len(neg_dicoms)}')\n        axes[1, i].axis('off')\n        if i == 0:\n            axes[1, i].set_ylabel('NO\\nAneurysm', rotation=0, ha='right', va='center')\n    \n    # Get additional information\n    pos_info = train_df[train_df['SeriesInstanceUID'] == positive_series].iloc[0]\n    neg_info = train_df[train_df['SeriesInstanceUID'] == negative_series].iloc[0]\n    \n    # Find affected arteries for positive case\n    artery_cols = [col for col in train_df.columns if col not in \n                   ['SeriesInstanceUID', 'PatientAge', 'PatientSex', 'Modality', 'Aneurysm Present']]\n    affected = [col for col in artery_cols if pos_info[col] == 1]\n    \n    plt.suptitle(f'Comparison: Aneurysm Cases\\n'\n                 f'Positive: {pos_info[\"Modality\"]}, Age {pos_info[\"PatientAge\"]}, {pos_info[\"PatientSex\"]} '\n                 f'- Affected: {\", \".join(affected[:2]) if affected else \"Unknown\"}\\n'\n                 f'Negative: {neg_info[\"Modality\"]}, Age {neg_info[\"PatientAge\"]}, {neg_info[\"PatientSex\"]}',\n                 fontsize=11)\n    \n    plt.tight_layout()\n    plt.savefig('aneurysm_case_comparison.png')\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T15:35:12.792038Z","iopub.execute_input":"2025-09-16T15:35:12.792414Z","iopub.status.idle":"2025-09-16T15:35:12.810257Z","shell.execute_reply.started":"2025-09-16T15:35:12.792387Z","shell.execute_reply":"2025-09-16T15:35:12.809244Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"compare_aneurysm_cases(train_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T15:35:12.811265Z","iopub.execute_input":"2025-09-16T15:35:12.811608Z","iopub.status.idle":"2025-09-16T15:35:15.558592Z","shell.execute_reply.started":"2025-09-16T15:35:12.811569Z","shell.execute_reply":"2025-09-16T15:35:15.557539Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 10: Analyze series with multiple aneurysm locations\ndef analyze_multi_location_cases(train_df):\n    \"\"\"Find and analyze cases with aneurysms in multiple locations.\"\"\"\n    \n    # Get artery columns\n    artery_cols = [col for col in train_df.columns if col not in \n                   ['SeriesInstanceUID', 'PatientAge', 'PatientSex', 'Modality', 'Aneurysm Present']]\n    \n    # Count aneurysm locations per case\n    positive_cases = train_df[train_df['Aneurysm Present'] == 1].copy()\n    positive_cases['num_locations'] = positive_cases[artery_cols].sum(axis=1)\n    \n    # Find cases with multiple locations\n    multi_location = positive_cases[positive_cases['num_locations'] > 1]\n    \n    print(f\"Cases with multiple aneurysm locations: {len(multi_location)} out of {len(positive_cases)} positive cases\")\n    print(f\"Percentage: {len(multi_location) / len(positive_cases) * 100:.1f}%\")\n    \n    # Display distribution\n    location_counts = positive_cases['num_locations'].value_counts().sort_index()\n    \n    plt.figure(figsize=(10, 5))\n    \n    # Bar chart of location counts\n    plt.subplot(1, 2, 1)\n    location_counts.plot(kind='bar', color='mediumpurple')\n    plt.xlabel('Number of Aneurysm Locations')\n    plt.ylabel('Number of Cases')\n    plt.title('Distribution of Aneurysm Location Counts')\n    plt.grid(True, alpha=0.3, axis='y')\n    \n    # Show an example case with multiple locations\n    if len(multi_location) > 0:\n        example = multi_location.iloc[0]\n        affected = [col for col in artery_cols if example[col] == 1]\n        \n        plt.subplot(1, 2, 2)\n        plt.text(0.1, 0.9, f\"Example Multi-Location Case:\", fontsize=12, fontweight='bold')\n        plt.text(0.1, 0.8, f\"Series: {example['SeriesInstanceUID'][:40]}...\", fontsize=10)\n        plt.text(0.1, 0.7, f\"Age: {example['PatientAge']}, Sex: {example['PatientSex']}\", fontsize=10)\n        plt.text(0.1, 0.6, f\"Modality: {example['Modality']}\", fontsize=10)\n        plt.text(0.1, 0.5, f\"Number of locations: {len(affected)}\", fontsize=10)\n        \n        plt.text(0.1, 0.35, \"Affected Arteries:\", fontsize=11, fontweight='bold')\n        for i, artery in enumerate(affected):\n            plt.text(0.1, 0.25 - i*0.08, f\"  • {artery}\", fontsize=9)\n        \n        plt.xlim(0, 1)\n        plt.ylim(0, 1)\n        plt.axis('off')\n    \n    plt.tight_layout()\n    plt.savefig('analyze_multi_location_cases.png')\n    plt.show()\n    \n    return multi_location","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T15:35:15.560002Z","iopub.execute_input":"2025-09-16T15:35:15.560287Z","iopub.status.idle":"2025-09-16T15:35:15.572696Z","shell.execute_reply.started":"2025-09-16T15:35:15.560265Z","shell.execute_reply":"2025-09-16T15:35:15.571515Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"multi_location_cases = analyze_multi_location_cases(train_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T15:35:15.573678Z","iopub.execute_input":"2025-09-16T15:35:15.573909Z","iopub.status.idle":"2025-09-16T15:35:16.038493Z","shell.execute_reply.started":"2025-09-16T15:35:15.57389Z","shell.execute_reply":"2025-09-16T15:35:16.037457Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 11: Dataset summary statistics and visualizations\ndef create_comprehensive_analysis(train_df):\n    \"\"\"Create a comprehensive analysis dashboard.\"\"\"\n    \n    fig = plt.figure(figsize=(16, 10))\n    \n    # 1. Aneurysm prevalence\n    ax1 = plt.subplot(3, 4, 1)\n    aneurysm_counts = train_df['Aneurysm Present'].value_counts()\n    colors = ['#3498db', '#e74c3c']\n    ax1.pie(aneurysm_counts.values, labels=['No Aneurysm', 'Aneurysm'], \n            autopct='%1.1f%%', colors=colors, startangle=90)\n    ax1.set_title('Overall Aneurysm Prevalence', fontsize=11, fontweight='bold')\n    \n    # 2. Age distribution\n    ax2 = plt.subplot(3, 4, 2)\n    train_df['PatientAge'].hist(bins=20, color='#2ecc71', edgecolor='black', ax=ax2)\n    ax2.set_xlabel('Age (years)')\n    ax2.set_ylabel('Count')\n    ax2.set_title('Age Distribution', fontsize=11, fontweight='bold')\n    ax2.grid(True, alpha=0.3)\n    \n    # 3. Sex distribution\n    ax3 = plt.subplot(3, 4, 3)\n    sex_counts = train_df['PatientSex'].value_counts()\n    sex_counts.plot(kind='bar', color=['#9b59b6', '#f39c12'], ax=ax3)\n    ax3.set_xlabel('Sex')\n    ax3.set_ylabel('Count')\n    ax3.set_title('Sex Distribution', fontsize=11, fontweight='bold')\n    ax3.grid(True, alpha=0.3, axis='y')\n    \n    # 4. Modality distribution\n    ax4 = plt.subplot(3, 4, 4)\n    modality_counts = train_df['Modality'].value_counts()\n    modality_counts.plot(kind='bar', color=['#1abc9c', '#34495e'], ax=ax4)\n    ax4.set_xlabel('Modality')\n    ax4.set_ylabel('Count')\n    ax4.set_title('Imaging Modality', fontsize=11, fontweight='bold')\n    ax4.grid(True, alpha=0.3, axis='y')\n    \n    # 5. Age by aneurysm status\n    ax5 = plt.subplot(3, 4, 5)\n    train_df[train_df['Aneurysm Present']==0]['PatientAge'].hist(bins=15, alpha=0.6, \n                                                                  label='No Aneurysm', \n                                                                  color='#3498db', ax=ax5)\n    train_df[train_df['Aneurysm Present']==1]['PatientAge'].hist(bins=15, alpha=0.6, \n                                                                  label='Aneurysm', \n                                                                  color='#e74c3c', ax=ax5)\n    ax5.set_xlabel('Age (years)')\n    ax5.set_ylabel('Count')\n    ax5.set_title('Age by Aneurysm Status', fontsize=11, fontweight='bold')\n    ax5.legend()\n    ax5.grid(True, alpha=0.3)\n    \n    # 6. Top aneurysm locations\n    ax6 = plt.subplot(3, 4, (6, 10))  # Span multiple cells\n    artery_cols = [col for col in train_df.columns if col not in \n                   ['SeriesInstanceUID', 'PatientAge', 'PatientSex', 'Modality', 'Aneurysm Present']]\n    \n    location_counts = {}\n    for col in artery_cols:\n        count = train_df[col].sum()\n        if count > 0:\n            # Shorten names for display\n            short_name = col.replace('Internal Carotid Artery', 'ICA')\\\n                           .replace('Middle Cerebral Artery', 'MCA')\\\n                           .replace('Anterior Cerebral Artery', 'ACA')\\\n                           .replace('Posterior Communicating Artery', 'PComA')\\\n                           .replace('Anterior Communicating Artery', 'AComA')\n            location_counts[short_name] = count\n    \n    locations = list(location_counts.keys())\n    counts = list(location_counts.values())\n    \n    y_pos = np.arange(len(locations))\n    ax6.barh(y_pos, counts, color='#8e44ad')\n    ax6.set_yticks(y_pos)\n    ax6.set_yticklabels(locations, fontsize=9)\n    ax6.set_xlabel('Number of Cases')\n    ax6.set_title('Aneurysm Locations', fontsize=11, fontweight='bold')\n    ax6.grid(True, alpha=0.3, axis='x')\n    \n    # 7. Sex vs Aneurysm\n    ax7 = plt.subplot(3, 4, 7)\n    sex_aneurysm = pd.crosstab(train_df['PatientSex'], train_df['Aneurysm Present'], normalize='index') * 100\n    sex_aneurysm.plot(kind='bar', stacked=True, color=['#3498db', '#e74c3c'], ax=ax7)\n    ax7.set_xlabel('Sex')\n    ax7.set_ylabel('Percentage (%)')\n    ax7.set_title('Aneurysm Rate by Sex', fontsize=11, fontweight='bold')\n    ax7.legend(['No Aneurysm', 'Aneurysm'], loc='upper right')\n    ax7.grid(True, alpha=0.3, axis='y')\n    \n    # 8. Modality vs Aneurysm\n    ax8 = plt.subplot(3, 4, 8)\n    mod_aneurysm = pd.crosstab(train_df['Modality'], train_df['Aneurysm Present'], normalize='index') * 100\n    mod_aneurysm.plot(kind='bar', stacked=True, color=['#3498db', '#e74c3c'], ax=ax8)\n    ax8.set_xlabel('Modality')\n    ax8.set_ylabel('Percentage (%)')\n    ax8.set_title('Aneurysm Rate by Modality', fontsize=11, fontweight='bold')\n    ax8.legend(['No Aneurysm', 'Aneurysm'], loc='upper right')\n    ax8.grid(True, alpha=0.3, axis='y')\n    \n    # Add summary statistics as text\n    ax9 = plt.subplot(3, 4, (11, 12))\n    ax9.axis('off')\n    \n    summary_text = f\"\"\"\n    SUMMARY STATISTICS\n    ==================\n    Total Series: {len(train_df):,}\n    Positive Cases: {train_df['Aneurysm Present'].sum():,} ({train_df['Aneurysm Present'].mean()*100:.1f}%)\n    \n    Age Range: {train_df['PatientAge'].min()}-{train_df['PatientAge'].max()} years\n    Mean Age: {train_df['PatientAge'].mean():.1f} ± {train_df['PatientAge'].std():.1f}\n    \n    Most Common Location:\n    {artery_cols[train_df[artery_cols].sum().argmax()]}\n    ({train_df[artery_cols].sum().max()} cases)\n    \"\"\"\n    \n    ax9.text(0.1, 0.5, summary_text, fontsize=10, family='monospace', va='center')\n    \n    plt.suptitle('RSNA Intracranial Aneurysm Detection - Dataset Analysis', fontsize=14, fontweight='bold')\n    plt.tight_layout()\n    plt.savefig('dataset_analysis.png')\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T15:35:16.039545Z","iopub.execute_input":"2025-09-16T15:35:16.039807Z","iopub.status.idle":"2025-09-16T15:35:16.061443Z","shell.execute_reply.started":"2025-09-16T15:35:16.03978Z","shell.execute_reply":"2025-09-16T15:35:16.060456Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"create_comprehensive_analysis(train_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T15:35:16.062571Z","iopub.execute_input":"2025-09-16T15:35:16.063079Z","iopub.status.idle":"2025-09-16T15:35:18.435936Z","shell.execute_reply.started":"2025-09-16T15:35:16.063048Z","shell.execute_reply":"2025-09-16T15:35:18.434878Z"}},"outputs":[],"execution_count":null}]}