{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":99552,"databundleVersionId":13190393,"sourceType":"competition"}],"dockerImageVersionId":31091,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\n\n# List all files in the dataset folder\ndataset_path = \"/kaggle/input/rsna-intracranial-aneurysm-detection\"\nprint(\"Files in dataset:\", os.listdir(dataset_path))\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-08-01T10:44:46.019862Z","iopub.execute_input":"2025-08-01T10:44:46.02043Z","iopub.status.idle":"2025-08-01T10:44:46.024853Z","shell.execute_reply.started":"2025-08-01T10:44:46.020409Z","shell.execute_reply":"2025-08-01T10:44:46.024286Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install pydicom","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-01T10:44:50.11353Z","iopub.execute_input":"2025-08-01T10:44:50.113772Z","iopub.status.idle":"2025-08-01T10:44:53.98609Z","shell.execute_reply.started":"2025-08-01T10:44:50.113751Z","shell.execute_reply":"2025-08-01T10:44:53.985131Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nimport os\nimport pydicom\nimport matplotlib.pyplot as plt\n\n# Paths\ntrain_csv_path = \"/kaggle/input/rsna-intracranial-aneurysm-detection/train.csv\"\nimage_path = \"/kaggle/input/rsna-intracranial-aneurysm-detection/series\"\n\n# Load CSV\ntrain_df = pd.read_csv(train_csv_path)\n\n# Get patient folders\npatient_folders = os.listdir(image_path)\npatient_id = patient_folders[0]  # First patient\n\n# Get one DICOM slice\nslice_files = os.listdir(os.path.join(image_path, patient_id))\ndicom_path = os.path.join(image_path, patient_id, slice_files[len(slice_files)//2])\n\n# Read and display the DICOM image\nds = pydicom.dcmread(dicom_path)\nplt.imshow(ds.pixel_array, cmap=\"gray\")\nplt.axis('off')\nplt.title(f\"Patient: {patient_id}\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-01T10:44:55.855419Z","iopub.execute_input":"2025-08-01T10:44:55.855709Z","iopub.status.idle":"2025-08-01T10:44:57.143391Z","shell.execute_reply.started":"2025-08-01T10:44:55.855681Z","shell.execute_reply":"2025-08-01T10:44:57.142611Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_df.columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-01T10:45:00.606192Z","iopub.execute_input":"2025-08-01T10:45:00.606441Z","iopub.status.idle":"2025-08-01T10:45:00.610478Z","shell.execute_reply.started":"2025-08-01T10:45:00.606423Z","shell.execute_reply":"2025-08-01T10:45:00.60986Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"label_row = train_df[train_df[\"SeriesInstanceUID\"] == patient_id]\n\nif not label_row.empty:\n    has_aneurysm = label_row[\"Aneurysm Present\"].values[0]\n    print(\"Aneurysm Present\" if has_aneurysm else \"No Aneurysm Detected\")\nelse:\n    print(\"No label found for this SeriesInstanceUID\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-01T10:45:02.110401Z","iopub.execute_input":"2025-08-01T10:45:02.110665Z","iopub.status.idle":"2025-08-01T10:45:02.123713Z","shell.execute_reply.started":"2025-08-01T10:45:02.110645Z","shell.execute_reply":"2025-08-01T10:45:02.123014Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Look up the row in the CSV that matches this SeriesInstanceUID\nlabel_row = train_df[train_df[\"SeriesInstanceUID\"] == patient_id]\n\nif not label_row.empty:\n    left_mca = label_row[\"Left Middle Cerebral Artery\"].values[0]\n    \n    if left_mca == 1:\n        print(\"✅ Aneurysm Present in the Left Middle Cerebral Artery\")\n    else:\n        print(\"❌ No Aneurysm in the Left Middle Cerebral Artery\")\nelse:\n    print(\"SeriesInstanceUID not found in train.csv\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-01T10:45:05.17773Z","iopub.execute_input":"2025-08-01T10:45:05.17797Z","iopub.status.idle":"2025-08-01T10:45:05.185056Z","shell.execute_reply.started":"2025-08-01T10:45:05.177954Z","shell.execute_reply":"2025-08-01T10:45:05.184123Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"slices = [pydicom.dcmread(os.path.join(image_path, patient_id, s)) for s in slice_files]\nslices.sort(key=lambda x: float(x.ImagePositionPatient[2]))  # or x.InstanceNumber\nvolume = np.stack([s.pixel_array for s in slices], axis=0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-01T10:45:10.040166Z","iopub.execute_input":"2025-08-01T10:45:10.040939Z","iopub.status.idle":"2025-08-01T10:45:12.006536Z","shell.execute_reply.started":"2025-08-01T10:45:10.040906Z","shell.execute_reply":"2025-08-01T10:45:12.005952Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install scikit-image","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-01T10:44:39.365224Z","iopub.status.idle":"2025-08-01T10:44:39.365501Z","shell.execute_reply.started":"2025-08-01T10:44:39.365351Z","shell.execute_reply":"2025-08-01T10:44:39.365363Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from skimage.transform import resize\n\nvolume_resized = resize(volume, (64, 128, 128), preserve_range=True, anti_aliasing=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-01T10:45:13.813202Z","iopub.execute_input":"2025-08-01T10:45:13.813464Z","iopub.status.idle":"2025-08-01T10:45:14.866919Z","shell.execute_reply.started":"2025-08-01T10:45:13.813445Z","shell.execute_reply":"2025-08-01T10:45:14.866345Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"volume = np.clip(volume, -1000, 1000)\nvolume = (volume + 1000) / 2000  # scales to 0–1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-01T10:45:17.219206Z","iopub.execute_input":"2025-08-01T10:45:17.220037Z","iopub.status.idle":"2025-08-01T10:45:17.317307Z","shell.execute_reply.started":"2025-08-01T10:45:17.220015Z","shell.execute_reply":"2025-08-01T10:45:17.316697Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def apply_window(image, window_center=50, window_width=100):\n    min_val = window_center - (window_width / 2)\n    max_val = window_center + (window_width / 2)\n    windowed = np.clip(image, min_val, max_val)\n    normalized = (windowed - min_val) / (max_val - min_val)\n    return normalized","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-01T10:45:24.737059Z","iopub.execute_input":"2025-08-01T10:45:24.737296Z","iopub.status.idle":"2025-08-01T10:45:24.741738Z","shell.execute_reply.started":"2025-08-01T10:45:24.737281Z","shell.execute_reply":"2025-08-01T10:45:24.740933Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Assume 'volume' is a 3D NumPy array [D, H, W] loaded from DICOMs\nwindowed_volume = apply_window(volume, window_center=50, window_width=100)\n\n# To view a slice\nplt.imshow(windowed_volume[len(windowed_volume)//2], cmap='gray')\nplt.axis('off')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-01T10:45:28.570091Z","iopub.execute_input":"2025-08-01T10:45:28.570848Z","iopub.status.idle":"2025-08-01T10:45:28.788639Z","shell.execute_reply.started":"2025-08-01T10:45:28.57074Z","shell.execute_reply":"2025-08-01T10:45:28.787959Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport pydicom\nimport scipy.ndimage\nfrom skimage.transform import resize  # Optional if you prefer skimage resize\n\n# Define paths\nimage_path = \"/kaggle/input/rsna-intracranial-aneurysm-detection/series\"\ntrain_csv_path = \"/kaggle/input/rsna-intracranial-aneurysm-detection/train.csv\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-01T10:45:33.124226Z","iopub.execute_input":"2025-08-01T10:45:33.124476Z","iopub.status.idle":"2025-08-01T10:45:33.129454Z","shell.execute_reply.started":"2025-08-01T10:45:33.124459Z","shell.execute_reply":"2025-08-01T10:45:33.128891Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load CSV\ntrain_df = pd.read_csv(train_csv_path)\n\n# Apply windowing (optional, use if needed)\ndef apply_window(volume, window_center=50, window_width=100):\n    lower = window_center - window_width // 2\n    upper = window_center + window_width // 2\n    volume = np.clip(volume, lower, upper)\n    return (volume - lower) / (upper - lower)\n\n# Fast 3D resize using scipy\ndef resize_volume(volume, target_shape=(64, 128, 128)):\n    zoom_factors = [t / s for t, s in zip(target_shape, volume.shape)]\n    return scipy.ndimage.zoom(volume, zoom_factors, order=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-01T10:45:34.937784Z","iopub.execute_input":"2025-08-01T10:45:34.938038Z","iopub.status.idle":"2025-08-01T10:45:34.954571Z","shell.execute_reply.started":"2025-08-01T10:45:34.938021Z","shell.execute_reply":"2025-08-01T10:45:34.953861Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Initialize storage\nall_volumes = []\nall_labels = []","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-01T10:45:37.186947Z","iopub.execute_input":"2025-08-01T10:45:37.187741Z","iopub.status.idle":"2025-08-01T10:45:37.191156Z","shell.execute_reply.started":"2025-08-01T10:45:37.187712Z","shell.execute_reply":"2025-08-01T10:45:37.190388Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport pydicom\nimport scipy.ndimage\nfrom skimage.transform import resize\nimport matplotlib.pyplot as plt\n\n# --- Configuration ---\nimage_path = \"/kaggle/input/rsna-intracranial-aneurysm-detection/series\"\ntrain_csv_path = \"/kaggle/input/rsna-intracranial-aneurysm-detection/train.csv\"\nMAX_PATIENTS = 3000\nTARGET_SHAPE = (64, 128, 128)\n\n# --- Load CSV ---\ntrain_df = pd.read_csv(train_csv_path)\n\n# --- Helper Functions ---\ndef resize_volume(volume, target_shape=TARGET_SHAPE):\n    zoom_factors = [t / s for t, s in zip(target_shape, volume.shape)]\n    return scipy.ndimage.zoom(volume, zoom_factors, order=1)\n\ndef apply_window(volume, window_center=50, window_width=100):\n    lower = window_center - window_width // 2\n    upper = window_center + window_width // 2\n    volume = np.clip(volume, lower, upper)\n    return (volume - lower) / (upper - lower)\n\n# --- Initialize Storage ---\nall_volumes = []\nall_labels = []\n\n# --- Process Only First 1000 Patients ---\npatient_folders = sorted(os.listdir(image_path))[:MAX_PATIENTS]\n\nfor idx, patient_id in enumerate(patient_folders, 1):\n    patient_folder = os.path.join(image_path, patient_id)\n    if not os.path.isdir(patient_folder):\n        continue\n\n    slice_files = os.listdir(patient_folder)\n    slices = []\n\n    for s in slice_files:\n        dcm_path = os.path.join(patient_folder, s)\n        try:\n            dcm = pydicom.dcmread(dcm_path)\n            if hasattr(dcm, 'ImagePositionPatient'):\n                slices.append(dcm)\n        except Exception as e:\n            print(f\"Failed to read {dcm_path}: {e}\")\n\n    if len(slices) < 3:\n        print(f\"Skipping {patient_id}: Not enough valid slices\")\n        continue\n\n    try:\n        slices.sort(key=lambda x: float(x.ImagePositionPatient[2]))\n        volume = np.stack([s.pixel_array for s in slices], axis=0)\n        volume = resize_volume(volume)\n        volume = np.clip(volume, -1000, 1000)\n        volume = (volume + 1000) / 2000\n        volume = apply_window(volume, window_center=50, window_width=100)\n\n        # Get label\n        row = train_df[train_df[\"SeriesInstanceUID\"] == patient_id]\n        if row.empty:\n            print(f\"No label found for {patient_id}, skipping.\")\n            continue\n\n        label = row[\"Aneurysm Present\"].values[0]\n        all_volumes.append(volume)\n        all_labels.append(label)\n\n        print(f\"[{idx}/{MAX_PATIENTS}] ✅ Processed patient: {patient_id}\")\n\n    except Exception as e:\n        print(f\"❌ Error processing {patient_id}: {e}\")\n\n# --- Convert and Save ---\nall_volumes = np.array(all_volumes)\nall_labels = np.array(all_labels)\n\nnp.save(\"volumes_first1000.npy\", all_volumes)\nnp.save(\"labels_first1000.npy\", all_labels)\n\nprint(\"\\n🎉 Processing complete.\")\nprint(\"✅ Volumes shape:\", all_volumes.shape)\nprint(\"✅ Labels shape:\", all_labels.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-01T10:46:01.670273Z","iopub.execute_input":"2025-08-01T10:46:01.670568Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# --- Imports ---\nimport numpy as np\nimport tensorflow as tf\nfrom tensorflow.keras import layers, models\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report, accuracy_score\n\n# --- Debug: Print available devices ---\nprint(\"✅ Devices:\", tf.config.list_physical_devices())\n\n# --- Load Preprocessed Data ---\nX = np.load(\"/kaggle/working/volumes_first1000.npy\")  # Shape: (N, 64, 128, 128)\ny = np.load(\"/kaggle/working/labels_first1000.npy\")   # Shape: (N,)\n\n# --- Reshape and Normalize ---\nX = X[..., np.newaxis]  # Add channel dimension => (N, 64, 128, 128, 1)\nX = X.astype(\"float32\") / 255.0  # Normalize to [0,1]\n\n# --- Train-Test Split ---\nX_train, X_val, y_train, y_val = train_test_split(\n    X, y, test_size=0.2, stratify=y, random_state=42\n)\n\n# --- Define 3D CNN Model ---\nmodel = models.Sequential([\n    layers.Input(shape=(64, 128, 128, 1)),\n\n    layers.Conv3D(32, kernel_size=3, activation='relu', padding='same'),\n    layers.MaxPooling3D(pool_size=2),\n\n    layers.Conv3D(64, kernel_size=3, activation='relu', padding='same'),\n    layers.MaxPooling3D(pool_size=2),\n\n    layers.Conv3D(128, kernel_size=3, activation='relu', padding='same'),\n    layers.GlobalAveragePooling3D(),\n\n    layers.Dense(64, activation='relu'),\n    layers.Dropout(0.3),\n    layers.Dense(1, activation='sigmoid')\n])\n\n# --- Compile ---\nmodel.compile(optimizer='adam',\n              loss='binary_crossentropy',\n              metrics=['accuracy'])\n\n# --- Train ---\nhistory = model.fit(\n    X_train, y_train,\n    validation_data=(X_val, y_val),\n    epochs=10,\n    batch_size=8,  # Smaller batch to avoid memory issues\n    callbacks=[\n        tf.keras.callbacks.EarlyStopping(patience=3, restore_best_weights=True)\n    ],\n    verbose=2\n)\n\n# --- Evaluate ---\nval_preds = (model.predict(X_val) > 0.5).astype(\"int32\")\n\nprint(\"\\n🔍 Classification Report:\\n\")\nprint(classification_report(y_val, val_preds))\nprint(\"✅ Validation Accuracy:\", accuracy_score(y_val, val_preds))\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}