{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":99552,"databundleVersionId":13190393,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!ls -la /kaggle/input/rsna-intracranial-aneurysm-detection","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T03:05:15.963618Z","iopub.execute_input":"2025-07-30T03:05:15.964262Z","iopub.status.idle":"2025-07-30T03:05:16.133991Z","shell.execute_reply.started":"2025-07-30T03:05:15.964233Z","shell.execute_reply":"2025-07-30T03:05:16.132766Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import time\nimport os\nimport torch\nimport pandas as pd\nfrom skimage import io, transform\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import transforms, utils\nimport torch.nn.functional as F\n\nfrom scipy import ndimage\n\nimport pandas as pd\nfrom pathlib import Path\n\n# Ignore warnings\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\nplt.ion()   # interactive mode","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T04:56:09.485821Z","iopub.execute_input":"2025-07-30T04:56:09.486631Z","iopub.status.idle":"2025-07-30T04:56:09.495379Z","shell.execute_reply.started":"2025-07-30T04:56:09.486599Z","shell.execute_reply":"2025-07-30T04:56:09.494362Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Competition constants\nID_COL = 'SeriesInstanceUID'\nLABEL_COLS = [\n    'Left Infraclinoid Internal Carotid Artery',\n    'Right Infraclinoid Internal Carotid Artery',\n    'Left Supraclinoid Internal Carotid Artery',\n    'Right Supraclinoid Internal Carotid Artery',\n    'Left Middle Cerebral Artery',\n    'Right Middle Cerebral Artery',\n    'Anterior Communicating Artery',\n    'Left Anterior Cerebral Artery',\n    'Right Anterior Cerebral Artery',\n    'Left Posterior Communicating Artery',\n    'Right Posterior Communicating Artery',\n    'Basilar Tip',\n    'Other Posterior Circulation',\n    'Aneurysm Present',\n]\n\nDICOM_TAG_ALLOWLIST = [\n    'BitsAllocated', 'BitsStored', 'Columns', 'FrameOfReferenceUID', 'HighBit',\n    'ImageOrientationPatient', 'ImagePositionPatient', 'InstanceNumber', 'Modality',\n    'PatientID', 'PhotometricInterpretation', 'PixelRepresentation', 'PixelSpacing',\n    'PlanarConfiguration', 'RescaleIntercept', 'RescaleSlope', 'RescaleType', 'Rows',\n    'SOPClassUID', 'SOPInstanceUID', 'SamplesPerPixel', 'SliceThickness',\n    'SpacingBetweenSlices', 'StudyInstanceUID', 'TransferSyntaxUID',\n]\n\n# Model configuration\nTARGET_SIZE = (64, 64, 64)  # Reduced size for memory efficiency\nDEVICE = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T04:59:09.839227Z","iopub.execute_input":"2025-07-30T04:59:09.839613Z","iopub.status.idle":"2025-07-30T04:59:09.846702Z","shell.execute_reply.started":"2025-07-30T04:59:09.839588Z","shell.execute_reply":"2025-07-30T04:59:09.845718Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TRAIN = True                     # ← set to True when you want to train\nRAW_DIR = Path(\"/kaggle/input/rsna-intracranial-aneurysm-detection\")\nPRETRAINED_DIR = Path(\"/kaggle/input/\")# used when TRAIN=False\nSERIES_DIR = RAW_DIR / \"series\"\nEXPORT_DIR = Path(\"./\")                                    # artefacts will be saved here\nBATCH_SIZE = 64\nPAD_PERCENTILE = 95\nLR_INIT = 5e-4\nWD = 3e-3\nMIXUP_ALPHA = 0.4\nEPOCHS = 160\nPATIENCE = 40","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T03:05:23.665767Z","iopub.execute_input":"2025-07-30T03:05:23.66614Z","iopub.status.idle":"2025-07-30T03:05:23.671722Z","shell.execute_reply.started":"2025-07-30T03:05:23.666114Z","shell.execute_reply":"2025-07-30T03:05:23.670763Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from torchvision import transforms\n\nimage_size = 64\n\ndata_transforms = {\n    'train': transforms.Compose([\n        transforms.Resize((image_size, image_size)),\n        transforms.ToTensor(),\n        # add normalization if needed\n    ]),\n    'val': transforms.Compose([\n        transforms.Resize((image_size, image_size)),\n        transforms.ToTensor(),\n    ]),\n}\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T05:32:25.814677Z","iopub.execute_input":"2025-07-30T05:32:25.815156Z","iopub.status.idle":"2025-07-30T05:32:25.821382Z","shell.execute_reply.started":"2025-07-30T05:32:25.815062Z","shell.execute_reply":"2025-07-30T05:32:25.820364Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pydicom\nimport numpy as np\n\n# def collate_fn(batch):\n#     volumes = [item['images'] for item in batch]  # shape: (1, D, H, W)\n#     labels = [item['labels'] for item in batch]\n\n\n#     print(volumes[0].shape)\n#     volumes = torch.stack(volumes)\n#     labels = torch.stack(labels)  # shape: (batch, label_dim)\n\n#     print(volumes.shape)\n#     print(labels.shape)\n#     return {'images': volumes, 'labels': labels}\n\n\nclass RSNADataset(Dataset):\n    \"\"\"Face Landmarks dataset.\"\"\"\n\n    def __init__(self, df, series_dir = SERIES_DIR, transform=None, target_size = [64,64,64]):\n        \"\"\"\n        Arguments:\n            csv_file (string): Path to the csv file with annotations.\n             (string): Directory with all the images.\n            transform (callable, optional): Optional transform to be applied\n                on a sample.\n        \"\"\"\n        if isinstance(df, (str, Path)):\n            self.df = pd.read_csv(df)\n        elif isinstance(df, pd.DataFrame):\n            self.df = df\n        else:\n            raise TypeError(\"df must be either a file path or a pandas DataFrame\")\n\n        self.series_dir = SERIES_DIR\n        self.transform = transform\n        self.target_size = target_size\n\n    def __len__(self):\n        return len(self.df)\n        \n    def preprocess_volume(self, volume: np.ndarray) -> np.ndarray:\n        \"\"\"Preprocess 3D volume: normalize, clip, resize\"\"\"\n        # Handle potential issues\n        if volume.size == 0:\n            return np.zeros(self.target_size, dtype=np.float32)\n        \n        # Clip extreme values (robust to outliers)\n        p1, p99 = np.percentile(volume, [1, 99])\n        volume = np.clip(volume, p1, p99)\n        \n        # Normalize to [0, 1]\n        volume_min, volume_max = volume.min(), volume.max()\n        if volume_max > volume_min:\n            volume = (volume - volume_min) / (volume_max - volume_min)\n        \n        # Resize to target size\n        if volume.shape != self.target_size:\n            zoom_factors = [\n                self.target_size[i] / volume.shape[i] for i in range(3)\n            ]\n            volume = ndimage.zoom(volume, zoom_factors, order=1)\n        \n        return volume.astype(np.float32)\n\n    def __getitem__(self, idx):\n        if torch.is_tensor(idx):\n            idx = idx.tolist()\n\n        series_path = self.series_dir / self.df.iloc[idx, 0]\n        dicom_files = sorted(series_path.glob('**/*.dcm'))\n        \n        slices = []\n        for f in dicom_files:\n            img = pydicom.dcmread(str(f)).pixel_array.astype(np.float32)\n            # print(img.shape)\n            if img.ndim == 3:\n                img = np.squeeze(img)\n            img = Image.fromarray(img)\n            # print(img.size, img.mode)\n            if self.transform:\n                img = self.transform(img)  # Apply 2D transform\n            # print(img.shape)\n            img = img.squeeze(0)  \n            slices.append(img)\n        volume = np.stack(slices, axis=0) # (D, C, H, W) -> torch.Size([72, 1, 128, 128])\n        # print(slices)\n        volume = self.preprocess_volume(volume)\n        volume = torch.from_numpy(volume).unsqueeze(0) \n        # volume = volume.permute(1, 0, 2, 3) \n        # volume = torch.tensor(volume, dtype=torch.float32) # (B, C, D, H, W)\n        \n        labels = self.df.iloc[idx, 4:].values.astype(np.float32)\n        labels = torch.tensor(labels)\n        \n        sample = {'images': volume, 'labels': labels}\n        return sample","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T05:39:41.133291Z","iopub.execute_input":"2025-07-30T05:39:41.134108Z","iopub.status.idle":"2025-07-30T05:39:41.147627Z","shell.execute_reply.started":"2025-07-30T05:39:41.134036Z","shell.execute_reply":"2025-07-30T05:39:41.146463Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# fig = plt.figure(figsize=(15, 10))\n# columns = 5; rows = 4\n# for i in range(20):\n#     fig.add_subplot(rows, columns, i + 1)\n#     plt.imshow(sample['images'][0,i], cmap='gray')\n#     plt.axis('off')\n# plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T05:34:22.938486Z","iopub.execute_input":"2025-07-30T05:34:22.939141Z","iopub.status.idle":"2025-07-30T05:34:22.942908Z","shell.execute_reply.started":"2025-07-30T05:34:22.939111Z","shell.execute_reply":"2025-07-30T05:34:22.941852Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch.nn as nn\nimport torch.optim as optim\nfrom torch.optim import lr_scheduler\nimport torch.backends.cudnn as cudnn\nimport numpy as np\nimport torchvision\nfrom torchvision import datasets, models, transforms\nimport matplotlib.pyplot as plt\nimport time\nimport os\nfrom PIL import Image\nfrom tempfile import TemporaryDirectory","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T05:34:23.088078Z","iopub.execute_input":"2025-07-30T05:34:23.08845Z","iopub.status.idle":"2025-07-30T05:34:23.093902Z","shell.execute_reply.started":"2025-07-30T05:34:23.088422Z","shell.execute_reply":"2025-07-30T05:34:23.093029Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tqdm import tqdm\n\ndef train_model(model, criterion, optimizer, scheduler, num_epochs=25):\n    since = time.time()\n\n    # Create a temporary directory to save training checkpoints\n    with TemporaryDirectory() as tempdir:\n        best_model_params_path = os.path.join(tempdir, 'best_model_params.pt')\n\n        torch.save(model.state_dict(), best_model_params_path)\n        best_acc = 0.0\n\n        for epoch in range(num_epochs):\n            print(f'Epoch {epoch}/{num_epochs - 1}')\n            print('-' * 10)\n\n            # Each epoch has a training and validation phase\n            for phase in ['train', 'val']:\n                if phase == 'train':\n                    model.train()  # Set model to training mode\n                else:\n                    model.eval()   # Set model to evaluate mode\n\n                running_loss = 0.0\n                running_corrects = 0\n\n                # Iterate over data\n                                # Wrap dataloader with tqdm, show phase and epoch info in desc\n                dataloader = dataloaders[phase]\n                loop = tqdm(dataloader, desc=f\"{phase} Epoch {epoch}\", leave=False)\n\n                for batch in loop:\n                    inputs = batch['images'].to(device)\n                    labels = batch['labels'].to(device)\n                    # zero the parameter gradients\n                    optimizer.zero_grad()\n\n                    # forward\n                    # track history if only in train\n                    with torch.set_grad_enabled(phase == 'train'):\n                        outputs = model(inputs)\n                        loss = criterion(outputs, labels)\n                        preds = torch.sigmoid(outputs) > 0.5\n\n                        # backward + optimize only if in training phase\n                        if phase == 'train':\n                            loss.backward()\n                            optimizer.step()\n\n                    # statistics\n                    running_loss += loss.item() * inputs.size(0)\n                    running_corrects += torch.sum(preds == labels.data)\n                if phase == 'train':\n                    scheduler.step()\n\n                epoch_loss = running_loss / dataset_sizes[phase]\n                epoch_acc = running_corrects.double() / dataset_sizes[phase]\n\n                print(f'{phase} Loss: {epoch_loss:.4f} Acc: {epoch_acc:.4f}')\n\n                # deep copy the model\n                if phase == 'val' and epoch_acc > best_acc:\n                    best_acc = epoch_acc\n                    torch.save(model.state_dict(), best_model_params_path)\n\n            print()\n\n        time_elapsed = time.time() - since\n        print(f'Training complete in {time_elapsed // 60:.0f}m {time_elapsed % 60:.0f}s')\n        print(f'Best val Acc: {best_acc:4f}')\n\n        # load best model weights\n        model.load_state_dict(torch.load(best_model_params_path, weights_only=True))\n    return model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T05:34:23.23546Z","iopub.execute_input":"2025-07-30T05:34:23.235771Z","iopub.status.idle":"2025-07-30T05:34:23.246859Z","shell.execute_reply.started":"2025-07-30T05:34:23.235749Z","shell.execute_reply":"2025-07-30T05:34:23.245823Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# df = pd.read_csv(RAW_DIR / \"train.csv\")\n# train_ds = RSNADataset(train_df, transform=data_transforms['train'])\n# sample = train_ds[0]\n# sample['images'].shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T05:34:23.347303Z","iopub.execute_input":"2025-07-30T05:34:23.347654Z","iopub.status.idle":"2025-07-30T05:34:23.352166Z","shell.execute_reply.started":"2025-07-30T05:34:23.347633Z","shell.execute_reply":"2025-07-30T05:34:23.351216Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\ndf = pd.read_csv(RAW_DIR / \"train.csv\")\ntrain_df, val_df = train_test_split(df, test_size=0.2) # or last column\n\ntrain_ds = RSNADataset(train_df, transform=data_transforms['train'])\nval_ds = RSNADataset(val_df, transform=data_transforms['val'])\n\ntrain_dl = DataLoader(train_ds, batch_size=1, shuffle=True) #collate_fn=collate_fn)\nval_dl = DataLoader(val_ds, batch_size=1)#, collate_fn=collate_fn)\n\ndataloaders = {'train': train_dl, 'val': val_dl}\ndataset_sizes = {'train': len(train_ds), 'val': len(val_ds)}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T05:39:47.131888Z","iopub.execute_input":"2025-07-30T05:39:47.132204Z","iopub.status.idle":"2025-07-30T05:39:47.156568Z","shell.execute_reply.started":"2025-07-30T05:39:47.132182Z","shell.execute_reply":"2025-07-30T05:39:47.155532Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nclass Simple3DCNN(nn.Module):\n    \"\"\"Lightweight 3D CNN for aneurysm detection\"\"\"\n    \n    def __init__(self, num_classes: int = len(LABEL_COLS)):\n        super(Simple3DCNN, self).__init__()\n        \n        # 3D Convolutional layers\n        self.conv1 = nn.Conv3d(1, 16, kernel_size=3, padding=1)\n        self.pool1 = nn.MaxPool3d(2)\n        self.conv2 = nn.Conv3d(16, 32, kernel_size=3, padding=1)\n        self.pool2 = nn.MaxPool3d(2)\n        self.conv3 = nn.Conv3d(32, 64, kernel_size=3, padding=1)\n        self.pool3 = nn.MaxPool3d(2)\n        self.conv4 = nn.Conv3d(64, 128, kernel_size=3, padding=1)\n        self.pool4 = nn.MaxPool3d(2)\n        \n        # Adaptive pooling to handle variable sizes\n        self.adaptive_pool = nn.AdaptiveAvgPool3d((2, 2, 2))\n        \n        # Fully connected layers\n        self.fc1 = nn.Linear(128 * 2 * 2 * 2, 256)\n        self.dropout1 = nn.Dropout(0.5)\n        self.fc2 = nn.Linear(256, 128)\n        self.dropout2 = nn.Dropout(0.3)\n        self.fc3 = nn.Linear(128, num_classes)\n        \n        # Batch normalization\n        self.bn1 = nn.BatchNorm3d(16)\n        self.bn2 = nn.BatchNorm3d(32)\n        self.bn3 = nn.BatchNorm3d(64)\n        self.bn4 = nn.BatchNorm3d(128)\n        \n    def forward(self, x):\n        # Input shape: (batch_size, 1, depth, height, width)\n        x = self.pool1(F.relu(self.bn1(self.conv1(x))))\n        x = self.pool2(F.relu(self.bn2(self.conv2(x))))\n        x = self.pool3(F.relu(self.bn3(self.conv3(x))))\n        x = self.pool4(F.relu(self.bn4(self.conv4(x))))\n        \n        # Adaptive pooling\n        x = self.adaptive_pool(x)\n        \n        # Flatten\n        x = x.view(x.size(0), -1)\n        \n        # Fully connected layers\n        x = F.relu(self.fc1(x))\n        x = self.dropout1(x)\n        x = F.relu(self.fc2(x))\n        x = self.dropout2(x)\n        x = self.fc3(x)\n        \n        return torch.sigmoid(x)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T05:39:47.158293Z","iopub.execute_input":"2025-07-30T05:39:47.158618Z","iopub.status.idle":"2025-07-30T05:39:47.170195Z","shell.execute_reply.started":"2025-07-30T05:39:47.158594Z","shell.execute_reply":"2025-07-30T05:39:47.169196Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\nmodel = Simple3DCNN().to(device)\ncriterion = nn.BCEWithLogitsLoss()\noptimizer = optim.Adam(model.parameters(), lr=1e-4)\nscheduler = lr_scheduler.StepLR(optimizer, step_size=5, gamma=0.5)\n\n# trained_model = train_model(model, criterion, optimizer, scheduler, num_epochs=5)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T05:39:47.228989Z","iopub.execute_input":"2025-07-30T05:39:47.230196Z","iopub.status.idle":"2025-07-30T05:41:34.532896Z","shell.execute_reply.started":"2025-07-30T05:39:47.230161Z","shell.execute_reply":"2025-07-30T05:41:34.531472Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}