{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":99552,"databundleVersionId":13851420,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import gc\nimport torch\nimport kaggle_evaluation.rsna_inference_server\n\ngc.collect()\ntorch.cuda.empty_cache()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T16:57:52.508946Z","iopub.execute_input":"2025-08-05T16:57:52.509242Z","iopub.status.idle":"2025-08-05T16:57:58.312064Z","shell.execute_reply.started":"2025-08-05T16:57:52.509189Z","shell.execute_reply":"2025-08-05T16:57:58.311532Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Imports\nimport pydicom\nimport os\nfrom collections import defaultdict\nimport numpy as np\nimport torch\nimport torch.nn as nn\nfrom torch.utils.data import Dataset, DataLoader\nimport torchvision.transforms as T\nimport torch.nn.functional as F\nimport torch.optim as optim\nfrom skimage.transform import resize\n!pip install tqdm\nfrom tqdm import tqdm\nimport random\nfrom glob import glob\n\n\n#Generate masks\nimport pandas as pd\nimport pydicom\nfrom tqdm import tqdm\nimport cv2\n\n\nos.environ[\"PYTORCH_CUDA_ALLOC_CONF\"] = \"expandable_segments:True\"\nos.environ[\"PYTORCH_CUDA_ALLOC_CONF\"] = \"max_split_size_mb:128\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T16:57:58.313147Z","iopub.execute_input":"2025-08-05T16:57:58.313482Z","iopub.status.idle":"2025-08-05T16:58:07.121969Z","shell.execute_reply.started":"2025-08-05T16:57:58.313458Z","shell.execute_reply":"2025-08-05T16:58:07.121266Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install nibabel\n\nimport os\nimport nibabel as nib\nimport numpy as np\n\n# Config\nseg_root = \"/kaggle/input/rsna-intracranial-aneurysm-detection/segmentations\"  # path to your .nii.gz segmentation folder\nout_root = \"/kaggle/working/masks_npy\"\nos.makedirs(out_root, exist_ok=True)\n\n# Define label map for the 13 locations (use same index order as your train.csv)\nLABEL_MAP = {\n    \"Left Infraclinoid Internal Carotid Artery\": 0,\n    \"Right Infraclinoid Internal Carotid Artery\": 1,\n    \"Left Supraclinoid Internal Carotid Artery\": 2,\n    \"Right Supraclinoid Internal Carotid Artery\": 3,\n    \"Left Middle Cerebral Artery\": 4,\n    \"Right Middle Cerebral Artery\": 5,\n    \"Anterior Communicating Artery\": 6,\n    \"Left Anterior Cerebral Artery\": 7,\n    \"Right Anterior Cerebral Artery\": 8,\n    \"Left Posterior Communicating Artery\": 9,\n    \"Right Posterior Communicating Artery\": 10,\n    \"Basilar Tip\": 11,\n    \"Other Posterior Circulation\": 12,\n}\n\nfor fname in os.listdir(seg_root):\n    if not fname.endswith(\".nii.gz\"):\n        continue\n\n    series_uid = fname.replace(\".nii.gz\", \"\")\n    path = os.path.join(seg_root, fname)\n\n    # Load segmentation volume\n    seg_nii = nib.load(path)\n    seg_data = seg_nii.get_fdata()  # shape: [H, W, D]\n\n    # Transpose to [D, H, W] (if needed)\n    seg_data = np.transpose(seg_data, (2, 0, 1))\n\n    # One-hot encode to [13, D, H, W]\n    mask = np.zeros((13, *seg_data.shape), dtype=np.uint8)\n    for label_name, class_idx in LABEL_MAP.items():\n        mask[class_idx] = (seg_data == (class_idx + 1))  # Labels often start from 1\n\n    np.save(os.path.join(out_root, f\"{series_uid}.npy\"), mask)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T16:58:07.12285Z","iopub.execute_input":"2025-08-05T16:58:07.123431Z","iopub.status.idle":"2025-08-05T16:58:10.372369Z","shell.execute_reply.started":"2025-08-05T16:58:07.123403Z","shell.execute_reply":"2025-08-05T16:58:10.371686Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport torch\nimport numpy as np\nimport pydicom\nimport torch.nn.functional as F\nfrom tqdm import tqdm\n\ndef preprocess_and_save_all(raw_dicom_root, output_dir, scale_factor=0.25):\n    os.makedirs(output_dir, exist_ok=True)\n    uids = os.listdir(raw_dicom_root)\n\n    for uid in tqdm(uids, desc=\"Preprocessing DICOM volumes\"):\n        try:\n            series_path = os.path.join(raw_dicom_root, uid)\n            dicom_files = sorted([os.path.join(series_path, f) for f in os.listdir(series_path)])\n\n            # Read and stack slices\n            slices = np.stack([pydicom.dcmread(p).pixel_array for p in dicom_files])\n            volume = (slices.astype(np.float32) - np.mean(slices)) / np.std(slices)  # Normalize\n\n            volume_tensor = torch.tensor(volume).unsqueeze(0).unsqueeze(0)  # [1, 1, D, H, W]\n            volume_downsampled = F.interpolate(volume_tensor, scale_factor=scale_factor, mode='trilinear', align_corners=False)\n            volume_downsampled = volume_downsampled.squeeze(0).squeeze(0)  # [D, H, W]\n\n            # Save\n            torch.save(volume_downsampled, os.path.join(output_dir, f'{uid}.pt'))\n\n        except Exception as e:\n            print(f\"Skipping {uid}: {e}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T16:58:10.3741Z","iopub.execute_input":"2025-08-05T16:58:10.37461Z","iopub.status.idle":"2025-08-05T16:58:10.380782Z","shell.execute_reply.started":"2025-08-05T16:58:10.374588Z","shell.execute_reply":"2025-08-05T16:58:10.380266Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nclass PatchedDicomDataset(Dataset):\n    def __init__(self, df_meta, dicom_root, preprocessed_root, transforms=None, crop_size=None):\n        self.df = df_meta\n        self.root = dicom_root\n        self.preprocessed_root = preprocessed_root\n        self.transforms = transforms\n        self.crop_size = crop_size  # (depth, height, width), e.g. (32, 64, 64)\n\n    def __len__(self):\n        return len(self.df)\n\n    def random_crop_3d(self, volume, crop_size):\n        C, D, H, W = volume.shape\n        cd, ch, cw = crop_size\n        if D < cd or H < ch or W < cw:\n            # Instead of error, return original volume or center crop smaller crop\n            print(f\"Warning: crop size {crop_size} bigger than volume {volume.shape}, skipping crop.\")\n            return volume\n        d1 = random.randint(0, D - cd)\n        h1 = random.randint(0, H - ch)\n        w1 = random.randint(0, W - cw)\n    \n        return volume[:, d1:d1+cd, h1:h1+ch, w1:w1+cw]\n\n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        uid = row['SeriesInstanceUID']\n    \n        volume_path = os.path.join(self.preprocessed_root, f\"{uid}.pt\")\n        if not os.path.exists(volume_path):\n            raise FileNotFoundError(f\"Preprocessed volume not found: {volume_path}\")\n    \n        volume_tensor = torch.load(volume_path).float()  # e.g. [1, 1, 26, 560, 560]\n        \n        # Remove extra dims if any (expect [C, D, H, W])\n        while volume_tensor.dim() > 4:\n            volume_tensor = volume_tensor.squeeze(0)\n        if volume_tensor.dim() != 4:\n            raise ValueError(f\"Volume tensor shape invalid after squeezing: {volume_tensor.shape}\")\n    \n        if self.crop_size is not None:\n            volume_tensor = self.random_crop_3d(volume_tensor, self.crop_size)\n    \n            LABEL_COLS = [\n                'Left Infraclinoid Internal Carotid Artery',\n                'Right Infraclinoid Internal Carotid Artery',\n                'Left Supraclinoid Internal Carotid Artery',\n                'Right Supraclinoid Internal Carotid Artery',\n                'Left Middle Cerebral Artery',\n                'Right Middle Cerebral Artery',\n                'Anterior Communicating Artery',\n                'Left Anterior Cerebral Artery',\n                'Right Anterior Cerebral Artery',\n                'Left Posterior Communicating Artery',\n                'Right Posterior Communicating Artery',\n                'Basilar Tip',\n                'Other Posterior Circulation',\n                'Aneurysm Present',\n            ]\n            \n            label = torch.tensor(row[LABEL_COLS].astype(float).values, dtype=torch.float32)\n\n    \n        mask_downsampled = None\n        if 'mask_path' in row and isinstance(row['mask_path'], str) and os.path.exists(row['mask_path']):\n            mask = torch.tensor(np.load(row['mask_path'])).unsqueeze(0).long()\n            if self.crop_size is not None:\n                mask_downsampled = self.random_crop_3d(mask, self.crop_size)\n            else:\n                mask_downsampled = mask\n    \n        return volume_tensor, label, mask_downsampled","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T16:58:10.381544Z","iopub.execute_input":"2025-08-05T16:58:10.381753Z","iopub.status.idle":"2025-08-05T16:58:10.398445Z","shell.execute_reply.started":"2025-08-05T16:58:10.381729Z","shell.execute_reply":"2025-08-05T16:58:10.397689Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class FilteredDataset(Dataset):\n    def __init__(self, base_dataset, min_depth=2):\n        self.valid_indices = []\n        print(\"Filtering dataset for minimum depth:\", min_depth)\n        for idx in range(len(base_dataset)):\n            try:\n                volume, _, _ = base_dataset[idx]\n                depth = volume.shape[1]  # shape is [1, D, H, W]\n                if depth >= min_depth:\n                    self.valid_indices.append(idx)\n            except (IndexError, ValueError, AssertionError) as e:\n                print(f\"Skipping index {idx} due to data error: {e}\")\n        self.base = base_dataset\n\n    def __len__(self):\n        return len(self.valid_indices)\n\n    def __getitem__(self, idx):\n        return self.base[self.valid_indices[idx]]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T16:58:10.399295Z","iopub.execute_input":"2025-08-05T16:58:10.399506Z","iopub.status.idle":"2025-08-05T16:58:10.413656Z","shell.execute_reply.started":"2025-08-05T16:58:10.399484Z","shell.execute_reply":"2025-08-05T16:58:10.412921Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pydicom\nfrom tqdm import tqdm\nimport pickle\n\nCACHE_PATH = \"/kaggle/working/filtered_series_uids.pkl\"\nDICOM_ROOT = \"/kaggle/input/rsna-intracranial-aneurysm-detection/series\"\nMIN_DEPTH = 2\n\nif os.path.exists(CACHE_PATH):\n    print(\"🔄 Loading cached filtered SeriesInstanceUIDs...\")\n    with open(CACHE_PATH, \"rb\") as f:\n        filtered_uids = pickle.load(f)\nelse:\n    print(\"🔍 Fast filtering using DICOM metadata only...\")\n\n    filtered_uids = []\n    for uid in tqdm(os.listdir(DICOM_ROOT)):\n        series_dir = os.path.join(DICOM_ROOT, uid)\n        if not os.path.isdir(series_dir):\n            continue\n        files = sorted(os.listdir(series_dir))\n        if len(files) < MIN_DEPTH:\n            continue\n        first_file = os.path.join(series_dir, files[0])\n        try:\n            dcm = pydicom.dcmread(first_file, stop_before_pixels=True)\n            if hasattr(dcm, 'NumberOfFrames') and dcm.NumberOfFrames < MIN_DEPTH:\n                continue\n            filtered_uids.append(uid)\n        except Exception as e:\n            continue\n\n    print(f\"✅ Filtered {len(filtered_uids)} valid SeriesInstanceUIDs.\")\n    with open(CACHE_PATH, \"wb\") as f:\n        pickle.dump(filtered_uids, f)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T16:58:10.414503Z","iopub.execute_input":"2025-08-05T16:58:10.415011Z","iopub.status.idle":"2025-08-05T17:02:37.020322Z","shell.execute_reply.started":"2025-08-05T16:58:10.414989Z","shell.execute_reply":"2025-08-05T17:02:37.019507Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pd.Series(filtered_uids, name=\"SeriesInstanceUID\").to_csv(\"/kaggle/working/filtered_uids.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T17:02:37.021186Z","iopub.execute_input":"2025-08-05T17:02:37.021822Z","iopub.status.idle":"2025-08-05T17:02:37.046143Z","shell.execute_reply.started":"2025-08-05T17:02:37.021794Z","shell.execute_reply":"2025-08-05T17:02:37.04547Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/rsna-intracranial-aneurysm-detection/train.csv\")\nfiltered_uids = pd.read_csv(\"/kaggle/working/filtered_uids.csv\")[\"SeriesInstanceUID\"].tolist()\n\ndf_filtered = df[df[\"SeriesInstanceUID\"].isin(filtered_uids)].reset_index(drop=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T17:02:37.046839Z","iopub.execute_input":"2025-08-05T17:02:37.047056Z","iopub.status.idle":"2025-08-05T17:02:37.096098Z","shell.execute_reply.started":"2025-08-05T17:02:37.047041Z","shell.execute_reply":"2025-08-05T17:02:37.095422Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"subset_df = df.sample(n=10, random_state=42)\ndicom_root = \"/kaggle/input/rsna-intracranial-aneurysm-detection/series\"\n\n\npreprocessed_root = \"/kaggle/working/preprocessed\"\nos.makedirs(preprocessed_root, exist_ok=True)\n\nfor idx, row in tqdm(subset_df.iterrows(), total=len(subset_df)):\n    uid = row['SeriesInstanceUID']\n    paths = sorted([os.path.join(dicom_root, uid, f) for f in os.listdir(os.path.join(dicom_root, uid))])\n    \n    slices = np.stack([pydicom.dcmread(p).pixel_array for p in paths], axis=0)\n    volume = (slices.astype(np.float32) - np.mean(slices)) / np.std(slices)\n    \n    volume_tensor = torch.tensor(volume).float().unsqueeze(0)  # shape [1, D, H, W]\n    \n    # Optional: downsample volume here if you want to save memory later\n    \n    save_path = os.path.join(preprocessed_root, f\"{uid}.pt\")\n    torch.save(volume_tensor, save_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T17:02:37.098896Z","iopub.execute_input":"2025-08-05T17:02:37.099096Z","iopub.status.idle":"2025-08-05T17:04:01.068934Z","shell.execute_reply.started":"2025-08-05T17:02:37.09908Z","shell.execute_reply":"2025-08-05T17:04:01.068183Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class ClassifierNet(nn.Module):\n    def __init__(self, input_shape=(1, 32, 64, 64), num_classes=14):\n        super().__init__()\n        self.conv1 = nn.Conv3d(1, 16, kernel_size=3, padding=1)\n        self.pool = nn.MaxPool3d(2)\n        self.adaptive_pool = nn.AdaptiveAvgPool3d((1, 1, 1))\n        self.fc = nn.Linear(16, 14)\n\n    def forward(self, x):\n        x = self.pool(F.relu(self.conv1(x)))\n        x = self.adaptive_pool(x)\n        x = x.view(x.size(0), -1)\n        x = self.fc(x)\n        return x  # [B, 14]\n\n# Example usage:\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nclf = ClassifierNet(input_shape=(1, 32, 64, 64)).to(device)\n\n# Test with dummy input\ndummy_input = torch.randn(1, 1, 32, 64, 64).to(device)\noutput = clf(dummy_input)\nprint(output.shape)  # Should print: torch.Size([1, 1])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T17:04:01.069792Z","iopub.execute_input":"2025-08-05T17:04:01.070066Z","iopub.status.idle":"2025-08-05T17:04:01.928433Z","shell.execute_reply.started":"2025-08-05T17:04:01.070043Z","shell.execute_reply":"2025-08-05T17:04:01.927482Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class Simple3DSegmentationNet(nn.Module):\n    def __init__(self, in_channels=1, out_channels=2):\n        super().__init__()\n        self.enc1 = nn.Sequential(\n            nn.Conv3d(in_channels, 16, kernel_size=3, padding=1),\n            nn.BatchNorm3d(16),\n            nn.ReLU(inplace=True),\n            nn.Conv3d(16, 16, kernel_size=3, padding=1),\n            nn.BatchNorm3d(16),\n            nn.ReLU(inplace=True),\n        )\n        self.pool1 = nn.MaxPool3d(2)\n\n        self.enc2 = nn.Sequential(\n            nn.Conv3d(16, 32, kernel_size=3, padding=1),\n            nn.BatchNorm3d(32),\n            nn.ReLU(inplace=True),\n            nn.Conv3d(32, 32, kernel_size=3, padding=1),\n            nn.BatchNorm3d(32),\n            nn.ReLU(inplace=True),\n        )\n        self.pool2 = nn.MaxPool3d(2)\n\n        self.bottleneck = nn.Sequential(\n            nn.Conv3d(32, 64, kernel_size=3, padding=1),\n            nn.BatchNorm3d(64),\n            nn.ReLU(inplace=True),\n            nn.Conv3d(64, 64, kernel_size=3, padding=1),\n            nn.BatchNorm3d(64),\n            nn.ReLU(inplace=True),\n        )\n\n        self.up2 = nn.ConvTranspose3d(64, 32, kernel_size=2, stride=2)\n        self.dec2 = nn.Sequential(\n            nn.Conv3d(64, 32, kernel_size=3, padding=1),\n            nn.BatchNorm3d(32),\n            nn.ReLU(inplace=True),\n            nn.Conv3d(32, 32, kernel_size=3, padding=1),\n            nn.BatchNorm3d(32),\n            nn.ReLU(inplace=True),\n        )\n\n        self.up1 = nn.ConvTranspose3d(32, 16, kernel_size=2, stride=2)\n        self.dec1 = nn.Sequential(\n            nn.Conv3d(32, 16, kernel_size=3, padding=1),\n            nn.BatchNorm3d(16),\n            nn.ReLU(inplace=True),\n            nn.Conv3d(16, 16, kernel_size=3, padding=1),\n            nn.BatchNorm3d(16),\n            nn.ReLU(inplace=True),\n        )\n        \n        self.out_conv = nn.Conv3d(16, out_channels, kernel_size=1)\n\n    def forward(self, x):\n        enc1 = self.enc1(x)          # [B,16,D,H,W]\n        p1 = self.pool1(enc1)        # [B,16,D/2,H/2,W/2]\n    \n        enc2 = self.enc2(p1)         # [B,32,D/2,H/2,W/2]\n        p2 = self.pool2(enc2)        # [B,32,D/4,H/4,W/4]\n    \n        bottleneck = self.bottleneck(p2)  # [B,64,D/4,H/4,W/4]\n    \n        up2 = self.up2(bottleneck)          # [B,32,D/2,H/2,W/2]\n    \n        # Resize enc2 to match up2 spatial dims before concatenation\n        if up2.shape[2:] != enc2.shape[2:]:\n            enc2 = F.interpolate(enc2, size=up2.shape[2:], mode='trilinear', align_corners=False)\n    \n        cat2 = torch.cat([up2, enc2], dim=1)  # skip connection\n    \n        dec2 = self.dec2(cat2)               # [B,32,D/2,H/2,W/2]\n    \n        up1 = self.up1(dec2)                 # [B,16,D,H,W]\n    \n        # Resize enc1 if needed before concat\n        if up1.shape[2:] != enc1.shape[2:]:\n            enc1 = F.interpolate(enc1, size=up1.shape[2:], mode='trilinear', align_corners=False)\n    \n        cat1 = torch.cat([up1, enc1], dim=1)\n    \n        dec1 = self.dec1(cat1)               # [B,16,D,H,W]\n    \n        out = self.out_conv(dec1)            # [B,out_channels,D,H,W]\n    \n        return out","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T17:04:01.929469Z","iopub.execute_input":"2025-08-05T17:04:01.929766Z","iopub.status.idle":"2025-08-05T17:04:02.224995Z","shell.execute_reply.started":"2025-08-05T17:04:01.929742Z","shell.execute_reply":"2025-08-05T17:04:02.224256Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from torch.utils.data import DataLoader\nimport torch.optim as optim\nfrom torch.cuda.amp import GradScaler\n\ndicom_root = \"/kaggle/input/rsna-intracranial-aneurysm-detection/series\"\npreprocessed_root = '/kaggle/working/preprocessed'\n\n# Load metadata CSV\ndf = pd.read_csv(\"/kaggle/input/rsna-intracranial-aneurysm-detection/train.csv\")\n\n# Filter dataframe to keep only UIDs that have a .pt preprocessed file\npreprocessed_files = set(f[:-3] for f in os.listdir(preprocessed_root) if f.endswith('.pt'))\ndf_filtered = df[df['SeriesInstanceUID'].isin(preprocessed_files)].reset_index(drop=True)\nprint(f\"Filtered dataframe size: {len(df_filtered)}\")\n\n# Initialize dataset with filtered dataframe\ndataset = PatchedDicomDataset(df_meta=df_filtered, dicom_root=dicom_root, crop_size=(32, 64, 64), preprocessed_root=preprocessed_root)\n\n# DataLoader\ndef custom_collate(batch):\n    # Filter None samples (whole sample)\n    batch = [b for b in batch if b is not None]\n    vols = torch.utils.data.dataloader.default_collate([b[0] for b in batch])\n    labels = torch.utils.data.dataloader.default_collate([b[1] for b in batch])\n    masks = []\n    for b in batch:\n        if b[2] is None:\n            masks.append(torch.zeros_like(b[0], dtype=torch.long))  # or whatever shape you want\n        else:\n            masks.append(b[2])\n    masks = torch.utils.data.dataloader.default_collate(masks)\n    return vols, labels, masks\n\nloader = DataLoader(dataset, batch_size=1, shuffle=True, num_workers=2, collate_fn=custom_collate)\n\nclf = ClassifierNet(input_shape=(1, 32, 64, 64)).to(device)\nseg = Simple3DSegmentationNet(in_channels=1, out_channels=2).to(device)\n\n# Losses\nclassification_loss = nn.BCEWithLogitsLoss()\nsegmentation_loss = nn.CrossEntropyLoss()\n\n# Models, loss, optimizer, scaler (your existing definitions)\noptimizer = optim.Adam(list(clf.parameters()) + list(seg.parameters()), lr=1e-4)\nscaler = GradScaler()\n\n# Training loop\nn_epochs = 5\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\nfor epoch in range(n_epochs):\n    running_loss = 0.0\n    for vol, label, mask in tqdm(loader, desc=f\"Training Epoch {epoch+1}\", leave=True):\n        vol = vol.to(device).float()\n        label = label.to(device).float()\n        if mask is not None:\n            mask = mask.to(device).long()\n\n        optimizer.zero_grad()\n\n        with torch.amp.autocast(device_type='cuda'):\n            pred = clf(vol)\n            loss_c = classification_loss(pred, label)\n\n\n            seg_pred = seg(vol)\n            if mask.dim() == 4:\n                mask = mask.unsqueeze(1)  # [B, 1, D, H, W]\n            \n            # Resize and compute loss\n            mask_resized = F.interpolate(mask.float(), size=seg_pred.shape[2:], mode='trilinear', align_corners=False)\n            loss_s = segmentation_loss(seg_pred, mask_resized.squeeze(1).long())\n\n            loss = loss_c + loss_s\n\n        scaler.scale(loss).backward()\n        scaler.step(optimizer)\n        scaler.update()\n        torch.cuda.empty_cache()\n        gc.collect()\n\n        running_loss += loss.item()\n\n    print(f\"Epoch {epoch+1} Loss: {running_loss:.4f}\")\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T17:04:02.225753Z","iopub.execute_input":"2025-08-05T17:04:02.225932Z","iopub.status.idle":"2025-08-05T17:04:20.244234Z","shell.execute_reply.started":"2025-08-05T17:04:02.225918Z","shell.execute_reply":"2025-08-05T17:04:20.243281Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/rsna-intracranial-aneurysm-detection/train.csv\")\ndf_test = pd.read_csv(\"/kaggle/input/rsna-intracranial-aneurysm-detection/kaggle_evaluation/test.csv\")\nprint(\"Total SeriesInstanceUIDs:\", df[\"SeriesInstanceUID\"].nunique())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T17:04:20.245378Z","iopub.execute_input":"2025-08-05T17:04:20.24571Z","iopub.status.idle":"2025-08-05T17:04:20.270307Z","shell.execute_reply.started":"2025-08-05T17:04:20.245668Z","shell.execute_reply":"2025-08-05T17:04:20.269729Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#save weights\n\ntorch.save(clf.state_dict(), 'classification_model.pth')\ntorch.save(seg.state_dict(), 'segmentation_model.pth')\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T17:04:20.271055Z","iopub.execute_input":"2025-08-05T17:04:20.271332Z","iopub.status.idle":"2025-08-05T17:04:20.287297Z","shell.execute_reply.started":"2025-08-05T17:04:20.271306Z","shell.execute_reply":"2025-08-05T17:04:20.286566Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#load weights\nclf.load_state_dict(torch.load('classification_model.pth'))\nseg.load_state_dict(torch.load('segmentation_model.pth'))\n#clf.eval()\n#seg.eval()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T17:04:20.288169Z","iopub.execute_input":"2025-08-05T17:04:20.288945Z","iopub.status.idle":"2025-08-05T17:04:20.313376Z","shell.execute_reply.started":"2025-08-05T17:04:20.288922Z","shell.execute_reply":"2025-08-05T17:04:20.312788Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def preprocess_and_save_test_series(uid, dicom_root, save_root):\n\n    \n    series_path = os.path.join(dicom_root, uid)\n    dicom_files = sorted(\n        [os.path.join(series_path, f) for f in os.listdir(series_path) if f.endswith('.dcm')],\n        key=lambda x: int(pydicom.dcmread(x).InstanceNumber)\n    )\n    slices = [pydicom.dcmread(f).pixel_array for f in dicom_files]\n    if len(slices) == 0:\n        print(f\"No dicom slices found for {uid}\")\n        return False\n    \n    volume = np.stack(slices).astype(np.float32)\n    volume = (volume - volume.min()) / (volume.max() - volume.min() + 1e-5)\n    volume_tensor = torch.tensor(volume).unsqueeze(0)  # Shape: [1, D, H, W]\n    \n    # Save tensor\n    os.makedirs(save_root, exist_ok=True)\n    torch.save(volume_tensor, os.path.join(save_root, f\"{uid}.pt\"))\n    print(f\"Saved preprocessed tensor for {uid}\")\n    return True\n\n\n# Preprocess all test UIDs\nfor uid in df_test[\"SeriesInstanceUID\"]:\n    preprocess_and_save_test_series(uid, dicom_root=\"/kaggle/input/rsna-intracranial-aneurysm-detection/kaggle_evaluation/series\", save_root=preprocessed_root)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T17:04:20.314038Z","iopub.execute_input":"2025-08-05T17:04:20.314348Z","iopub.status.idle":"2025-08-05T17:04:32.939824Z","shell.execute_reply.started":"2025-08-05T17:04:20.314321Z","shell.execute_reply":"2025-08-05T17:04:32.93917Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#def run_model_on_uid_debug(uid, model, device, preprocessed_root, crop_size=(32, 64, 64)):\n#    import torch.nn.functional as F\n#    import os\n#\n #   pt_path = os.path.join(preprocessed_root, f\"{uid}.pt\")\n#\n #   if not os.path.isfile(pt_path):\n  #      print(f\"Warning: Preprocessed file not found for UID {uid}, returning 0.5\")\n   #     return 0.5\n#\n#    volume = torch.load(pt_path)\n\n#    if len(volume.shape) == 4:\n #       volume = volume.unsqueeze(0)\n#    elif len(volume.shape) == 3:\n#        volume = volume.unsqueeze(0).unsqueeze(0)\n#    else:\n      #  raise ValueError(f\"Unexpected volume shape: {volume.shape}\")\n\n#    _, C, D, H, W = volume.shape\n#    td, th, tw = crop_size\n\n #   pad_d = max(td - D, 0)\n#    pad_h = max(th - H, 0)\n#    pad_w = max(tw - W, 0)\n#    volume = F.pad(volume, \n#                   [pad_w // 2, pad_w - pad_w // 2,\n#                    pad_h // 2, pad_h - pad_h // 2,\n#                    pad_d // 2, pad_d - pad_d // 2])\n\n#    _, _, D, H, W = volume.shape\n#    start_d = (D - td) // 2\n#    start_h = (H - th) // 2\n#    start_w = (W - tw) // 2\n#    volume = volume[:, :, start_d:start_d+td, start_h:start_h+th, start_w:start_w+tw]\n\n#    volume = volume.to(device).float()\n\n#    model.eval()\n#    with torch.no_grad():\n #       logits = model(volume)\n#        prob = torch.sigmoid(logits).item()\n\n#    print(f\"UID: {uid}, prediction: {prob:.4f}\")\n\n#    return prob\n\n#df_test[\"label\"] = df_test[\"SeriesInstanceUID\"].apply(\n#    lambda uid: run_model_on_uid_debug(uid, clf, device, preprocessed_root)\n#)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T17:04:32.940533Z","iopub.execute_input":"2025-08-05T17:04:32.940744Z","iopub.status.idle":"2025-08-05T17:04:32.944593Z","shell.execute_reply.started":"2025-08-05T17:04:32.940727Z","shell.execute_reply":"2025-08-05T17:04:32.944046Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#def run_model_on_uid(uid, model, device, preprocessed_root, crop_size=(32, 64, 64)):\n#    pt_path = os.path.join(preprocessed_root, f\"{uid}.pt\")\n#    if not os.path.isfile(pt_path):\n#        print(f\"Warning: Preprocessed file not found for UID {uid}, returning 0.5\")\n#        return 0.5\n\n#    volume = torch.load(pt_path)\n\n#    if len(volume.shape) == 4:\n#        volume = volume.unsqueeze(0)\n#    elif len(volume.shape) == 3:\n #       volume = volume.unsqueeze(0).unsqueeze(0)\n#    else:\n #       raise ValueError(f\"Unexpected volume shape: {volume.shape}\")\n\n#    _, C, D, H, W = volume.shape\n#    td, th, tw = crop_size\n\n#    pad_d = max(td - D, 0)\n#    pad_h = max(th - H, 0)\n#    pad_w = max(tw - W, 0)\n#    volume = F.pad(volume,\n#                   [pad_w // 2, pad_w - pad_w // 2,\n#                    pad_h // 2, pad_h - pad_h // 2,\n#                    pad_d // 2, pad_d - pad_d // 2])\n\n#    _, _, D, H, W = volume.shape\n#    start_d = (D - td) // 2\n#    start_h = (H - th) // 2\n#    start_w = (W - tw) // 2\n#    volume = volume[:, :, start_d:start_d + td, start_h:start_h + th, start_w:start_w + tw]\n\n#    volume = volume.to(device).float()\n\n#    model.eval()\n#    with torch.no_grad():\n#        logits = model(volume)\n#        prob = torch.sigmoid(logits).item()\n\n#    print(f\"UID: {uid}, prediction: {prob:.4f}\")\n #   return prob\n\n\n# Load test CSV with SeriesInstanceUID\n#df_test = pd.read_csv(\"/kaggle/input/rsna-intracranial-aneurysm-detection/kaggle_evaluation/test.csv\")\n\n# Generate predictions for all UIDs in test set\n#df_test[\"label\"] = df_test[\"SeriesInstanceUID\"].apply(\n#    lambda uid: run_model_on_uid(uid, clf, device, preprocessed_root)\n#)\n\n# Save submission file as submission.parquet (competition required format)\n#df_test[[\"SeriesInstanceUID\", \"label\"]].to_parquet(\"submission.parquet\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T17:04:32.945327Z","iopub.execute_input":"2025-08-05T17:04:32.94552Z","iopub.status.idle":"2025-08-05T17:04:32.962203Z","shell.execute_reply.started":"2025-08-05T17:04:32.945505Z","shell.execute_reply":"2025-08-05T17:04:32.961594Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport torch\nimport polars as pl\nimport numpy as np\nimport pydicom\nimport shutil\nfrom torchvision import transforms\n\n# Your label columns (exactly as required)\nLABEL_COLS = [\n    'Left Infraclinoid Internal Carotid Artery',\n    'Right Infraclinoid Internal Carotid Artery',\n    'Left Supraclinoid Internal Carotid Artery',\n    'Right Supraclinoid Internal Carotid Artery',\n    'Left Middle Cerebral Artery',\n    'Right Middle Cerebral Artery',\n    'Anterior Communicating Artery',\n    'Left Anterior Cerebral Artery',\n    'Right Anterior Cerebral Artery',\n    'Left Posterior Communicating Artery',\n    'Right Posterior Communicating Artery',\n    'Basilar Tip',\n    'Other Posterior Circulation',\n    'Aneurysm Present',\n]\n\nID_COL = \"SeriesInstanceUID\"\n\n# Load model weights if saved\n# clf.load_state_dict(torch.load(\"path/to/classifier_weights.pth\"))  # <-- OPTIONAL\n\ndef preprocess_volume(series_path):\n    \"\"\"Loads DICOMs and returns a preprocessed volume tensor.\"\"\"\n    dcm_files = sorted([os.path.join(series_path, f) for f in os.listdir(series_path) if f.endswith(\".dcm\")])\n    slices = [pydicom.dcmread(f).pixel_array.astype(np.float32) for f in dcm_files]\n    volume = np.stack(slices, axis=0)  # Shape: [D, H, W]\n\n    # Normalize\n    volume = (volume - np.mean(volume)) / (np.std(volume) + 1e-5)\n\n    # Resize or pad to expected shape (32, 64, 64)\n    volume = resize_or_pad(volume, (32, 64, 64))\n\n    # Add channel dimension\n    tensor = torch.tensor(volume).unsqueeze(0).unsqueeze(0)  # [1, 1, D, H, W]\n    return tensor\n\ndef resize_or_pad(vol, target_shape):\n    \"\"\"Pads or center-crops to match target_shape=(D,H,W).\"\"\"\n    d, h, w = vol.shape\n    td, th, tw = target_shape\n\n    # Pad or crop depth\n    if d < td:\n        pad_d = (td - d) // 2\n        vol = np.pad(vol, ((pad_d, td - d - pad_d), (0,0), (0,0)), mode='constant')\n    elif d > td:\n        start_d = (d - td) // 2\n        vol = vol[start_d:start_d+td]\n\n    # Pad or crop height\n    if h < th:\n        pad_h = (th - h) // 2\n        vol = np.pad(vol, ((0,0), (pad_h, th - h - pad_h), (0,0)), mode='constant')\n    elif h > th:\n        start_h = (h - th) // 2\n        vol = vol[:, start_h:start_h+th]\n\n    # Pad or crop width\n    if w < tw:\n        pad_w = (tw - w) // 2\n        vol = np.pad(vol, ((0,0), (0,0), (pad_w, tw - w - pad_w)), mode='constant')\n    elif w > tw:\n        start_w = (w - tw) // 2\n        vol = vol[:, :, start_w:start_w+tw]\n\n    return vol\n\n# Your trained classifier\nclf.eval()\n\n@torch.no_grad()\ndef predict(series_path: str) -> pl.DataFrame:\n    try:\n        series_id = os.path.basename(series_path)\n        volume_tensor = preprocess_volume(series_path).to(device)\n    \n        output = clf(volume_tensor)\n        probs = torch.sigmoid(output).cpu().numpy().flatten().tolist()\n    \n        if probs is None or len(probs) != 14 or any(np.isnan(probs)):\n            print(f\"Invalid prediction for {series_id}\")\n            probs = [0.5] * 14\n    \n        # Here we output the same value for all 14 columns (for demo purposes)\n        # You can extend your model to predict all LABEL_COLS if you train for them\n        pred_row = [series_id] + probs\n        result = pl.DataFrame([pred_row], schema=[ID_COL] + LABEL_COLS)\n    \n        #shutil.rmtree('/kaggle/shared', ignore_errors=True)\n        return result\n    except Exception as e:\n        print(f\"Error processing {series_path}: {e}\")\n        raise e    \n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T17:04:32.96294Z","iopub.execute_input":"2025-08-05T17:04:32.963338Z","iopub.status.idle":"2025-08-05T17:04:33.061815Z","shell.execute_reply.started":"2025-08-05T17:04:32.963322Z","shell.execute_reply":"2025-08-05T17:04:33.061185Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for fname in os.listdir(\"/kaggle/working\"):\n    if fname != \"submission.parquet\":\n        path = os.path.join(\"/kaggle/working\", fname)\n        if os.path.isfile(path):\n            os.remove(path)\n        elif os.path.isdir(path):\n            shutil.rmtree(path)\n\nshutil.rmtree('/kaggle/shared', ignore_errors=True)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T17:04:33.062705Z","iopub.execute_input":"2025-08-05T17:04:33.063006Z","iopub.status.idle":"2025-08-05T17:04:34.08971Z","shell.execute_reply.started":"2025-08-05T17:04:33.062933Z","shell.execute_reply":"2025-08-05T17:04:34.089084Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"inference_server = kaggle_evaluation.rsna_inference_server.RSNAInferenceServer(predict)\n\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway()\n    display(pl.read_parquet('/kaggle/working/submission.parquet'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T17:04:34.090441Z","iopub.execute_input":"2025-08-05T17:04:34.090648Z","iopub.status.idle":"2025-08-05T17:04:41.162533Z","shell.execute_reply.started":"2025-08-05T17:04:34.090632Z","shell.execute_reply":"2025-08-05T17:04:41.161764Z"}},"outputs":[],"execution_count":null}]}