{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":99552,"databundleVersionId":13441085,"sourceType":"competition"},{"sourceId":12945953,"sourceType":"datasetVersion","datasetId":8192522}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader, Subset\nfrom sklearn.model_selection import StratifiedKFold\nfrom tqdm import tqdm\nimport timm\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-09-03T03:36:31.062361Z","iopub.execute_input":"2025-09-03T03:36:31.062732Z","iopub.status.idle":"2025-09-03T03:36:31.800869Z","shell.execute_reply.started":"2025-09-03T03:36:31.062704Z","shell.execute_reply":"2025-09-03T03:36:31.800223Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch.nn as nn\nimport torch.nn.functional as F\n\nclass FocalLoss(nn.Module):\n    def __init__(self, alpha=0.25, gamma=2.0, reduction=\"mean\"):\n        super(FocalLoss, self).__init__()\n        self.alpha = alpha\n        self.gamma = gamma\n        self.reduction = reduction\n\n    def forward(self, inputs, targets):\n        bce_loss = F.binary_cross_entropy_with_logits(inputs, targets, reduction=\"none\")\n        probas = torch.sigmoid(inputs)\n        pt = probas * targets + (1 - probas) * (1 - targets)\n        focal_loss = self.alpha * (1 - pt) ** self.gamma * bce_loss\n\n        if self.reduction == \"mean\":\n            return focal_loss.mean()\n        elif self.reduction == \"sum\":\n            return focal_loss.sum()\n        else:\n            return focal_loss","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class AneurysmDataset(Dataset):\n    def __init__(self, npz_dir):\n        self.npz_files = [os.path.join(npz_dir, f) for f in os.listdir(npz_dir) if f.endswith(\".npz\")]\n        self.labels = []\n        for f in self.npz_files:\n            data = np.load(f, allow_pickle=True)\n            labels_dict = data[\"labels\"].item()\n            # stratify using the main target \"Aneurysm Present\"\n            self.labels.append(labels_dict[\"Aneurysm Present\"])\n\n    def __len__(self):\n        return len(self.npz_files)\n\n    def __getitem__(self, idx):\n        data = np.load(self.npz_files[idx], allow_pickle=True)\n        volume = data[\"volume\"].astype(np.float32) / 255.0  # normalize\n        label = torch.tensor(list(data[\"labels\"].item().values()), dtype=torch.float32)\n        volume = torch.tensor(volume)\n        return volume, label\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-03T03:12:13.195651Z","iopub.execute_input":"2025-09-03T03:12:13.196116Z","iopub.status.idle":"2025-09-03T03:12:13.202249Z","shell.execute_reply.started":"2025-09-03T03:12:13.196092Z","shell.execute_reply":"2025-09-03T03:12:13.20145Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class EfficientNetV2MultiLabel(nn.Module):\n    def __init__(self, num_classes=14):\n        super().__init__()\n        self.backbone = timm.create_model(\"tf_efficientnetv2_s_in21k\", pretrained=True, in_chans=32)\n        in_features = self.backbone.classifier.in_features\n        self.backbone.classifier = nn.Linear(in_features, num_classes)\n\n    def forward(self, x):\n        return self.backbone(x)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-03T03:12:15.448984Z","iopub.execute_input":"2025-09-03T03:12:15.44972Z","iopub.status.idle":"2025-09-03T03:12:15.454042Z","shell.execute_reply.started":"2025-09-03T03:12:15.449691Z","shell.execute_reply":"2025-09-03T03:12:15.45335Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport gc\nfrom pathlib import Path\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import roc_auc_score\nfrom tqdm import tqdm\nimport torch\nimport torch.nn as nn\nfrom torch.utils.data import Dataset, DataLoader\nimport timm  # EfficientNetV2\n\n# ------------------------------\n# Weighted AUC (competition metric)\n# ------------------------------\nCLASS_WEIGHTS = np.array([1]*13 + [13], dtype=np.float32)  # last label is 'Aneurysm Present'\n\ndef compute_weighted_auc(y_true, y_pred, class_weights=CLASS_WEIGHTS):\n    y_true = y_true.cpu().numpy()\n    y_pred = y_pred.detach().cpu().numpy()  # detach first\n    try:\n        individual_aucs = roc_auc_score(y_true, y_pred, average=None)\n    except ValueError:\n        individual_aucs = np.zeros(y_true.shape[1], dtype=np.float32)\n    weights = class_weights / class_weights.sum()\n    return float(np.sum(individual_aucs * weights))\n\n# ------------------------------\n# Dataset\n# ------------------------------\nclass AneurysmDataset(Dataset):\n    def __init__(self, npz_dir, transforms=None):\n        self.npz_dir = Path(npz_dir)\n        self.files = list(self.npz_dir.glob(\"*.npz\"))\n        self.transforms = transforms\n        self.labels = [np.load(f, allow_pickle=True)[\"labels\"].item()[\"Aneurysm Present\"] for f in self.files]\n\n    def __len__(self):\n        return len(self.files)\n\n    def __getitem__(self, idx):\n        f = self.files[idx]\n        data = np.load(f, allow_pickle=True)\n        volume = data[\"volume\"].astype(np.float32) / 255.0  # (32, H, W)\n        labels = np.array(list(data[\"labels\"].item().values()), dtype=np.float32)\n    \n        if self.transforms:\n            augmented_slices = []\n            for i in range(volume.shape[0]):\n                img = (volume[i] * 255).astype(np.uint8)\n                augmented = self.transforms(image=img)\n                # Remove channel dim if present\n                img_aug = augmented[\"image\"]\n                if img_aug.ndim == 3 and img_aug.shape[0] == 1:\n                    img_aug = img_aug.squeeze(0)\n                augmented_slices.append(img_aug)\n            volume = np.stack(augmented_slices)  # (32, H, W)\n    \n        x = torch.from_numpy(volume)  # (32, H, W)\n        return x, torch.from_numpy(labels)\n\n\n# ------------------------------\n# Training / Validation\n# ------------------------------\ndef train_epoch(model, loader, criterion, optimizer, device):\n    model.train()\n    running_loss = 0.0\n    all_labels, all_preds = [], []\n\n    for x, y in tqdm(loader, desc=\"Train\", leave=False):\n        x, y = x.to(device), y.to(device)\n        optimizer.zero_grad()\n        outputs = model(x)\n        loss = criterion(outputs, y)\n        loss.backward()\n        optimizer.step()\n        running_loss += loss.item() * x.size(0)\n\n        all_labels.append(y)\n        all_preds.append(torch.sigmoid(outputs))\n\n    all_labels = torch.cat(all_labels, dim=0)\n    all_preds = torch.cat(all_preds, dim=0)\n    auc = compute_weighted_auc(all_labels, all_preds)\n\n    return running_loss / len(loader.dataset), auc\n\ndef validate_epoch(model, loader, criterion, device):\n    model.eval()\n    running_loss = 0.0\n    all_labels, all_preds = [], []\n\n    with torch.no_grad():\n        for x, y in tqdm(loader, desc=\"Valid\", leave=False):\n            x, y = x.to(device), y.to(device)\n            outputs = model(x)\n            loss = criterion(outputs, y)\n            running_loss += loss.item() * x.size(0)\n\n            all_labels.append(y)\n            all_preds.append(torch.sigmoid(outputs))\n\n    all_labels = torch.cat(all_labels, dim=0)\n    all_preds = torch.cat(all_preds, dim=0)\n    auc = compute_weighted_auc(all_labels, all_preds)\n\n    return running_loss / len(loader.dataset), auc","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-03T03:41:42.666199Z","iopub.execute_input":"2025-09-03T03:41:42.666812Z","iopub.status.idle":"2025-09-03T03:41:42.683654Z","shell.execute_reply.started":"2025-09-03T03:41:42.666784Z","shell.execute_reply":"2025-09-03T03:41:42.682665Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_transforms = A.Compose([\n    # A.HorizontalFlip(p=0.5),\n    # A.ShiftScaleRotate(shift_limit=0.05, scale_limit=0.1, rotate_limit=10, p=0.5),\n    A.RandomBrightnessContrast(p=0.3),\n    A.GaussianBlur(p=0.2),\n    A.Normalize(mean=0.5, std=0.5),\n    ToTensorV2()\n])\n\nval_transforms = A.Compose([\n    A.Normalize(mean=0.5, std=0.5),\n    ToTensorV2()\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-03T03:42:34.374899Z","iopub.execute_input":"2025-09-03T03:42:34.375803Z","iopub.status.idle":"2025-09-03T03:42:34.383944Z","shell.execute_reply.started":"2025-09-03T03:42:34.375759Z","shell.execute_reply":"2025-09-03T03:42:34.383065Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def main(npz_dir, num_epochs=10, batch_size=16):\n    device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n    dataset = AneurysmDataset(npz_dir)\n\n    skf = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n\n    for fold, (train_idx, val_idx) in enumerate(skf.split(np.zeros(len(dataset)), dataset.labels)):\n        print(f\"\\n=== Fold {fold+1} ===\")\n        best_auc = 0\n        train_subset = Subset(AneurysmDataset(npz_dir, transforms=train_transforms), train_idx)\n        val_subset = Subset(AneurysmDataset(npz_dir, transforms=val_transforms), val_idx)\n\n        train_loader = DataLoader(train_subset, batch_size=batch_size, shuffle=True, num_workers=4)\n        val_loader = DataLoader(val_subset, batch_size=batch_size, shuffle=False, num_workers=4)\n\n        model = EfficientNetV2MultiLabel(num_classes=14).to(device)\n        # criterion = nn.BCEWithLogitsLoss()\n        criterion = FocalLoss(alpha=0.25, gamma=2.0)\n        optimizer = optim.Adam(model.parameters(), lr=1e-4)\n\n        for epoch in range(num_epochs):\n            \n            train_loss, train_auc = train_epoch(model, train_loader, criterion, optimizer, device)\n            val_loss, val_auc = validate_epoch(model, val_loader, criterion, device)\n\n            print(f\"Epoch {epoch+1}/{num_epochs} | \"\n                  f\"Train Loss: {train_loss:.4f} | Train AUC: {train_auc:.4f} | \"\n                  f\"Val Loss: {val_loss:.4f} | Val AUC: {val_auc:.4f}\")\n\n            if val_auc > best_auc:\n                    best_auc = val_auc\n                    print(f'Saving best model, fold :{fold} , Val AUC : {val_auc}')\n                    torch.save(model.state_dict(), f\"efficientnetv2_fold{fold+1}.pth\")\n        del model, optimizer, train_loader, val_loader\n        torch.cuda.empty_cache()\n\n\nif __name__ == \"__main__\":\n    main(\"/kaggle/input/rsna-iad-3d-volumes-512/processed_train\", num_epochs=5, batch_size=16)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-03T03:51:36.305847Z","iopub.execute_input":"2025-09-03T03:51:36.306216Z","iopub.status.idle":"2025-09-03T03:52:05.935766Z","shell.execute_reply.started":"2025-09-03T03:51:36.306172Z","shell.execute_reply":"2025-09-03T03:52:05.93447Z"}},"outputs":[],"execution_count":null}]}