{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":99552,"databundleVersionId":13851420,"sourceType":"competition"},{"sourceId":13258161,"sourceType":"datasetVersion","datasetId":8401368},{"sourceId":13266765,"sourceType":"datasetVersion","datasetId":8407116},{"sourceId":295264803,"sourceType":"kernelVersion"}],"dockerImageVersionId":31260,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# RSNA Aneurysm Detection - Inference with Ensemble\n\n## Updated to Match New Training Approach\n\nThis notebook:\n1. Loads your trained models (5-fold ensemble)\n2. Can evaluate on **validation set**, **test set**, or **new data**\n3. Handles both positive and negative samples\n4. Uses the `data_split.csv` from training to identify test samples\n5. Generates predictions with ensemble averaging\n\n## Requirements:\n1. Add your trained models as a Kaggle Dataset\n2. Add the `data_split.csv` from training\n3. Add your preprocessed images datasets\n4. Update paths in Configuration cell","metadata":{}},{"cell_type":"markdown","source":"## 1. Install Dependencies","metadata":{}},{"cell_type":"code","source":"%%capture\n!pip install timm albumentations","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-03T09:26:06.623049Z","iopub.execute_input":"2026-02-03T09:26:06.623359Z","iopub.status.idle":"2026-02-03T09:26:11.116318Z","shell.execute_reply.started":"2026-02-03T09:26:06.62333Z","shell.execute_reply":"2026-02-03T09:26:11.115585Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2. Import Libraries","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nfrom torch.cuda.amp import autocast\nimport timm\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\nimport cv2\nfrom pathlib import Path\nfrom tqdm.notebook import tqdm\nimport glob\nimport warnings\nfrom sklearn.metrics import accuracy_score, classification_report, confusion_matrix\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nwarnings.filterwarnings('ignore')\n\nprint(f\"PyTorch version: {torch.__version__}\")\nprint(f\"CUDA available: {torch.cuda.is_available()}\")","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","trusted":true,"execution":{"iopub.status.busy":"2026-02-03T09:26:11.117866Z","iopub.execute_input":"2026-02-03T09:26:11.118116Z","iopub.status.idle":"2026-02-03T09:26:25.564598Z","shell.execute_reply.started":"2026-02-03T09:26:11.118087Z","shell.execute_reply":"2026-02-03T09:26:25.563921Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3. Configuration","metadata":{}},{"cell_type":"code","source":"class InferenceConfig:\n    # Model settings\n    model_name = \"tf_efficientnetv2_s.in21k_ft_in1k\"\n    size = 384\n    in_chans = 32\n    \n    # Paths - UPDATE THESE!\n    model_dir = \"/kaggle/input/kaggle-training-notebook-updated\"  # ⚠️ UPDATE: Your models dataset\n    data_split_csv = \"/kaggle/input/kaggle-training-notebook-updated/data_split.csv\"  # ⚠️ UPDATE: Split from training\n    \n    # Data directories (same as training)\n    data_dirs = [\n        \"/kaggle/input/rsna-preprocessed-images\",      # ⚠️ UPDATE\n        \"/kaggle/input/rsna-processed-images-2\"        # ⚠️ UPDATE\n    ]\n    \n    # CSV with ground truth labels (for evaluation)\n    csv_path = \"/kaggle/input/rsna-intracranial-aneurysm-detection/train_localizers.csv\"\n    \n    # Inference settings\n    eval_on = \"test\"  # Options: \"test\", \"val\", \"train\", \"all\", \"custom\"\n    n_fold = 5\n    trn_fold = [0, 1, 2, 3, 4]  # Which folds to use for ensemble\n    \n    # Target columns\n    target_cols = [\n        'Left Infraclinoid Internal Carotid Artery',\n        'Right Infraclinoid Internal Carotid Artery',\n        'Left Supraclinoid Internal Carotid Artery',\n        'Right Supraclinoid Internal Carotid Artery',\n        'Left Middle Cerebral Artery',\n        'Right Middle Cerebral Artery',\n        'Anterior Communicating Artery',\n        'Left Anterior Cerebral Artery',\n        'Right Anterior Cerebral Artery',\n        'Left Posterior Communicating Artery',\n        'Right Posterior Communicating Artery',\n        'Basilar Tip',\n        'Other Posterior Circulation',\n        'Aneurysm Present',\n    ]\n    num_classes = len(target_cols)\n    \n    # Inference settings\n    use_amp = True\n    batch_size = 4\n    \n    # Device\n    device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n\nCFG = InferenceConfig()\n\nprint(f\"Device: {CFG.device}\")\nprint(f\"Model directory: {CFG.model_dir}\")\nprint(f\"Evaluating on: {CFG.eval_on} set\")\nprint(f\"Data split CSV: {CFG.data_split_csv}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-03T09:26:25.565514Z","iopub.execute_input":"2026-02-03T09:26:25.565953Z","iopub.status.idle":"2026-02-03T09:26:25.572941Z","shell.execute_reply.started":"2026-02-03T09:26:25.565931Z","shell.execute_reply":"2026-02-03T09:26:25.57209Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 4. Load Data Split","metadata":{}},{"cell_type":"code","source":"print(\"=\"*60)\nprint(\"Loading Data Split from Training\")\nprint(\"=\"*60)\n\n# Load the split CSV from training\nsplit_df = pd.read_csv(CFG.data_split_csv)\n\nprint(f\"\\nLoaded split with {len(split_df)} total samples\")\nprint(f\"\\nSplit distribution:\")\nprint(split_df['Split'].value_counts())\n\nprint(f\"\\nPositive/Negative by split:\")\nprint(split_df.groupby(['Split', 'HasAneurysm']).size().unstack(fill_value=0))\n\n# Get series IDs for the split we want to evaluate\nif CFG.eval_on == \"all\":\n    eval_series_ids = set(split_df['SeriesInstanceUID'].values)\n    print(f\"\\n✅ Will evaluate on ALL {len(eval_series_ids)} samples\")\nelse:\n    eval_series_ids = set(split_df[split_df['Split'] == CFG.eval_on]['SeriesInstanceUID'].values)\n    n_pos = split_df[(split_df['Split'] == CFG.eval_on) & (split_df['HasAneurysm'] == 1.0)].shape[0]\n    n_neg = split_df[(split_df['Split'] == CFG.eval_on) & (split_df['HasAneurysm'] == 0.0)].shape[0]\n    print(f\"\\n✅ Will evaluate on {CFG.eval_on.upper()} set: {len(eval_series_ids)} samples\")\n    print(f\"   Positive: {n_pos} | Negative: {n_neg}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-03T09:26:25.574738Z","iopub.execute_input":"2026-02-03T09:26:25.57508Z","iopub.status.idle":"2026-02-03T09:26:25.665436Z","shell.execute_reply.started":"2026-02-03T09:26:25.575054Z","shell.execute_reply":"2026-02-03T09:26:25.664873Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 5. Build Label Map (Positive + Negative)","metadata":{}},{"cell_type":"code","source":"def build_complete_label_map(csv_path, all_series_ids):\n    \"\"\"\n    Build label map for ALL series (positive AND negative)\n    Same as training notebook\n    \"\"\"\n    print(\"\\n\" + \"=\"*60)\n    print(\"Building Label Map for Evaluation\")\n    print(\"=\"*60)\n    \n    df = pd.read_csv(csv_path)\n    label_map = {}\n    \n    # STEP 1: Positive samples from CSV\n    positive_series = set()\n    for sid in df['SeriesInstanceUID'].unique():\n        labels = np.zeros(CFG.num_classes, dtype=np.float32)\n        series_data = df[df['SeriesInstanceUID'] == sid]\n        \n        for _, row in series_data.iterrows():\n            location = row['location']\n            if location in CFG.target_cols[:-1]:\n                idx = CFG.target_cols.index(location)\n                labels[idx] = 1.0\n        \n        if labels[:-1].sum() > 0:\n            labels[-1] = 1.0\n        \n        label_map[sid] = labels\n        positive_series.add(sid)\n    \n    print(f\"\\n✅ Positive series: {len(positive_series)}\")\n    \n    # STEP 2: Negative samples (rest)\n    negative_series = all_series_ids - positive_series\n    for sid in negative_series:\n        labels = np.zeros(CFG.num_classes, dtype=np.float32)\n        label_map[sid] = labels\n    \n    print(f\"✅ Negative series: {len(negative_series)}\")\n    print(f\"✅ Total series: {len(label_map)}\")\n    \n    return label_map\n\n# Gather all files\nprint(\"\\nGathering all files...\")\nall_files = []\nfor data_dir in CFG.data_dirs:\n    if os.path.exists(data_dir):\n        files = glob.glob(os.path.join(data_dir, \"*.npy\"))\n        all_files.extend(files)\n        print(f\"  {data_dir}: {len(files)} files\")\n\nprint(f\"\\nTotal files: {len(all_files)}\")\n\n# Extract series IDs\nall_series_ids = set([os.path.basename(f).replace('.npy', '') for f in all_files])\nseries_to_file = {os.path.basename(f).replace('.npy', ''): f for f in all_files}\n\n# Build label map\nlabel_map = build_complete_label_map(CFG.csv_path, all_series_ids)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-03T09:26:48.093191Z","iopub.execute_input":"2026-02-03T09:26:48.093838Z","iopub.status.idle":"2026-02-03T09:26:49.035072Z","shell.execute_reply.started":"2026-02-03T09:26:48.093809Z","shell.execute_reply":"2026-02-03T09:26:49.034254Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 6. Get Files for Evaluation","metadata":{}},{"cell_type":"code","source":"# Filter files to only those in our evaluation set\neval_files = [series_to_file[sid] for sid in eval_series_ids if sid in series_to_file]\n\nprint(f\"\\n\" + \"=\"*60)\nprint(f\"Evaluation Set: {CFG.eval_on.upper()}\")\nprint(\"=\"*60)\nprint(f\"Total files: {len(eval_files)}\")\n\n# Count positive/negative in eval set\neval_labels = np.array([label_map[os.path.basename(f).replace('.npy', '')][-1] for f in eval_files])\nn_pos = eval_labels.sum()\nn_neg = len(eval_labels) - n_pos\n\nprint(f\"Positive (with aneurysm): {n_pos:.0f} ({n_pos/len(eval_files)*100:.2f}%)\")\nprint(f\"Negative (no aneurysm):   {n_neg:.0f} ({n_neg/len(eval_files)*100:.2f}%)\")\n\nif len(eval_files) == 0:\n    raise ValueError(f\"No files found for '{CFG.eval_on}' set. Check your data_split.csv and data directories.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-03T09:26:51.475784Z","iopub.execute_input":"2026-02-03T09:26:51.476061Z","iopub.status.idle":"2026-02-03T09:26:51.483834Z","shell.execute_reply.started":"2026-02-03T09:26:51.476039Z","shell.execute_reply":"2026-02-03T09:26:51.482919Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 7. Model Definition","metadata":{}},{"cell_type":"code","source":"class AneurysmClassifier(nn.Module):\n    def __init__(self, model_name=CFG.model_name, num_classes=CFG.num_classes, \n                 in_chans=CFG.in_chans, pretrained=False):\n        super().__init__()\n        self.backbone = timm.create_model(\n            model_name,\n            pretrained=pretrained,\n            num_classes=num_classes,\n            in_chans=in_chans\n        )\n    \n    def forward(self, x):\n        return self.backbone(x)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-03T09:26:53.478043Z","iopub.execute_input":"2026-02-03T09:26:53.478344Z","iopub.status.idle":"2026-02-03T09:26:53.483608Z","shell.execute_reply.started":"2026-02-03T09:26:53.47832Z","shell.execute_reply":"2026-02-03T09:26:53.482863Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 8. Data Preprocessing","metadata":{}},{"cell_type":"code","source":"def get_inference_transform():\n    return A.Compose([\n        A.Resize(CFG.size, CFG.size),\n        A.Normalize(mean=[0.485] * 32, std=[0.229] * 32),\n        ToTensorV2(),\n    ])\n\ndef convert_to_32_channels(img):\n    \"\"\"Convert (512, 512, 3) to (384, 384, 32)\"\"\"\n    img = cv2.resize(img, (384, 384))\n    \n    if img.max() > 1.0:\n        img = img.astype(np.float32) / 255.0\n    \n    channels = []\n    for i in range(32):\n        ratio = (i / 31.0) * 2\n        if ratio < 1:\n            ch = img[:, :, 0] * (1 - ratio) + img[:, :, 1] * ratio\n        else:\n            ratio = ratio - 1\n            ch = img[:, :, 1] * (1 - ratio) + img[:, :, 2] * ratio\n        channels.append(ch)\n    \n    volume = np.stack(channels, axis=-1).astype(np.float32)\n    return volume\n\ndef preprocess_image(npy_path):\n    \"\"\"Load and preprocess a single .npy file\"\"\"\n    img = np.load(npy_path)\n    img = convert_to_32_channels(img)\n    \n    transform = get_inference_transform()\n    augmented = transform(image=img)\n    img_tensor = augmented['image']\n    \n    return img_tensor","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-03T09:26:55.153119Z","iopub.execute_input":"2026-02-03T09:26:55.153689Z","iopub.status.idle":"2026-02-03T09:26:55.160137Z","shell.execute_reply.started":"2026-02-03T09:26:55.153662Z","shell.execute_reply":"2026-02-03T09:26:55.159448Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 9. Load Models","metadata":{}},{"cell_type":"code","source":"def load_model_fold(fold):\n    \"\"\"Load a single fold model\"\"\"\n    model_path = Path(CFG.model_dir) / f'{CFG.model_name}_fold{fold}_best.pth'\n    \n    if not model_path.exists():\n        raise FileNotFoundError(f\"Model file not found: {model_path}\")\n    \n    print(f\"Loading fold {fold} model from {model_path}\")\n    \n    checkpoint = torch.load(model_path, map_location=CFG.device,weights_only=False)\n    \n    model = AneurysmClassifier(\n        model_name=CFG.model_name,\n        num_classes=CFG.num_classes,\n        in_chans=CFG.in_chans,\n        pretrained=False\n    )\n    \n    model.load_state_dict(checkpoint['model'])\n    model = model.to(CFG.device)\n    model.eval()\n    \n    acc = checkpoint.get('accuracy', 'N/A')\n    epoch = checkpoint.get('epoch', 'N/A')\n    print(f\"✅ Loaded fold {fold} (Epoch: {epoch}, Accuracy: {acc})\")\n    return model\n\ndef load_all_models():\n    \"\"\"Load all fold models\"\"\"\n    models = {}\n    \n    print(\"\\n\" + \"=\"*60)\n    print(\"Loading Models\")\n    print(\"=\"*60)\n    \n    for fold in CFG.trn_fold:\n        try:\n            models[fold] = load_model_fold(fold)\n        except Exception as e:\n            print(f\"⚠️  Warning: Could not load fold {fold}: {e}\")\n    \n    if not models:\n        raise ValueError(\"No models were loaded successfully\")\n    \n    print(f\"\\n✅ Loaded {len(models)} models: folds {list(models.keys())}\")\n    \n    # Warm up\n    print(\"Warming up models...\")\n    dummy_image = torch.randn(1, CFG.in_chans, CFG.size, CFG.size).to(CFG.device)\n    with torch.no_grad():\n        for fold, model in models.items():\n            _ = model(dummy_image)\n    \n    print(\"✅ Models ready for inference!\")\n    return models\n\n# Load models\nmodels = load_all_models()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-03T09:27:32.502844Z","iopub.execute_input":"2026-02-03T09:27:32.503133Z","iopub.status.idle":"2026-02-03T09:27:37.59452Z","shell.execute_reply.started":"2026-02-03T09:27:32.50311Z","shell.execute_reply":"2026-02-03T09:27:37.593785Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 10. Prediction Functions","metadata":{}},{"cell_type":"code","source":"@torch.no_grad()\ndef predict_single_model(model, image_tensor):\n    \"\"\"Make prediction with a single model\"\"\"\n    image_tensor = image_tensor.unsqueeze(0).to(CFG.device)\n    \n    with autocast(enabled=CFG.use_amp):\n        output = model(image_tensor)\n        probs = torch.sigmoid(output).cpu().numpy().squeeze()\n    \n    return probs\n\ndef predict_ensemble(models, image_tensor):\n    \"\"\"Make ensemble prediction across all folds\"\"\"\n    all_predictions = []\n    \n    for fold, model in models.items():\n        pred = predict_single_model(model, image_tensor)\n        all_predictions.append(pred)\n    \n    ensemble_pred = np.mean(all_predictions, axis=0)\n    return ensemble_pred\n\ndef predict_single_file(npy_path, models):\n    \"\"\"Predict for a single .npy file\"\"\"\n    image_tensor = preprocess_image(npy_path)\n    predictions = predict_ensemble(models, image_tensor)\n    return predictions","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-03T09:27:40.960764Z","iopub.execute_input":"2026-02-03T09:27:40.961052Z","iopub.status.idle":"2026-02-03T09:27:40.966909Z","shell.execute_reply.started":"2026-02-03T09:27:40.961028Z","shell.execute_reply":"2026-02-03T09:27:40.966166Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 11. Make Predictions on Evaluation Set","metadata":{}},{"cell_type":"code","source":"print(\"\\n\" + \"=\"*60)\nprint(f\"Making Predictions on {CFG.eval_on.upper()} Set\")\nprint(\"=\"*60)\n\nresults = []\nall_predictions = []\nall_labels = []\n\nfor npy_path in tqdm(eval_files, desc=\"Predicting\"):\n    series_id = os.path.basename(npy_path).replace('.npy', '')\n    \n    try:\n        # Get prediction\n        predictions = predict_single_file(npy_path, models)\n        \n        # Get ground truth\n        ground_truth = label_map[series_id]\n        \n        # Store result\n        result = {'SeriesInstanceUID': series_id}\n        for i, col in enumerate(CFG.target_cols):\n            result[f'pred_{col}'] = predictions[i]\n            result[f'true_{col}'] = ground_truth[i]\n        \n        results.append(result)\n        all_predictions.append(predictions)\n        all_labels.append(ground_truth)\n        \n    except Exception as e:\n        print(f\"❌ Failed to process {series_id}: {e}\")\n\n# Convert to arrays\nall_predictions = np.array(all_predictions)\nall_labels = np.array(all_labels)\n\n# Create DataFrame\npredictions_df = pd.DataFrame(results)\nprint(f\"\\n✅ Generated predictions for {len(predictions_df)} samples\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-03T09:27:44.956936Z","iopub.execute_input":"2026-02-03T09:27:44.957684Z","iopub.status.idle":"2026-02-03T09:27:48.317853Z","shell.execute_reply.started":"2026-02-03T09:27:44.957655Z","shell.execute_reply":"2026-02-03T09:27:48.316955Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 12. Evaluation Metrics","metadata":{}},{"cell_type":"code","source":"print(\"\\n\" + \"=\"*60)\nprint(\"EVALUATION METRICS\")\nprint(\"=\"*60)\n\n# Convert predictions to binary (threshold = 0.5)\npred_binary = (all_predictions > 0.5).astype(int)\n\n# Overall accuracy\noverall_accuracy = (pred_binary == all_labels).mean()\nprint(f\"\\n📊 Overall Accuracy: {overall_accuracy:.4f}\")\n\n# Per-class metrics\nprint(\"\\n\" + \"=\"*60)\nprint(\"Per-Class Performance\")\nprint(\"=\"*60)\nprint(f\"{'Class':<45} {'Accuracy':>10} {'Precision':>10} {'Recall':>10}\")\nprint(\"=\"*60)\n\nfor i, col in enumerate(CFG.target_cols):\n    y_true = all_labels[:, i]\n    y_pred = pred_binary[:, i]\n    \n    # Skip if no positive samples\n    if y_true.sum() == 0:\n        print(f\"{col:<45} {'N/A':>10} {'N/A':>10} {'N/A':>10} (no positives)\")\n        continue\n    \n    acc = accuracy_score(y_true, y_pred)\n    \n    # Calculate precision and recall\n    tp = ((y_pred == 1) & (y_true == 1)).sum()\n    fp = ((y_pred == 1) & (y_true == 0)).sum()\n    fn = ((y_pred == 0) & (y_true == 1)).sum()\n    \n    precision = tp / (tp + fp) if (tp + fp) > 0 else 0\n    recall = tp / (tp + fn) if (tp + fn) > 0 else 0\n    \n    print(f\"{col:<45} {acc:>10.4f} {precision:>10.4f} {recall:>10.4f}\")\n\nprint(\"=\"*60)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 13. Confusion Matrix for \"Aneurysm Present\"","metadata":{}},{"cell_type":"code","source":"# Focus on the main task: detecting presence of aneurysm\ny_true_present = all_labels[:, -1]  # Last column is \"Aneurysm Present\"\ny_pred_present = pred_binary[:, -1]\n\n# Confusion matrix\ncm = confusion_matrix(y_true_present, y_pred_present)\n\n# Plot\nplt.figure(figsize=(8, 6))\nsns.heatmap(cm, annot=True, fmt='d', cmap='Blues', \n            xticklabels=['No Aneurysm', 'Aneurysm'],\n            yticklabels=['No Aneurysm', 'Aneurysm'])\nplt.title(f'Confusion Matrix - Aneurysm Detection\\n({CFG.eval_on.upper()} Set)', fontsize=14)\nplt.ylabel('True Label', fontsize=12)\nplt.xlabel('Predicted Label', fontsize=12)\nplt.tight_layout()\nplt.savefig('confusion_matrix.png', dpi=150, bbox_inches='tight')\nplt.show()\n\n# Calculate metrics\ntn, fp, fn, tp = cm.ravel()\naccuracy = (tp + tn) / (tp + tn + fp + fn)\nprecision = tp / (tp + fp) if (tp + fp) > 0 else 0\nrecall = tp / (tp + fn) if (tp + fn) > 0 else 0\nspecificity = tn / (tn + fp) if (tn + fp) > 0 else 0\nf1 = 2 * (precision * recall) / (precision + recall) if (precision + recall) > 0 else 0\n\nprint(\"\\n\" + \"=\"*60)\nprint(\"Aneurysm Detection Metrics\")\nprint(\"=\"*60)\nprint(f\"Accuracy:    {accuracy:.4f}\")\nprint(f\"Precision:   {precision:.4f} (of predicted positives, how many are correct)\")\nprint(f\"Recall:      {recall:.4f} (of true positives, how many did we find)\")\nprint(f\"Specificity: {specificity:.4f} (of true negatives, how many did we identify)\")\nprint(f\"F1 Score:    {f1:.4f}\")\nprint(f\"\\nTrue Positives:  {tp}\")\nprint(f\"True Negatives:  {tn}\")\nprint(f\"False Positives: {fp}\")\nprint(f\"False Negatives: {fn}\")\nprint(\"=\"*60)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 14. Save Results","metadata":{}},{"cell_type":"code","source":"# Save predictions\noutput_path = f'predictions_{CFG.eval_on}_set.csv'\npredictions_df.to_csv(output_path, index=False)\nprint(f\"✅ Saved predictions to {output_path}\")\n\n# Save metrics summary\nmetrics_summary = {\n    'eval_set': CFG.eval_on,\n    'n_samples': len(predictions_df),\n    'n_positive': y_true_present.sum(),\n    'n_negative': len(y_true_present) - y_true_present.sum(),\n    'overall_accuracy': overall_accuracy,\n    'aneurysm_accuracy': accuracy,\n    'aneurysm_precision': precision,\n    'aneurysm_recall': recall,\n    'aneurysm_specificity': specificity,\n    'aneurysm_f1': f1,\n    'true_positives': int(tp),\n    'true_negatives': int(tn),\n    'false_positives': int(fp),\n    'false_negatives': int(fn)\n}\n\nmetrics_df = pd.DataFrame([metrics_summary])\nmetrics_df.to_csv(f'metrics_{CFG.eval_on}_set.csv', index=False)\nprint(f\"✅ Saved metrics to metrics_{CFG.eval_on}_set.csv\")","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 15. Prediction Distribution","metadata":{}},{"cell_type":"code","source":"# Plot prediction distribution\nfig, axes = plt.subplots(1, 2, figsize=(14, 5))\n\n# True positives\ntrue_pos_preds = all_predictions[y_true_present == 1, -1]\naxes[0].hist(true_pos_preds, bins=50, alpha=0.7, color='red', edgecolor='black')\naxes[0].axvline(0.5, color='black', linestyle='--', linewidth=2, label='Threshold=0.5')\naxes[0].set_title(f'Prediction Distribution: TRUE POSITIVES (n={len(true_pos_preds)})', fontsize=12)\naxes[0].set_xlabel('Predicted Probability', fontsize=11)\naxes[0].set_ylabel('Count', fontsize=11)\naxes[0].legend()\naxes[0].grid(alpha=0.3)\n\n# True negatives\ntrue_neg_preds = all_predictions[y_true_present == 0, -1]\naxes[1].hist(true_neg_preds, bins=50, alpha=0.7, color='blue', edgecolor='black')\naxes[1].axvline(0.5, color='black', linestyle='--', linewidth=2, label='Threshold=0.5')\naxes[1].set_title(f'Prediction Distribution: TRUE NEGATIVES (n={len(true_neg_preds)})', fontsize=12)\naxes[1].set_xlabel('Predicted Probability', fontsize=11)\naxes[1].set_ylabel('Count', fontsize=11)\naxes[1].legend()\naxes[1].grid(alpha=0.3)\n\nplt.tight_layout()\nplt.savefig('prediction_distribution.png', dpi=150, bbox_inches='tight')\nplt.show()\n\nprint(\"\\nPrediction Statistics:\")\nprint(f\"True Positives - Mean: {true_pos_preds.mean():.4f}, Median: {np.median(true_pos_preds):.4f}\")\nprint(f\"True Negatives - Mean: {true_neg_preds.mean():.4f}, Median: {np.median(true_neg_preds):.4f}\")","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 16. Summary and Next Steps","metadata":{}},{"cell_type":"code","source":"print(\"\\n\" + \"=\"*60)\nprint(\"INFERENCE COMPLETE\")\nprint(\"=\"*60)\n\nprint(f\"\\n📊 Evaluated on: {CFG.eval_on.upper()} set\")\nprint(f\"   Samples: {len(predictions_df)}\")\nprint(f\"   Positive: {int(y_true_present.sum())} | Negative: {int(len(y_true_present) - y_true_present.sum())}\")\n\nprint(f\"\\n🎯 Key Metrics:\")\nprint(f\"   Overall Accuracy: {overall_accuracy:.4f}\")\nprint(f\"   Aneurysm Detection Accuracy: {accuracy:.4f}\")\nprint(f\"   Precision: {precision:.4f}\")\nprint(f\"   Recall (Sensitivity): {recall:.4f}\")\nprint(f\"   Specificity: {specificity:.4f}\")\nprint(f\"   F1 Score: {f1:.4f}\")\n\nprint(f\"\\n📁 Output Files:\")\nprint(f\"   predictions_{CFG.eval_on}_set.csv\")\nprint(f\"   metrics_{CFG.eval_on}_set.csv\")\nprint(f\"   confusion_matrix.png\")\nprint(f\"   prediction_distribution.png\")\n\nprint(f\"\\n💡 Next Steps:\")\nif CFG.eval_on == \"val\":\n    print(\"   - Validation results look good? Try evaluating on TEST set\")\n    print(\"   - Set CFG.eval_on = 'test' and re-run\")\nelif CFG.eval_on == \"test\":\n    print(\"   - These are your FINAL results on held-out test set\")\n    print(\"   - Use these metrics for reporting model performance\")\n    print(\"   - Consider threshold tuning if precision/recall trade-off needed\")\n\nprint(\"\\n\" + \"=\"*60)","metadata":{},"outputs":[],"execution_count":null}]}