{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":99552,"databundleVersionId":13851420,"sourceType":"competition"}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport gc\nimport warnings\nwarnings.filterwarnings('ignore')\n\n\nimport numpy as np\nimport pandas as pd\nimport pydicom\nimport nibabel as nib\nfrom pathlib import Path\nfrom typing import Dict, List, Tuple, Optional, Union\nfrom collections import defaultdict, Counter\nfrom tqdm.auto import tqdm\ntqdm.pandas()\n\nimport cv2\nfrom scipy import ndimage\nfrom scipy.ndimage import zoom, binary_dilation\nfrom skimage import morphology, filters, exposure\nfrom skimage.filters import frangi, sato, meijering\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom IPython.display import display\n\nfrom sklearn.model_selection import StratifiedGroupKFold\nfrom sklearn.preprocessing import LabelEncoder\nimport torch\nimport torch.nn.functional as F\n\ntry:\n    import SimpleITK as sitk\nexcept:\n    print(\"SimpleITK not installed, some features may be limited\")\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-10-06T00:00:55.108566Z","iopub.execute_input":"2025-10-06T00:00:55.10877Z","iopub.status.idle":"2025-10-06T00:01:03.702615Z","shell.execute_reply.started":"2025-10-06T00:00:55.108744Z","shell.execute_reply":"2025-10-06T00:01:03.702048Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install cupy-cuda12x cucim-cu12 --extra-index-url=https://pypi.ngc.nvidia.com","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-06T00:01:03.70399Z","iopub.execute_input":"2025-10-06T00:01:03.704379Z","iopub.status.idle":"2025-10-06T00:01:21.579924Z","shell.execute_reply.started":"2025-10-06T00:01:03.70436Z","shell.execute_reply":"2025-10-06T00:01:21.57921Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from joblib import Parallel, delayed","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-06T00:01:21.58093Z","iopub.execute_input":"2025-10-06T00:01:21.581161Z","iopub.status.idle":"2025-10-06T00:01:21.585373Z","shell.execute_reply.started":"2025-10-06T00:01:21.581137Z","shell.execute_reply":"2025-10-06T00:01:21.584575Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cupy as cp\nimport cucim\nfrom cucim.skimage.filters import frangi, sato, meijering\n\nprint(\"CuPy:\", cp.__version__, \"cuCIM:\", cucim.__version__)\nprint(\"GPU:\", cp.cuda.runtime.getDeviceProperties(0)['name'])\n\n# tiny smoke test on GPU\nx = cp.random.rand(8, 256, 256).astype(cp.float32)  # (D,H,W) on GPU\ny = frangi(x, sigmas=(1,2,3))\nprint(type(y), y.shape, y.device)  # should be <class 'cupy.ndarray'> (8,256,256) <Device 0>\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-06T00:01:21.586015Z","iopub.execute_input":"2025-10-06T00:01:21.586199Z","iopub.status.idle":"2025-10-06T00:01:34.346994Z","shell.execute_reply.started":"2025-10-06T00:01:21.586185Z","shell.execute_reply":"2025-10-06T00:01:34.346215Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def seed_everything(seed=42):\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    if torch.cuda.is_available():\n        torch.cuda.manual_seed(seed)\n        torch.cuda.manual_seed_all(seed)\n\nseed_everything(42)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-06T00:01:34.34788Z","iopub.execute_input":"2025-10-06T00:01:34.348632Z","iopub.status.idle":"2025-10-06T00:01:34.360577Z","shell.execute_reply.started":"2025-10-06T00:01:34.348609Z","shell.execute_reply":"2025-10-06T00:01:34.359834Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"CONFIG = {\n    'TRAIN_CSV': '/kaggle/input/rsna-intracranial-aneurysm-detection/train.csv',\n    'SERIES_DIR': '/kaggle/input/rsna-intracranial-aneurysm-detection/series',\n    'SEGMENTATION_DIR': '/kaggle/input/rsna-intracranial-aneurysm-detection/segmentations',\n    'OUTPUT_DIR': '/kaggle/working/preprocessed',\n    'CACHE_DIR': '/kaggle/working/cache',\n    'N_FOLDS': 5,\n    'SEED': 42,\n    'DEBUG': False,  # Set True to process only subset\n    'DEBUG_SIZE': 100\n}\n\nos.makedirs(CONFIG['OUTPUT_DIR'], exist_ok=True)\nos.makedirs(CONFIG['CACHE_DIR'], exist_ok=True)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-06T00:01:34.361422Z","iopub.execute_input":"2025-10-06T00:01:34.361749Z","iopub.status.idle":"2025-10-06T00:01:34.400024Z","shell.execute_reply.started":"2025-10-06T00:01:34.361725Z","shell.execute_reply":"2025-10-06T00:01:34.399274Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = pd.read_csv(CONFIG['TRAIN_CSV'])\n\nif CONFIG['DEBUG']:\n    train_df = train_df.sample(CONFIG['DEBUG_SIZE'], random_state=42)\n    print(f\"⚠️ DEBUG MODE: Using only {CONFIG['DEBUG_SIZE']} samples\")\n\n\nLABEL_COLS = [\n    'Left Infraclinoid Internal Carotid Artery',\n    'Right Infraclinoid Internal Carotid Artery', \n    'Left Supraclinoid Internal Carotid Artery',\n    'Right Supraclinoid Internal Carotid Artery',\n    'Left Middle Cerebral Artery',\n    'Right Middle Cerebral Artery',\n    'Anterior Communicating Artery',\n    'Left Anterior Cerebral Artery',\n    'Right Anterior Cerebral Artery',\n    'Left Posterior Communicating Artery',\n    'Right Posterior Communicating Artery',\n    'Basilar Tip',\n    'Other Posterior Circulation',\n    'Aneurysm Present'\n]\n\nprint(\"📊 Dataset Overview:\")\nprint(f\"Total series: {len(train_df)}\")\nprint(f\"Unique patients: {train_df['PatientAge'].nunique()}\")\nprint(f\"Modalities: {train_df['Modality'].value_counts().to_dict()}\")\nprint(f\"\\n🎯 Aneurysm prevalence: {train_df['Aneurysm Present'].mean():.2%}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-06T00:01:34.402496Z","iopub.execute_input":"2025-10-06T00:01:34.402705Z","iopub.status.idle":"2025-10-06T00:01:34.455042Z","shell.execute_reply.started":"2025-10-06T00:01:34.402689Z","shell.execute_reply":"2025-10-06T00:01:34.454403Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"location_prev = train_df[LABEL_COLS[:-1]].mean().sort_values(ascending=False)\nprint(\"\\n🔍 Top 5 most common aneurysm locations:\")\nfor loc, prev in location_prev.head().items():\n    print(f\"  {loc}: {prev:.2%}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-06T00:01:34.455687Z","iopub.execute_input":"2025-10-06T00:01:34.455975Z","iopub.status.idle":"2025-10-06T00:01:34.464315Z","shell.execute_reply.started":"2025-10-06T00:01:34.455957Z","shell.execute_reply":"2025-10-06T00:01:34.463726Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class DICOMProcessor:\n    def __init__(self):\n        self.stats_cache = {}\n        \n    @staticmethod\n    def load_dicom_series(series_path: str):\n        dicom_files = list(Path(series_path).glob(\"*.dcm\"))\n\n        if not dicom_files:\n            raise ValueError(f\"No DICOM files found in {series_path}\")\n\n        slices = []\n        for file_path in dicom_files:\n            try:\n                ds = pydicom.dcmread(str(file_path))\n                if hasattr(ds, 'PixelData'):\n                    slices.append(ds)\n            except:\n                continue\n\n        if not slices:\n            raise ValueError(f\"No valid DICOM files with pixel data in {series_path}\")\n\n        slices = DICOMProcessor._sort_slices(slices)\n        metadata = DICOMProcessor._extract_metadata(slices[0])\n        volume = DICOMProcessor._stack_slices(slices)\n        \n        # Handle multi-slice volumes\n        if len(slices) == 1 and volume.ndim == 3:\n            pass\n        elif len(slices) == 1 and volume.ndim == 2:\n            volume = volume[np.newaxis, ...]\n        elif volume.ndim == 4 and volume.shape[0] == 1:\n            volume = volume.squeeze(0)\n        \n        # FIXED: Smart HU conversion - detect if already in HU\n        volume = DICOMProcessor._apply_hu_conversion(\n            volume, \n            metadata.get('rescale_slope', 1.0),\n            metadata.get('rescale_intercept', 0.0),\n            metadata.get('modality', 'Unknown')\n        )\n        \n        return volume.astype(np.float32), metadata\n    \n    @staticmethod\n    def _apply_hu_conversion(volume, slope, intercept, modality):\n        \"\"\"Apply HU conversion only if data is NOT already in HU\"\"\"\n        if modality != 'CT':\n            return volume  # Non-CT doesn't use HU\n        \n        # Detect if already in HU range\n        min_val, max_val = float(np.min(volume)), float(np.max(volume))\n        \n        # HU range check: -1024 (air) to +3071 (dense bone/metal)\n        # If we see negative values or very high positive, it's already HU\n        is_already_hu = (min_val < -100 or max_val > 1000)\n        \n        if is_already_hu:\n            return volume  # Already in HU, don't convert\n        \n        # Raw pixel values - apply conversion\n        return volume * slope + intercept\n        \n    @staticmethod\n    def _sort_slices(slices: List) -> List:\n        \"\"\"Sort DICOM slices by image position\"\"\"\n        def get_position(ds):\n            if hasattr(ds, 'ImagePositionPatient'):\n                return float(ds.ImagePositionPatient[2])\n            elif hasattr(ds, 'SliceLocation'):\n                return float(ds.SliceLocation)\n            else:\n                return float(ds.InstanceNumber)\n        \n        return sorted(slices, key=get_position)\n\n    @staticmethod\n    def _extract_metadata(ds) -> Dict:\n        \"\"\"Extract relevant metadata from DICOM\"\"\"\n        metadata = {\n            'modality': getattr(ds, 'Modality', 'Unknown'),\n            'pixel_spacing': list(map(float, getattr(ds, 'PixelSpacing', [1.0, 1.0]))),\n            'slice_thickness': float(getattr(ds, 'SliceThickness', 1.0)),\n            'spacing_between_slices': float(getattr(ds, 'SpacingBetweenSlices', \n                                           getattr(ds, 'SliceThickness', 1.0))),\n            'rescale_slope': float(getattr(ds, 'RescaleSlope', 1.0)),\n            'rescale_intercept': float(getattr(ds, 'RescaleIntercept', 0.0)),\n            'rows': int(ds.Rows),\n            'columns': int(ds.Columns),\n            'photometric': getattr(ds, 'PhotometricInterpretation', 'MONOCHROME2')\n        }\n        \n        metadata['spacing'] = [\n            metadata['spacing_between_slices'],\n            metadata['pixel_spacing'][0],\n            metadata['pixel_spacing'][1]\n        ]\n        \n        return metadata\n        \n    @staticmethod\n    def _stack_slices(slices: List) -> np.ndarray:\n        \"\"\"Stack DICOM slices into 3D volume\"\"\"\n        if len(slices) == 1:\n            slice_array = slices[0].pixel_array\n            if slices[0].PhotometricInterpretation == \"MONOCHROME1\":\n                slice_array = np.max(slice_array) - slice_array\n            \n            if slice_array.ndim == 3:\n                return slice_array\n            else:\n                return slice_array[np.newaxis, ...]\n        \n        first_slice = slices[0].pixel_array\n        if slices[0].PhotometricInterpretation == \"MONOCHROME1\":\n            first_slice = np.max(first_slice) - first_slice\n        \n        is_rgb = len(first_slice.shape) == 3 and first_slice.shape[-1] == 3\n        if is_rgb:\n            first_slice = cv2.cvtColor(first_slice, cv2.COLOR_RGB2GRAY)\n        \n        volume = np.zeros((len(slices), *first_slice.shape), dtype=first_slice.dtype)\n        \n        for i, ds in enumerate(slices):\n            slice_array = ds.pixel_array\n            if ds.PhotometricInterpretation == \"MONOCHROME1\":\n                slice_array = np.max(slice_array) - slice_array\n            \n            if is_rgb:\n                slice_array = cv2.cvtColor(slice_array, cv2.COLOR_RGB2GRAY)\n            \n            volume[i] = slice_array\n        \n        return volume\n\nprocessor = DICOMProcessor()\nprint(\"✅ DICOMProcessor with smart HU detection initialized\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-06T00:01:34.465065Z","iopub.execute_input":"2025-10-06T00:01:34.465242Z","iopub.status.idle":"2025-10-06T00:01:34.482582Z","shell.execute_reply.started":"2025-10-06T00:01:34.465227Z","shell.execute_reply":"2025-10-06T00:01:34.481903Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class ModalityProcessor:\n    \"\"\"Modality-specific processing with DATA-DRIVEN windows from analysis\"\"\"\n    \n    # Data-driven windows based on actual dataset analysis\n    WINDOWS = {\n        'CT': {\n            # Cleaned median from analysis: Center=60, Width=400\n            'vessel': {'center': 60, 'width': 400},      \n            'soft_tissue': {'center': 60, 'width': 360},  \n            'wide': {'center': 80, 'width': 700},\n        },\n        'MR': {\n            # MR uses percentile normalization, not fixed windows\n            'percentile_range': (1, 99),  # From analysis: [0.0, 259.0]\n        }\n    }\n    \n    @staticmethod\n    def apply_window(volume: np.ndarray, center: float, width: float) -> np.ndarray:\n        \"\"\"Apply CT windowing to volume\"\"\"\n        lower = center - width / 2\n        upper = center + width / 2\n        \n        volume_windowed = np.clip(volume, lower, upper)\n        volume_windowed = (volume_windowed - lower) / width\n        \n        return volume_windowed.astype(np.float32)\n    \n    @staticmethod\n    def process_cta(volume: np.ndarray, metadata: Dict) -> Dict[str, np.ndarray]:\n        \"\"\"Process CT/CTA with data-driven windows\"\"\"\n        outputs = {}\n        \n        # Check if data is in valid HU range\n        min_val, max_val = float(np.min(volume)), float(np.max(volume))\n        is_valid_hu = (min_val < -100 or max_val > 1000)\n        \n        if not is_valid_hu:\n            # Fallback to percentile normalization\n            p1, p99 = np.percentile(volume, [1, 99])\n            if p99 - p1 == 0: \n                p99 = p1 + 1\n            volume_norm = np.clip((volume - p1) / (p99 - p1), 0, 1)\n            outputs['vessel'] = volume_norm.astype(np.float32)\n        else:\n            # Apply data-driven vessel window\n            outputs['vessel'] = ModalityProcessor.apply_window(\n                volume, \n                ModalityProcessor.WINDOWS['CT']['vessel']['center'],\n                ModalityProcessor.WINDOWS['CT']['vessel']['width']\n            )\n        \n        # Create multi-window representation\n        outputs['multi_window'] = np.stack([outputs['vessel']] * 3, axis=0)\n        return outputs\n    \n    @staticmethod\n    def process_mra(volume: np.ndarray, metadata: Dict) -> Dict[str, np.ndarray]:\n        \"\"\"Process MRA with MIP and vessel enhancement\"\"\"\n        outputs = {}\n        \n        # Percentile-based normalization using data-driven range\n        p_low, p_high = ModalityProcessor.WINDOWS['MR']['percentile_range']\n        p1, p99 = np.percentile(volume, [p_low, p_high])\n        \n        volume_norm = np.clip(volume, p1, p99)\n        if p99 > p1:\n            volume_norm = (volume_norm - p1) / (p99 - p1)\n        else:\n            volume_norm = np.zeros_like(volume)\n            \n        outputs['normalized'] = volume_norm\n        \n        # Maximum Intensity Projection (MIP) in 3 directions\n        outputs['mip_axial'] = np.max(volume_norm, axis=0)\n        outputs['mip_coronal'] = np.max(volume_norm, axis=1)\n        outputs['mip_sagittal'] = np.max(volume_norm, axis=2)\n        \n        # TOF enhancement\n        outputs['tof_enhanced'] = ModalityProcessor._enhance_tof(volume_norm)\n        \n        return outputs\n    \n    @staticmethod\n    def process_mri(volume: np.ndarray, metadata: Dict, \n                    sequence_type: str = 'T2') -> Dict[str, np.ndarray]:\n        \"\"\"Process MRI T1/T2 with bias correction and normalization\"\"\"\n        outputs = {}\n        \n        # Simple bias correction\n        outputs['bias_corrected'] = ModalityProcessor._simple_bias_correction(volume)\n        \n        # Z-score normalization with clipping\n        mean = np.mean(outputs['bias_corrected'])\n        std = np.std(outputs['bias_corrected'])\n        \n        volume_zscore = (outputs['bias_corrected'] - mean) / (std + 1e-8)\n        volume_zscore = np.clip(volume_zscore, -3, 3)\n        \n        # Scale to [0, 1]\n        outputs['normalized'] = (volume_zscore + 3) / 6\n        \n        return outputs\n    \n    @staticmethod\n    def _enhance_tof(volume: np.ndarray) -> np.ndarray:\n        \"\"\"Enhance Time-of-Flight MRA signal\"\"\"\n        if volume.shape[0] < 2:\n            return np.clip(volume, 0, 1)\n        \n        gradient = np.gradient(volume, axis=0)\n        flow_enhanced = volume * (1 + np.abs(gradient))\n        return np.clip(flow_enhanced, 0, 1)\n    \n    @staticmethod\n    def _simple_bias_correction(volume: np.ndarray, \n                               kernel_size: int = 50) -> np.ndarray:\n        \"\"\"Simplified N4-like bias correction\"\"\"\n        bias_field = ndimage.gaussian_filter(volume, sigma=kernel_size)\n        bias_field = np.maximum(bias_field, np.percentile(bias_field, 10))\n        \n        corrected = volume / (bias_field + 1e-8)\n        corrected = corrected * np.mean(volume) / np.mean(corrected)\n        \n        return corrected\n\nmodality_processor = ModalityProcessor()\nprint(\"✅ ModalityProcessor with data-driven windows initialized\")\nprint(f\"  CT vessel window: Center={ModalityProcessor.WINDOWS['CT']['vessel']['center']}, \"\n      f\"Width={ModalityProcessor.WINDOWS['CT']['vessel']['width']}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-06T00:01:34.483355Z","iopub.execute_input":"2025-10-06T00:01:34.483621Z","iopub.status.idle":"2025-10-06T00:01:34.503461Z","shell.execute_reply.started":"2025-10-06T00:01:34.4836Z","shell.execute_reply":"2025-10-06T00:01:34.50277Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cupy as cp\nimport numpy as np\nfrom cucim.skimage.filters import frangi as frangi_gpu, sato as sato_gpu, meijering as meijering_gpu\n\ndef preprocess_volume_cucim(volume_np, modality='CT', sigmas=(1, 2, 3), batch=64):\n    \"\"\"\n    GPU preprocessing with CuCIM - handles CT/MR data correctly\n    \n    Args:\n        volume_np: (D, H, W) float32 numpy array\n        modality: 'CT', 'MRA', or 'MRI'\n        sigmas: vessel scale ranges\n        batch: process slices in batches\n    \n    Returns:\n        (3, D, H, W) numpy float32 [frangi, sato, meijering]\n    \"\"\"\n    D, H, W = volume_np.shape\n    \n    # Modality-specific normalization\n    if modality == 'CT':\n        # CT data is already in HU - use dataset 1-99 percentile: [-2048, 595]\n        volume_clipped = np.clip(volume_np, -2048, 595)\n        # Normalize to [0, 1]\n        volume_norm = (volume_clipped + 2048) / (595 + 2048)\n    else:\n        # MR: use percentile normalization\n        p1, p99 = np.percentile(volume_np, [1, 99])\n        volume_norm = np.clip((volume_np - p1) / (p99 - p1 + 1e-6), 0, 1)\n    \n    parts = []\n    \n    for i in range(0, D, batch):\n        v = cp.asarray(volume_norm[i:i+batch], dtype=cp.float32)  # GPU [B,H,W]\n        \n        # Per-slice normalization for filter stability\n        vmin = v.min(axis=(1, 2), keepdims=True)\n        vmax = v.max(axis=(1, 2), keepdims=True)\n        v_stable = (v - vmin) / (vmax - vmin + 1e-6)\n        \n        # Apply vessel filters\n        frv = frangi_gpu(v_stable, sigmas=sigmas)\n        stv = sato_gpu(v_stable, sigmas=sigmas)\n        mjv = meijering_gpu(v_stable, sigmas=sigmas)\n        \n        parts.append(cp.stack([frv, stv, mjv], axis=1))  # (B, 3, H, W)\n        \n        # Clean GPU memory per batch\n        cp.get_default_memory_pool().free_all_blocks()\n    \n    x = cp.concatenate(parts, axis=0)  # (D, 3, H, W)\n    x = x.transpose(1, 0, 2, 3)        # -> (3, D, H, W)\n    \n    return cp.asnumpy(x.astype(cp.float32))\n\n\n# Alternative: Simple gradient-based (faster, no GPU needed)\ndef preprocess_volume_simple(volume_np, modality='CT'):\n    \"\"\"\n    Fast CPU-based vessel enhancement using gradients\n    \n    Args:\n        volume_np: (D, H, W) numpy array\n        modality: 'CT', 'MRA', or 'MRI'\n    \n    Returns:\n        (3, D, H, W) numpy float32\n    \"\"\"\n    # Normalize based on modality\n    if modality == 'CT':\n        # Use dataset 1-99 percentile: [-2048, 595]\n        volume_clipped = np.clip(volume_np, -2048, 595)\n        volume_norm = (volume_clipped + 2048) / (595 + 2048)\n    else:\n        p1, p99 = np.percentile(volume_np, [1, 99])\n        volume_norm = np.clip((volume_np - p1) / (p99 - p1 + 1e-6), 0, 1)\n    \n    if volume_norm.shape[0] < 2:\n        # Single slice - duplicate for 3 channels\n        return np.stack([volume_norm] * 3, axis=0)\n    \n    # Compute gradients\n    grad_z = np.abs(np.gradient(volume_norm, axis=0))\n    grad_y = np.abs(np.gradient(volume_norm, axis=1))\n    grad_x = np.abs(np.gradient(volume_norm, axis=2))\n    \n    # Combined gradient magnitude\n    grad_mag = np.sqrt(grad_x**2 + grad_y**2 + grad_z**2)\n    \n    # Vessel-like enhancement\n    vessel_enhanced = volume_norm + 0.3 * grad_mag\n    vessel_enhanced = np.clip(vessel_enhanced, 0, 1)\n    \n    # Stack as 3 channels: [original, gradient, enhanced]\n    output = np.stack([\n        volume_norm,\n        grad_mag,\n        vessel_enhanced\n    ], axis=0)\n    \n    return output.astype(np.float32)\n\n\n# Usage examples:\n# GPU version (if available):\n# enhanced = preprocess_volume_cucim(volume, modality='CT', sigmas=(1, 2, 3))\n\n# CPU fallback (faster, no GPU):\n# enhanced = preprocess_volume_simple(volume, modality='CT')\n\nprint(\"✅ CuCIM preprocessing functions ready\")\nprint(\"  - preprocess_volume_cucim: GPU-accelerated vessel filters\")\nprint(\"  - preprocess_volume_simple: Fast CPU gradient-based\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-06T00:01:34.504244Z","iopub.execute_input":"2025-10-06T00:01:34.504739Z","iopub.status.idle":"2025-10-06T00:01:34.523346Z","shell.execute_reply.started":"2025-10-06T00:01:34.504719Z","shell.execute_reply":"2025-10-06T00:01:34.522502Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cupy as cp\nimport torch\nimport time\nimport gc\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm.auto import tqdm\n\n# Configuration\nENABLE_GPU_BATCHING = True\nENABLE_PARALLEL = True\nGPU_BATCH_SIZE = 8  # Reduced for stability\nMAX_WORKERS = 2\n\nclass ProcessedDataStore:\n    \"\"\"In-memory storage with LRU eviction\"\"\"\n    def __init__(self, max_items=200):\n        self.data = {}\n        self.max_items = max_items\n        self.access_order = []\n    \n    def add(self, series_id, data):\n        if len(self.data) >= self.max_items and series_id not in self.data:\n            oldest = self.access_order.pop(0)\n            del self.data[oldest]\n        \n        self.data[series_id] = data\n        if series_id in self.access_order:\n            self.access_order.remove(series_id)\n        self.access_order.append(series_id)\n    \n    def get(self, series_id):\n        if series_id in self.data:\n            self.access_order.remove(series_id)\n            self.access_order.append(series_id)\n            return self.data[series_id]\n        return None\n    \n    def get_all(self):\n        return dict(self.data)\n\nprocessed_store = ProcessedDataStore(max_items=200)\n\ndef cleanup_gpu_memory():\n    \"\"\"Emergency GPU cleanup\"\"\"\n    try:\n        if hasattr(cp, 'get_default_memory_pool'):\n            cp.get_default_memory_pool().free_all_blocks()\n        if torch.cuda.is_available():\n            torch.cuda.empty_cache()\n    except:\n        pass\n    gc.collect()\n\ndef process_gpu_batch(batch_data):\n    \"\"\"Process batch on GPU\"\"\"\n    results = []\n    \n    with torch.no_grad():\n        for row_data in batch_data:\n            series_id = row_data['SeriesInstanceUID']\n            \n            try:\n                result = pipeline.process_series(\n                    series_id=series_id,\n                    series_info=row_data,\n                    save_preprocessed=True\n                )\n                \n                if result and 'error' not in result:\n                    processed_store.add(series_id, result)\n                    results.append((series_id, 'Success'))\n                else:\n                    error = result.get('error', 'Unknown') if result else 'No result'\n                    results.append((series_id, f'Error: {error[:100]}'))\n                    \n            except Exception as e:\n                results.append((series_id, f'Error: {str(e)[:100]}'))\n    \n    return results\n\ndef create_smart_batches(df, batch_size=GPU_BATCH_SIZE):\n    \"\"\"Create balanced batches by modality\"\"\"\n    batches = []\n    \n    for modality, group in df.groupby('Modality'):\n        group_batches = [\n            group.iloc[i:i+batch_size].to_dict('records')\n            for i in range(0, len(group), batch_size)\n        ]\n        batches.extend(group_batches)\n    \n    return batches\n\n# Main processing\ndef run_gpu_processing(train_df):\n    \"\"\"Main processing function\"\"\"\n    print(\"=\"*60)\n    print(\"🚀 GPU-ACCELERATED PROCESSING\")\n    print(\"=\"*60)\n    print(f\"Total series: {len(train_df)}\")\n    print(f\"GPU Batching: {ENABLE_GPU_BATCHING}\")\n    print(f\"Batch Size: {GPU_BATCH_SIZE}\")\n    print(f\"Parallel Workers: {MAX_WORKERS if ENABLE_PARALLEL else 'Sequential'}\")\n    print(\"=\"*60)\n    \n    results = []\n    failed = []\n    start_time = time.time()\n    \n    # Create batches\n    batches = create_smart_batches(train_df, GPU_BATCH_SIZE)\n    print(f\"\\nCreated {len(batches)} batches\")\n    \n    if ENABLE_PARALLEL and len(batches) > 1:\n        # Parallel processing\n        with ThreadPoolExecutor(max_workers=MAX_WORKERS) as executor:\n            futures = {executor.submit(process_gpu_batch, batch): i \n                      for i, batch in enumerate(batches)}\n            \n            with tqdm(total=len(train_df), desc=\"Processing\") as pbar:\n                for future in futures:\n                    batch_idx = futures[future]\n                    try:\n                        batch_results = future.result(timeout=300)\n                        \n                        for series_id, status in batch_results:\n                            if 'Error' in status:\n                                failed.append((series_id, status))\n                            results.append((series_id, status))\n                        \n                        pbar.update(len(batches[batch_idx]))\n                        \n                        # Periodic cleanup\n                        if batch_idx % 5 == 0:\n                            cleanup_gpu_memory()\n                            \n                    except Exception as e:\n                        print(f\"\\nBatch {batch_idx} timeout: {e}\")\n                        for row in batches[batch_idx]:\n                            series_id = row['SeriesInstanceUID']\n                            failed.append((series_id, f'Timeout: {str(e)[:50]}'))\n                        pbar.update(len(batches[batch_idx]))\n    else:\n        # Sequential processing\n        with tqdm(batches, desc=\"Processing\") as pbar:\n            for batch in pbar:\n                batch_results = process_gpu_batch(batch)\n                \n                for series_id, status in batch_results:\n                    if 'Error' in status:\n                        failed.append((series_id, status))\n                    results.append((series_id, status))\n                \n                pbar.set_postfix({'Store': len(processed_store.data)})\n    \n    total_time = time.time() - start_time\n    \n    # Summary\n    success_count = sum(1 for r in results if 'Success' in r[1])\n    \n    print(\"\\n\" + \"=\"*60)\n    print(\"✅ PROCESSING COMPLETE\")\n    print(\"=\"*60)\n    print(f\"Success: {success_count}/{len(train_df)} ({success_count/len(train_df)*100:.1f}%)\")\n    print(f\"Failed: {len(failed)}\")\n    print(f\"Stored: {len(processed_store.data)} series\")\n    print(f\"Time: {total_time:.1f}s ({total_time/len(train_df):.2f}s per series)\")\n    \n    if failed:\n        print(f\"\\nFirst 5 errors:\")\n        for series_id, error in failed[:5]:\n            print(f\"  {series_id[:30]}: {error[:80]}\")\n    \n    return results, failed\n\n# Usage\n# results, failed = run_gpu_processing(train_df_with_folds)\n# stored_data = processed_store.get_all()\n\nprint(\"✅ GPU processing script ready\")\nprint(\"Run: results, failed = run_gpu_processing(train_df_with_folds)\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-06T00:01:34.524274Z","iopub.execute_input":"2025-10-06T00:01:34.524545Z","iopub.status.idle":"2025-10-06T00:01:34.543287Z","shell.execute_reply.started":"2025-10-06T00:01:34.524519Z","shell.execute_reply":"2025-10-06T00:01:34.542635Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cupy as cp\nimport numpy as np\nfrom cucim.skimage.filters import frangi as frangi_gpu, sato as sato_gpu, meijering as meijering_gpu\nfrom scipy.ndimage import binary_dilation\nfrom skimage import morphology\n\nclass VesselEnhancer:\n    \"\"\"GPU-accelerated vessel enhancement with smart batching\"\"\"\n    \n    def __init__(self, use_gpu=True):\n        self.use_gpu = use_gpu and self._test_gpu()\n        \n    def _test_gpu(self):\n        \"\"\"Test if GPU is available\"\"\"\n        try:\n            test = cp.ones((10, 10))\n            return True\n        except:\n            return False\n    \n    def enhance_vessels_3d(self, volume, spacing, sigmas=(1, 2, 3), batch_size=64):\n        \"\"\"\n        GPU-accelerated vessel enhancement\n        \n        Args:\n            volume: (D, H, W) numpy array\n            spacing: [z, y, x] spacing in mm\n            sigmas: vessel scale ranges\n            batch_size: process slices in batches for memory efficiency\n        \"\"\"\n        if not self.use_gpu or volume.shape[0] < 2:\n            # Fallback: simple gradient-based enhancement\n            return self._simple_enhancement(volume)\n        \n        try:\n            # GPU processing in batches\n            D, H, W = volume.shape\n            parts = []\n            \n            for i in range(0, D, batch_size):\n                batch = volume[i:i+batch_size]\n                v_gpu = cp.asarray(batch, dtype=cp.float32)\n                \n                # Normalize to [0, 1] for stability\n                vmin = v_gpu.min(axis=(1, 2), keepdims=True)\n                vmax = v_gpu.max(axis=(1, 2), keepdims=True)\n                v_gpu = (v_gpu - vmin) / (vmax - vmin + 1e-6)\n                \n                # Apply vessel filters\n                frangi_result = frangi_gpu(v_gpu, sigmas=sigmas)\n                sato_result = sato_gpu(v_gpu, sigmas=sigmas)\n                meijering_result = meijering_gpu(v_gpu, sigmas=sigmas)\n                \n                # Weighted combination\n                combined = 0.5 * frangi_result + 0.3 * sato_result + 0.2 * meijering_result\n                \n                parts.append(cp.asnumpy(combined))\n                \n                # Clean GPU memory\n                cp.get_default_memory_pool().free_all_blocks()\n            \n            vessel_enhanced = np.concatenate(parts, axis=0)\n            vessel_prob = 1.0 / (1.0 + np.exp(-10.0 * (vessel_enhanced - 0.5)))\n            \n            return {\n                \"combined\": vessel_enhanced.astype(np.float32),\n                \"vessel_prob\": vessel_prob.astype(np.float32)\n            }\n            \n        except Exception as e:\n            print(f\"  GPU vessel enhancement failed: {e}, using fallback\")\n            return self._simple_enhancement(volume)\n    \n    def _simple_enhancement(self, volume):\n        \"\"\"Fast CPU fallback using gradient\"\"\"\n        if volume.shape[0] < 2:\n            return {\"combined\": volume, \"vessel_prob\": volume}\n        \n        grad_z = np.abs(np.gradient(volume, axis=0))\n        grad_y = np.abs(np.gradient(volume, axis=1)) \n        grad_x = np.abs(np.gradient(volume, axis=2))\n        \n        vessel_enhanced = volume + 0.3 * (grad_x + grad_y + grad_z)\n        vessel_prob = np.clip(vessel_enhanced, 0, 1)\n        \n        return {\"combined\": vessel_enhanced, \"vessel_prob\": vessel_prob}\n    \n    @staticmethod\n    def extract_vessel_features(vessel_prob: np.ndarray, percentile_threshold: float = 98.0) -> Dict:\n        \"\"\"Extract vessel features from probability map\"\"\"\n        if vessel_prob.size == 0:\n            return {\n                'vessel_volume_ratio': 0,\n                'vessel_surface_area': 0,\n                'mean_vessel_intensity': 0,\n                'vessel_mask': np.array([])\n            }\n\n        thr = np.percentile(vessel_prob, percentile_threshold) if np.max(vessel_prob) > 0 else 0\n        vessel_mask = vessel_prob > thr\n\n        if np.any(vessel_mask):\n            vessel_mask = morphology.remove_small_objects(vessel_mask, min_size=10)\n            vessel_mask = morphology.remove_small_holes(vessel_mask, area_threshold=10)\n\n        features = {\n            'vessel_volume_ratio': (np.sum(vessel_mask) / vessel_mask.size) if vessel_mask.size else 0,\n            'vessel_surface_area': VesselEnhancer._estimate_surface_area(vessel_mask),\n            'mean_vessel_intensity': float(np.mean(vessel_prob[vessel_mask])) if np.any(vessel_mask) else 0.0,\n            'vessel_mask': vessel_mask\n        }\n        return features\n\n    @staticmethod\n    def _estimate_surface_area(binary_mask: np.ndarray) -> float:\n        \"\"\"Estimate surface area from binary mask\"\"\"\n        if binary_mask.size == 0 or not np.any(binary_mask):\n            return 0.0\n        dilated = binary_dilation(binary_mask, iterations=1)\n        surface = dilated.astype(float) - binary_mask.astype(float)\n        return float(np.sum(surface))\n\nvessel_enhancer = VesselEnhancer()\nprint(f\"✅ VesselEnhancer initialized (GPU: {vessel_enhancer.use_gpu})\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-06T00:01:34.544115Z","iopub.execute_input":"2025-10-06T00:01:34.544433Z","iopub.status.idle":"2025-10-06T00:01:34.657477Z","shell.execute_reply.started":"2025-10-06T00:01:34.544411Z","shell.execute_reply":"2025-10-06T00:01:34.656844Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from collections import OrderedDict\nimport gc\nimport cupy as cp\nimport torch\nfrom scipy.ndimage import zoom\nimport numpy as np\nfrom pathlib import Path\n\nclass Phase1Pipeline:\n    \"\"\"Complete Phase 1 processing pipeline with memory management\"\"\"\n    \n    def __init__(self, \n                 processor: DICOMProcessor,\n                 modality_processor: ModalityProcessor,\n                 vessel_enhancer: VesselEnhancer):\n        self.processor = processor\n        self.modality_processor = modality_processor\n        self.vessel_enhancer = vessel_enhancer\n        \n        self.cache = OrderedDict()\n        self.max_cache_size = 5\n        \n        self.gpu_threshold = 0.75\n        self.ram_threshold = 0.80\n        \n    def _cleanup_cache(self):\n        \"\"\"Enforce cache size limits with LRU eviction\"\"\"\n        while len(self.cache) > self.max_cache_size:\n            oldest_key, _ = self.cache.popitem(last=False)\n    \n    def _force_memory_cleanup(self):\n        \"\"\"Clean memory aggressively\"\"\"\n        try:\n            if hasattr(cp, 'get_default_memory_pool'):\n                cp.get_default_memory_pool().free_all_blocks()\n        except:\n            pass\n        \n        try:\n            if torch.cuda.is_available():\n                torch.cuda.empty_cache()\n        except:\n            pass\n        \n        gc.collect()\n    \n    def process_series(self, \n                      series_id: str,\n                      series_info,\n                      save_preprocessed: bool = False) -> dict:\n        \"\"\"Complete processing for one series\"\"\"\n        \n        if series_id in self.cache:\n            self.cache.move_to_end(series_id)\n            return self.cache[series_id]\n        \n        outputs = {\n            'series_id': series_id,\n            'metadata': {},\n        }\n        \n        try:\n            # Step 1: Load DICOM (with smart HU detection)\n            series_path = Path(CONFIG['SERIES_DIR']) / series_id\n            volume, metadata = self.processor.load_dicom_series(str(series_path))\n            \n            # Downsample if needed\n            TARGET_SIZE = (64, 256,256)\n            if volume.shape[0] > TARGET_SIZE[0] or volume.shape[1] > TARGET_SIZE[1]:\n                zoom_factors = [\n                    min(1.0, TARGET_SIZE[0] / volume.shape[0]),\n                    min(1.0, TARGET_SIZE[1] / volume.shape[1]),\n                    min(1.0, TARGET_SIZE[2] / volume.shape[2])\n                ]\n                volume = zoom(volume, zoom_factors, order=1)\n            \n            outputs['metadata'] = metadata\n            \n            # Handle single-slice volumes\n            if volume.ndim == 2:\n                volume = volume[np.newaxis, ...]\n            elif volume.shape[0] == 1:\n                volume = np.repeat(volume, 3, axis=0)\n            \n            # Step 2: Modality-specific processing (uses data-driven windows)\n            modality = series_info['Modality'].upper()\n            \n            if 'CT' in modality or modality == 'CTA':\n    # Use percentile normalization instead of HU windows\n                p1, p99 = np.percentile(volume, [1, 99])\n                volume_norm = np.clip(volume, p1, p99)\n                volume_norm = (volume_norm - p1) / (p99 - p1 + 1e-8)\n                \n                # Create 3 channels with different percentile ranges for variety\n                p5, p95 = np.percentile(volume, [5, 95])\n                p10, p90 = np.percentile(volume, [10, 90])\n                \n                ch0 = np.clip((volume - p1) / (p99 - p1 + 1e-8), 0, 1)  # Full range\n                ch1 = np.clip((volume - p5) / (p95 - p5 + 1e-8), 0, 1)  # Narrower\n                ch2 = np.clip((volume - p10) / (p90 - p10 + 1e-8), 0, 1)  # Narrowest\n                \n                base_channels = np.stack([ch0, ch1, ch2], axis=0)\n                \n            elif modality == 'MRA':\n                processed = self.modality_processor.process_mra(volume, metadata)\n                primary_volume = processed['normalized'].copy()\n                base_channels = np.stack([primary_volume] * 3, axis=0)\n                del processed, primary_volume\n                \n            else:  # MRI\n                sequence_type = 'T1' if 'T1' in modality else 'T2'\n                processed = self.modality_processor.process_mri(volume, metadata, sequence_type)\n                primary_volume = processed['normalized'].copy()\n                base_channels = np.stack([primary_volume] * 3, axis=0)\n                del processed, primary_volume\n            \n            del volume\n            gc.collect()\n            \n            # Step 3: Create final input (skip vessel enhancement)\n            final_input = base_channels.astype(np.float16)\n            del base_channels\n            gc.collect()\n            \n            outputs['final_input'] = final_input\n            \n            # Save to NPZ file\n            if save_preprocessed:\n                output_path = Path(CONFIG['OUTPUT_DIR']) / f\"{series_id}.npz\"\n                np.savez_compressed(\n                    output_path,\n                    image_data=final_input,\n                    modality=metadata.get('modality', 'Unknown'),\n                    spacing=metadata.get('spacing', [1, 1, 1])\n                )\n            else:\n                output_path = None\n            \n            # Cache minimal result\n            cache_result = {\n                'series_id': series_id,\n                'shape': final_input.shape,\n                'processed': True,\n                'saved_to': str(output_path) if save_preprocessed else None\n            }\n            \n            self.cache[series_id] = cache_result\n            self._cleanup_cache()\n            self._force_memory_cleanup()\n            \n            return cache_result\n            \n        except Exception as e:\n            self._force_memory_cleanup()\n            \n            error_result = {'series_id': series_id, 'error': str(e)}\n            self.cache[series_id] = error_result\n            self._cleanup_cache()\n            \n            return error_result\n\n# Initialize pipeline\npipeline = Phase1Pipeline(\n    processor=processor,\n    modality_processor=modality_processor,\n    vessel_enhancer=vessel_enhancer\n)\n\nprint(\"✅ Phase1Pipeline initialized with 3-channel multi-window processing\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-06T00:01:34.658291Z","iopub.execute_input":"2025-10-06T00:01:34.658546Z","iopub.status.idle":"2025-10-06T00:01:34.674833Z","shell.execute_reply.started":"2025-10-06T00:01:34.658517Z","shell.execute_reply":"2025-10-06T00:01:34.674201Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class SmartSampler:\n    def __init__(self,train_df,n_folds):\n        self.train_df = train_df\n        self.n_folds = n_folds\n        self.fold_assignments = None\n\n    def create_stratified_folds(self):\n        # so here we create stratification based on Aneurysm presence , Modality and Number of aneurysm locations\n\n        df = self.train_df.copy()\n        location_cols = LABEL_COLS[:-1]\n        df['n_locations'] = df[location_cols].sum(axis=1)\n        df['strat_key'] = (\n            df['Aneurysm Present'].astype(str) + '_' +\n            df['Modality'] + '_' +\n            pd.cut(df['n_locations'], bins=[-1, 0, 1, 3, 100], \n                  labels=['0', '1', '2-3', '4+']).astype(str)\n        )\n        df['patient_id'] = df['SeriesInstanceUID'].apply(lambda x: x.split('.')[-1])\n        \n        sgkf = StratifiedGroupKFold(n_splits=self.n_folds, shuffle=True, \n                                    random_state=CONFIG['SEED'])\n        df['fold'] = -1\n        for fold, (train_idx, val_idx) in enumerate(sgkf.split(\n            X=df,\n            y=df['strat_key'],\n            groups=df['patient_id']\n        )):\n            df.loc[val_idx, 'fold'] = fold\n\n\n        print(\"📊 Fold Distribution:\")\n        fold_stats = df.groupby('fold')['Aneurysm Present'].agg(['count', 'mean'])\n        print(fold_stats)\n        \n        self.fold_assignments = df[['SeriesInstanceUID', 'fold']]\n        return df\n\n    def get_balanced_batch_sampler(self,fold,is_training,batch_size):\n        df = self.train_df.merge(self.fold_assignments, on='SeriesInstanceUID')\n        if is_training:\n            df = df[df['fold'] != fold]\n        else:\n            df = df[df['fold'] == fold]\n            \n        positive_samples = df[df['Aneurysm Present'] == 1]['SeriesInstanceUID'].tolist()\n        negative_samples = df[df['Aneurysm Present'] == 0]['SeriesInstanceUID'].tolist()\n\n        batches = []\n        n_positive_per_batch = batch_size // 2\n        n_negative_per_batch = batch_size - n_positive_per_batch\n\n        if len(positive_samples) < len(negative_samples):\n            positive_samples = positive_samples * (len(negative_samples) // len(positive_samples) + 1)\n        \n        # Create balanced batches\n        np.random.shuffle(positive_samples)\n        np.random.shuffle(negative_samples)\n        \n        n_batches = max(len(positive_samples), len(negative_samples)) // (batch_size // 2)\n        \n        for i in range(n_batches):\n            batch = []\n            \n            # Add positive samples\n            start_idx = (i * n_positive_per_batch) % len(positive_samples)\n            end_idx = start_idx + n_positive_per_batch\n            batch.extend(positive_samples[start_idx:end_idx])\n            \n            # Add negative samples\n            start_idx = (i * n_negative_per_batch) % len(negative_samples)\n            end_idx = start_idx + n_negative_per_batch\n            batch.extend(negative_samples[start_idx:end_idx])\n            \n            batches.append(batch)\n        \n        return batches\n        \n    def get_hard_negative_samples(self, \n                                  n_samples: int = 100) -> List[str]:\n        \"\"\"Identify hard negative samples (vessel bifurcations without aneurysms)\"\"\"\n        \n        # Focus on cases with complex vasculature but no aneurysms\n        df = self.train_df.copy()\n        \n        # Negative samples from CTA/MRA (better vessel visualization)\n        hard_negatives = df[\n            (df['Aneurysm Present'] == 0) & \n            (df['Modality'].isin(['CTA', 'MRA']))\n        ]\n        \n        # Prioritize certain age groups (higher aneurysm risk)\n        hard_negatives['age_risk'] = pd.cut(\n            hard_negatives['PatientAge'],\n            bins=[0, 40, 60, 100],\n            labels=[0.5, 1.0, 0.8]\n        ).astype(float)\n        \n        # Sample with weights\n        if len(hard_negatives) > n_samples:\n            hard_negatives = hard_negatives.sample(\n                n_samples,\n                weights='age_risk',\n                random_state=CONFIG['SEED']\n            )\n        \n        return hard_negatives['SeriesInstanceUID'].tolist()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-06T00:01:34.675585Z","iopub.execute_input":"2025-10-06T00:01:34.675818Z","iopub.status.idle":"2025-10-06T00:01:34.694196Z","shell.execute_reply.started":"2025-10-06T00:01:34.675803Z","shell.execute_reply":"2025-10-06T00:01:34.693574Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Initialize processor (was missing)\nprocessor = DICOMProcessor()\nprint(\"✅ DICOM Processor initialized\")\n\n# Initialize smart sampler and create folds\nsampler = SmartSampler(train_df, n_folds=CONFIG['N_FOLDS'])\ntrain_df_with_folds = sampler.create_stratified_folds()\nprint(f\"✅ Created {CONFIG['N_FOLDS']} stratified folds\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-06T00:01:34.694948Z","iopub.execute_input":"2025-10-06T00:01:34.695193Z","iopub.status.idle":"2025-10-06T00:01:36.11583Z","shell.execute_reply.started":"2025-10-06T00:01:34.695171Z","shell.execute_reply":"2025-10-06T00:01:36.115185Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Start preprocessing\ntrain_subset = train_df_with_folds.iloc[:1450]\nprint(f\"Notebook 1: Processing {len(train_subset)} series (0-1449)\")\nresults, failed = run_gpu_processing(train_subset)\n\n# Check results\nprint(f\"\\n✅ Processed: {len(processed_store.data)} series\")\nprint(f\"❌ Failed: {len(failed)}\")\n\n# Access processed data\nstored_data = processed_store.get_all()\nprint(f\"Available series: {len(stored_data)}\")\n\n# Example: Get one series\nif stored_data:\n    sample_id = list(stored_data.keys())[0]\n    sample_data = stored_data[sample_id]\n    print(f\"\\nSample shape: {sample_data['image_data'].shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-06T00:01:36.11666Z","iopub.execute_input":"2025-10-06T00:01:36.117362Z","iopub.status.idle":"2025-10-06T01:51:59.999373Z","shell.execute_reply.started":"2025-10-06T00:01:36.117335Z","shell.execute_reply":"2025-10-06T01:51:59.995467Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import zipfile\nfrom pathlib import Path\n\npreprocessed_dir = Path('/kaggle/working/preprocessed')\nnpz_files = sorted(preprocessed_dir.glob('*.npz'))\n\n# Split into 3 parts\nchunk_size = len(npz_files) // 3\n\nfor i in range(3):\n    start = i * chunk_size\n    end = start + chunk_size if i < 2 else len(npz_files)\n    chunk_files = npz_files[start:end]\n    \n    # Create zip directly without copying\n    zip_path = f'/kaggle/working/preprocessed_part{i+1}.zip'\n    \n    with zipfile.ZipFile(zip_path, 'w', zipfile.ZIP_DEFLATED) as zipf:\n        for f in chunk_files:\n            zipf.write(f, arcname=f.name)\n    \n    size = os.path.getsize(zip_path) / 1e9\n    print(f\"✅ Part {i+1}: {len(chunk_files)} files, {size:.1f} GB\")\n\nprint(\"\\nDownload all 3 zip files from Output tab\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-06T01:55:48.296008Z","iopub.execute_input":"2025-10-06T01:55:48.296289Z","iopub.status.idle":"2025-10-06T01:55:48.333062Z","shell.execute_reply.started":"2025-10-06T01:55:48.296269Z","shell.execute_reply":"2025-10-06T01:55:48.332078Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import shutil\nfrom pathlib import Path\n\n# Delete failed zip attempts\nPath('/kaggle/working/preprocessed_part1.zip').unlink(missing_ok=True)\nshutil.rmtree('/kaggle/working/preprocessed_part1', ignore_errors=True)\n\nprint(\"Cleaned up failed attempts\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-06T01:58:05.67504Z","iopub.execute_input":"2025-10-06T01:58:05.675735Z","iopub.status.idle":"2025-10-06T01:58:05.844036Z","shell.execute_reply.started":"2025-10-06T01:58:05.675709Z","shell.execute_reply":"2025-10-06T01:58:05.843394Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import zipfile\nfrom pathlib import Path\n\npreprocessed_dir = Path('/kaggle/working/preprocessed')\nnpz_files = sorted(preprocessed_dir.glob('*.npz'))\n\nchunk_size = len(npz_files) // 3\n\nfor i in range(3):\n    start = i * chunk_size\n    end = start + chunk_size if i < 2 else len(npz_files)\n    chunk_files = npz_files[start:end]\n    \n    zip_path = f'/kaggle/working/part{i+1}.zip'\n    \n    print(f\"Creating part{i+1}.zip with {len(chunk_files)} files...\")\n    \n    with zipfile.ZipFile(zip_path, 'w', zipfile.ZIP_DEFLATED) as zipf:\n        for j, f in enumerate(chunk_files):\n            zipf.write(f, arcname=f.name)\n            f.unlink()  # Delete NPZ after adding to zip\n            if j % 100 == 0:\n                print(f\"  Progress: {j}/{len(chunk_files)}\")\n    \n    size = Path(zip_path).stat().st_size / 1e9\n    print(f\"✅ Part {i+1}: {size:.1f} GB\\n\")\n\nprint(\"Done! Download part1.zip, part2.zip, part3.zip from Output tab\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-06T01:59:48.214045Z","iopub.execute_input":"2025-10-06T01:59:48.214329Z","iopub.status.idle":"2025-10-06T02:11:38.304956Z","shell.execute_reply.started":"2025-10-06T01:59:48.21431Z","shell.execute_reply":"2025-10-06T02:11:38.304242Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from kaggle_datasets import KaggleDatasets\nfrom kaggle.api.kaggle_api_extended import KaggleApi\n\napi = KaggleApi()\napi.authenticate()\n\n# Create/update dataset version\napi.dataset_create_version(\n    folder='/kaggle/working',\n    version_notes='Updated with new parts',\n    quiet=False\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-06T02:25:19.735784Z","iopub.execute_input":"2025-10-06T02:25:19.736372Z","iopub.status.idle":"2025-10-06T02:25:19.751953Z","shell.execute_reply.started":"2025-10-06T02:25:19.736353Z","shell.execute_reply":"2025-10-06T02:25:19.751019Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}