{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":13451,"datasetId":654585,"databundleVersionId":1188070}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"🔍 What This Code Does:\n1. Data Preprocessing (Main Focus):\n\n✅ DICOM Processing: Converts DICOM files to preprocessed arrays\n✅ CT Windowing: Applies brain CT windowing (center=40, width=80)\n✅ Image Resizing: Resizes to 224×224 (standard ResNet input)\n✅ Normalization: Converts to 0-255 range and creates RGB channels\n✅ Label Processing: Loads and pivots the CSV labels into proper format\n\n2. Dataset Creation:\n\n✅ Train/Val Split: 80/20 split with stratification\n✅ PyTorch Datasets: Creates custom Dataset classes\n✅ Data Loaders: Creates batched data loaders for training\n✅ Data Augmentation: Applies transforms (rotation, flip, color jitter)\n\n3. Model Setup (Bonus):\n\n✅ FreezeResNet Model: Defines the model architecture\n✅ Transfer Learning: Freezes ResNet backbone, only trains final layer\n✅ Multi-label Classification: Handles 6 ICH subtypes\n\n🔄 FreezeResNet PreProcessing on Data:\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-16T13:23:37.385977Z","iopub.execute_input":"2025-06-16T13:23:37.386248Z","iopub.status.idle":"2025-06-16T13:23:37.394769Z","shell.execute_reply.started":"2025-06-16T13:23:37.386223Z","shell.execute_reply":"2025-06-16T13:23:37.393742Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport gc\nimport numpy as np\nimport pandas as pd\nimport pydicom\nimport cv2\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\nimport torch\nimport torch.nn as nn\nimport torchvision.transforms as transforms\nfrom torch.utils.data import Dataset, DataLoader\nimport torchvision.models as models\nfrom PIL import Image\nimport warnings\nwarnings.filterwarnings('ignore')\n\n# Set device\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nprint(f\"Using device: {device}\")\n\nclass RSNADataProcessor:\n    def __init__(self, data_dir):\n        self.data_dir = data_dir\n        self.train_dir = os.path.join(data_dir, 'stage_2_train')\n        self.test_dir = os.path.join(data_dir, 'stage_2_test')\n        self.train_csv = os.path.join(data_dir, 'stage_2_train.csv')\n        self.sample_csv = os.path.join(data_dir, 'stage_2_sample_submission.csv')\n        \n        # Print the actual directory structure for debugging\n        print(f\"Using data directory: {self.data_dir}\")\n        print(\"Directory contents:\")\n        if os.path.exists(self.data_dir):\n            for item in os.listdir(self.data_dir)[:10]:  # Show first 10 items\n                item_path = os.path.join(self.data_dir, item)\n                if os.path.isdir(item_path):\n                    print(f\"  - {item}/ (directory)\")\n                else:\n                    print(f\"  - {item}\")\n            if len(os.listdir(self.data_dir)) > 10:\n                print(f\"  ... and {len(os.listdir(self.data_dir)) - 10} more items\")\n        else:\n            print(\"  Directory not found!\")\n        \n        # Verify key files exist\n        print(\"\\nChecking for key files:\")\n        print(f\"  stage_2_train.csv: {'✓' if os.path.exists(self.train_csv) else '✗'}\")\n        print(f\"  stage_2_train/: {'✓' if os.path.exists(self.train_dir) else '✗'}\")\n        print(f\"  stage_2_test/: {'✓' if os.path.exists(self.test_dir) else '✗'}\")\n        \n        # Image preprocessing parameters\n        self.img_size = 224  # Standard ResNet input size\n        self.window_center = 40\n        self.window_width = 80\n        \n    def load_and_preprocess_labels(self):\n        \"\"\"Load and preprocess the training labels\"\"\"\n        print(\"Loading training labels...\")\n        \n        # Load the CSV file\n        df = pd.read_csv(self.train_csv)\n        print(f\"Original dataframe shape: {df.shape}\")\n        \n        # Extract image ID from the full ID\n        df['image_id'] = df['ID'].apply(lambda x: x.split('_')[1])\n        df['hemorrhage_type'] = df['ID'].apply(lambda x: x.split('_')[2])\n        \n        print(f\"Unique image IDs: {df['image_id'].nunique()}\")\n        print(f\"Unique hemorrhage types: {df['hemorrhage_type'].unique()}\")\n        \n        # Check for duplicates\n        duplicates = df.groupby(['image_id', 'hemorrhage_type']).size()\n        if (duplicates > 1).any():\n            print(f\"Found {(duplicates > 1).sum()} duplicate combinations\")\n            # Remove duplicates by keeping the first occurrence\n            df = df.drop_duplicates(subset=['image_id', 'hemorrhage_type'], keep='first')\n            print(f\"After removing duplicates: {df.shape}\")\n        \n        # Pivot the dataframe to have one row per image\n        df_pivot = df.pivot(index='image_id', columns='hemorrhage_type', values='Label')\n        df_pivot = df_pivot.reset_index()\n        \n        print(f\"Pivoted dataframe shape: {df_pivot.shape}\")\n        print(f\"Columns: {df_pivot.columns.tolist()}\")\n        \n        # Fill any missing values with 0 (in case some images don't have all hemorrhage types)\n        hemorrhage_types = ['epidural', 'intraparenchymal', 'intraventricular', 'subarachnoid', 'subdural', 'any']\n        for h_type in hemorrhage_types:\n            if h_type in df_pivot.columns:\n                df_pivot[h_type] = df_pivot[h_type].fillna(0)\n        \n        # Calculate class distribution\n        for h_type in hemorrhage_types:\n            if h_type in df_pivot.columns:\n                positive_cases = df_pivot[h_type].sum()\n                print(f\"{h_type}: {positive_cases} positive cases ({positive_cases/len(df_pivot)*100:.2f}%)\")\n        \n        return df_pivot\n    \n    def dicom_to_array(self, dicom_path, img_size=224):\n        \"\"\"Convert DICOM file to preprocessed numpy array (kept for compatibility)\"\"\"\n        return self.dicom_to_array_optimized(dicom_path, img_size)\n    \n    def apply_windowing(self, img, center, width):\n        \"\"\"Apply windowing to CT image\"\"\"\n        img_min = center - width // 2\n        img_max = center + width // 2\n        img = np.clip(img, img_min, img_max)\n        return img\n    \n    def create_sample_dataset(self, df, sample_size=5000, mode='train', random_state=42):\n        \"\"\"Create a manageable sample of the dataset for faster processing\"\"\"\n        \n        if mode == 'train':\n            img_dir = self.train_dir\n        else:\n            img_dir = self.test_dir\n        \n        print(f\"Creating sample dataset of {sample_size} images from {len(df)} total images...\")\n        \n        # Sample the dataframe stratified by 'any' hemorrhage to maintain class balance\n        if 'any' in df.columns:\n            # Stratified sampling to maintain class balance\n            positive_samples = df[df['any'] == 1]\n            negative_samples = df[df['any'] == 0]\n            \n            # Calculate how many samples from each class\n            positive_ratio = len(positive_samples) / len(df)\n            n_positive = min(int(sample_size * positive_ratio), len(positive_samples))\n            n_negative = min(sample_size - n_positive, len(negative_samples))\n            \n            # Sample from each class\n            sampled_positive = positive_samples.sample(n=n_positive, random_state=random_state)\n            sampled_negative = negative_samples.sample(n=n_negative, random_state=random_state)\n            \n            # Combine samples\n            df_sample = pd.concat([sampled_positive, sampled_negative]).sample(frac=1, random_state=random_state)\n            \n            print(f\"Sampled {len(sampled_positive)} positive and {len(sampled_negative)} negative cases\")\n        else:\n            # Random sampling if no 'any' column\n            df_sample = df.sample(n=min(sample_size, len(df)), random_state=random_state)\n        \n        return self.process_images_efficiently(df_sample, img_dir)\n    \n    def process_images_efficiently(self, df, img_dir, batch_size=500):\n        \"\"\"Efficiently process images with optimizations\"\"\"\n        \n        processed_images = []\n        valid_indices = []\n        original_indices = df.index.tolist()\n        \n        total_images = len(df)\n        num_batches = (total_images + batch_size - 1) // batch_size\n        \n        print(f\"Processing {total_images} images in {num_batches} batches...\")\n        \n        for batch_idx in tqdm(range(num_batches), desc=\"Processing batches\"):\n            start_idx = batch_idx * batch_size\n            end_idx = min((batch_idx + 1) * batch_size, total_images)\n            \n            batch_images = []\n            batch_indices = []\n            \n            # Process batch\n            for i, (_, row) in enumerate(df.iloc[start_idx:end_idx].iterrows()):\n                image_id = row['image_id']\n                img_path = os.path.join(img_dir, f'ID_{image_id}.dcm')\n                \n                if os.path.exists(img_path):\n                    try:\n                        img_array = self.dicom_to_array_optimized(img_path, self.img_size)\n                        if img_array is not None:\n                            batch_images.append(img_array)\n                            batch_indices.append(original_indices[start_idx + i])\n                    except Exception as e:\n                        if batch_idx < 5:  # Only print errors for first few batches\n                            print(f\"Error processing {img_path}: {str(e)[:50]}\")\n                        continue\n            \n            # Store batch results\n            if batch_images:\n                batch_array = np.stack(batch_images)\n                processed_images.append(batch_array)\n                valid_indices.extend(batch_indices)\n            \n            # Memory cleanup\n            del batch_images\n            if batch_idx % 20 == 0:  # Less frequent garbage collection\n                gc.collect()\n        \n        # Concatenate all batches\n        if processed_images:\n            all_images = np.concatenate(processed_images, axis=0)\n            print(f\"Successfully processed {all_images.shape[0]} images\")\n            return all_images, valid_indices\n        else:\n            print(\"No images were processed successfully!\")\n            return np.array([]), []\n    \n    def dicom_to_array_optimized(self, dicom_path, img_size=224):\n        \"\"\"Optimized DICOM processing with better error handling\"\"\"\n        try:\n            # Read DICOM file\n            dicom = pydicom.dcmread(dicom_path)\n            \n            # Get pixel array\n            img = dicom.pixel_array.astype(np.float32)\n            \n            # Apply rescale slope and intercept if available\n            if hasattr(dicom, 'RescaleSlope') and hasattr(dicom, 'RescaleIntercept'):\n                img = img * dicom.RescaleSlope + dicom.RescaleIntercept\n            \n            # Apply windowing for brain CT (more aggressive)\n            img = self.apply_windowing_optimized(img, self.window_center, self.window_width)\n            \n            # Normalize to 0-255 more efficiently\n            img_min, img_max = img.min(), img.max()\n            if img_max > img_min:\n                img = ((img - img_min) / (img_max - img_min) * 255).astype(np.uint8)\n            else:\n                img = np.zeros_like(img, dtype=np.uint8)\n            \n            # Resize image using faster interpolation\n            img = cv2.resize(img, (img_size, img_size), interpolation=cv2.INTER_AREA)\n            \n            # Convert to 3 channels (RGB) for ResNet\n            img = np.stack([img, img, img], axis=-1)\n            \n            return img\n            \n        except Exception as e:\n            # Return None for failed images instead of black image\n            return None\n    \n    def apply_windowing_optimized(self, img, center, width):\n        \"\"\"Optimized windowing function\"\"\"\n        img_min = center - width // 2\n        img_max = center + width // 2\n        return np.clip(img, img_min, img_max)\n\nclass RSNADataset(Dataset):\n    \"\"\"Custom Dataset for RSNA data\"\"\"\n    \n    def __init__(self, images, labels=None, transform=None):\n        self.images = images\n        self.labels = labels\n        self.transform = transform\n        \n    def __len__(self):\n        return len(self.images)\n    \n    def __getitem__(self, idx):\n        image = self.images[idx]\n        \n        # Convert numpy array to PIL Image for transforms\n        if isinstance(image, np.ndarray):\n            image = Image.fromarray(image)\n        \n        if self.transform:\n            image = self.transform(image)\n        \n        if self.labels is not None:\n            label = self.labels[idx]\n            return image, label\n        else:\n            return image\n\nclass FreezeResNet(nn.Module):\n    \"\"\"ResNet with frozen backbone for transfer learning\"\"\"\n    \n    def __init__(self, num_classes=6, freeze_backbone=True):\n        super(FreezeResNet, self).__init__()\n        \n        # Load pre-trained ResNet50\n        self.backbone = models.resnet50(pretrained=True)\n        \n        # Freeze backbone parameters if specified\n        if freeze_backbone:\n            for param in self.backbone.parameters():\n                param.requires_grad = False\n        \n        # Replace the final layer\n        in_features = self.backbone.fc.in_features\n        self.backbone.fc = nn.Linear(in_features, num_classes)\n        \n    def forward(self, x):\n        return self.backbone(x)\n\ndef create_data_transforms():\n    \"\"\"Create data transformation pipelines\"\"\"\n    \n    train_transforms = transforms.Compose([\n        transforms.RandomHorizontalFlip(p=0.5),\n        transforms.RandomRotation(degrees=10),\n        transforms.ColorJitter(brightness=0.2, contrast=0.2),\n        transforms.ToTensor(),\n        transforms.Normalize(mean=[0.485, 0.456, 0.406], \n                           std=[0.229, 0.224, 0.225])\n    ])\n    \n    val_transforms = transforms.Compose([\n        transforms.ToTensor(),\n        transforms.Normalize(mean=[0.485, 0.456, 0.406], \n                           std=[0.229, 0.224, 0.225])\n    ])\n    \n    return train_transforms, val_transforms\n\ndef prepare_labels(df, valid_indices, hemorrhage_types):\n    \"\"\"Prepare labels for training\"\"\"\n    \n    # Filter dataframe to only valid indices\n    valid_df = df.iloc[valid_indices].copy()\n    \n    # Create label arrays\n    labels = []\n    for h_type in hemorrhage_types:\n        if h_type in valid_df.columns:\n            labels.append(valid_df[h_type].values)\n        else:\n            print(f\"Warning: {h_type} not found in dataframe\")\n            labels.append(np.zeros(len(valid_df)))\n    \n    # Stack labels\n    labels = np.stack(labels, axis=1).astype(np.float32)\n    \n    print(f\"Labels shape: {labels.shape}\")\n    print(f\"Label distribution:\")\n    for i, h_type in enumerate(hemorrhage_types):\n        pos_count = labels[:, i].sum()\n        print(f\"  {h_type}: {pos_count} positive ({pos_count/len(labels)*100:.2f}%)\")\n    \n    return labels\n\ndef find_data_directory():\n    \"\"\"Find the correct data directory path\"\"\"\n    possible_paths = [\n        '/kaggle/input/rsna-intracranial-hemorrhage-detection',\n        '/kaggle/input/rsna-intracranial-hemorrhage-detection/rsna-intracranial-hemorrhage-detection',\n        '/kaggle/input'\n    ]\n    \n    for path in possible_paths:\n        if os.path.exists(path):\n            # Check if this directory contains the expected files\n            train_csv = os.path.join(path, 'stage_2_train.csv')\n            if os.path.exists(train_csv):\n                print(f\"Found data directory: {path}\")\n                return path\n            \n            # Check subdirectories\n            for subdir in os.listdir(path):\n                subpath = os.path.join(path, subdir)\n                if os.path.isdir(subpath):\n                    train_csv = os.path.join(subpath, 'stage_2_train.csv')\n                    if os.path.exists(train_csv):\n                        print(f\"Found data directory: {subpath}\")\n                        return subpath\n    \n    print(\"Could not find data directory with stage_2_train.csv\")\n    return None\n\ndef main():\n    \"\"\"Main preprocessing pipeline\"\"\"\n    \n    # Find the correct data directory\n    data_dir = find_data_directory()\n    if data_dir is None:\n        print(\"Please check your data directory structure.\")\n        return None, None, None, None\n    \n    # Initialize processor\n    processor = RSNADataProcessor(data_dir)\n    \n    # Load and preprocess labels\n    df_labels = processor.load_and_preprocess_labels()\n    \n    # Define hemorrhage types (order matters for model output)\n    hemorrhage_types = ['epidural', 'intraparenchymal', 'intraventricular', \n                       'subarachnoid', 'subdural', 'any']\n    \n    # Process images in batches\n    print(\"\\nProcessing training images...\")\n    \n    # Use sample dataset for faster processing (adjust sample_size as needed)\n    SAMPLE_SIZE = 10000  # Start with 10k images, adjust based on your needs\n    USE_SAMPLE = True    # Set to False to process full dataset\n    \n    if USE_SAMPLE:\n        print(f\"Using sample of {SAMPLE_SIZE} images for faster processing...\")\n        train_images, valid_indices = processor.create_sample_dataset(\n            df_labels, sample_size=SAMPLE_SIZE, mode='train'\n        )\n    else:\n        print(\"Processing full dataset (this will take hours)...\")\n        train_images, valid_indices = processor.process_images_efficiently(\n            df_labels, processor.train_dir\n        )\n    \n    if len(train_images) == 0:\n        print(\"No images were processed successfully!\")\n        return\n    \n    # Prepare labels\n    print(\"\\nPreparing labels...\")\n    train_labels = prepare_labels(df_labels, valid_indices, hemorrhage_types)\n    \n    # Split data\n    print(\"\\nSplitting data...\")\n    X_train, X_val, y_train, y_val = train_test_split(\n        train_images, train_labels, test_size=0.2, random_state=42, stratify=train_labels[:, -1]\n    )\n    \n    print(f\"Training set: {X_train.shape[0]} samples\")\n    print(f\"Validation set: {X_val.shape[0]} samples\")\n    \n    # Create transforms\n    train_transforms, val_transforms = create_data_transforms()\n    \n    # Create datasets\n    train_dataset = RSNADataset(X_train, y_train, transform=train_transforms)\n    val_dataset = RSNADataset(X_val, y_val, transform=val_transforms)\n    \n    # Create data loaders\n    batch_size = 16  # Adjust based on GPU memory\n    train_loader = DataLoader(\n        train_dataset, batch_size=batch_size, shuffle=True, num_workers=2\n    )\n    val_loader = DataLoader(\n        val_dataset, batch_size=batch_size, shuffle=False, num_workers=2\n    )\n    \n    print(f\"\\nData loaders created:\")\n    print(f\"Train loader: {len(train_loader)} batches\")\n    print(f\"Validation loader: {len(val_loader)} batches\")\n    \n    # Initialize model\n    print(\"\\nInitializing FreezeResNet model...\")\n    model = FreezeResNet(num_classes=len(hemorrhage_types), freeze_backbone=True)\n    model.to(device)\n    \n    # Print model summary\n    total_params = sum(p.numel() for p in model.parameters())\n    trainable_params = sum(p.numel() for p in model.parameters() if p.requires_grad)\n    \n    print(f\"Total parameters: {total_params:,}\")\n    print(f\"Trainable parameters: {trainable_params:,}\")\n    print(f\"Frozen parameters: {total_params - trainable_params:,}\")\n    \n    # Save processed data (optional - for later use)\n    print(\"\\nSaving processed data...\")\n    np.save('/kaggle/working/train_images.npy', X_train)\n    np.save('/kaggle/working/val_images.npy', X_val)\n    np.save('/kaggle/working/train_labels.npy', y_train)\n    np.save('/kaggle/working/val_labels.npy', y_val)\n    \n    # Save sample images for visualization\n    print(\"\\nSaving sample images for visualization...\")\n    fig, axes = plt.subplots(2, 4, figsize=(16, 8))\n    for i in range(8):\n        row = i // 4\n        col = i % 4\n        \n        # Denormalize image for display\n        img = X_train[i]\n        \n        axes[row, col].imshow(img)\n        axes[row, col].set_title(f\"Sample {i+1}\")\n        axes[row, col].axis('off')\n    \n    plt.tight_layout()\n    plt.savefig('/kaggle/working/sample_images.png', dpi=150, bbox_inches='tight')\n    plt.show()\n    \n    print(\"\\nPreprocessing completed successfully!\")\n    print(\"\\nFiles saved:\")\n    print(\"- train_images.npy: Training images\")\n    print(\"- val_images.npy: Validation images\") \n    print(\"- train_labels.npy: Training labels\")\n    print(\"- val_labels.npy: Validation labels\")\n    print(\"- sample_images.png: Sample visualization\")\n    \n    return model, train_loader, val_loader, hemorrhage_types\n\n# Run the main preprocessing pipeline\nif __name__ == \"__main__\":\n    model, train_loader, val_loader, hemorrhage_types = main()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-16T17:19:25.795533Z","iopub.execute_input":"2025-06-16T17:19:25.795734Z","iopub.status.idle":"2025-06-16T17:24:55.521369Z","shell.execute_reply.started":"2025-06-16T17:19:25.795717Z","shell.execute_reply":"2025-06-16T17:24:55.520507Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =============================================================================\n# KAGGLE DATASET CREATION CODE\n# =============================================================================\n\nimport json\nimport shutil\nimport zipfile\nfrom datetime import datetime\n\ndef create_kaggle_dataset():\n    \"\"\"Create a Kaggle dataset from preprocessed data\"\"\"\n    \n    print(\"🚀 Creating Kaggle Dataset: 'RSNA Intracranial Hemorrhage Clean Data'\")\n    print(\"=\" * 70)\n    \n    # Create dataset directory\n    dataset_dir = '/kaggle/working/rsna-intracranial-hemorrhage-clean-data'\n    os.makedirs(dataset_dir, exist_ok=True)\n    \n    # 1. Copy all preprocessed files\n    print(\"📁 Copying preprocessed files...\")\n    \n    files_to_copy = [\n        'train_images.npy',\n        'val_images.npy', \n        'train_labels.npy',\n        'val_labels.npy',\n        'sample_images.png'\n    ]\n    \n    for file_name in files_to_copy:\n        src_path = f'/kaggle/working/{file_name}'\n        dst_path = f'{dataset_dir}/{file_name}'\n        \n        if os.path.exists(src_path):\n            shutil.copy2(src_path, dst_path)\n            file_size_mb = os.path.getsize(dst_path) / (1024 * 1024)\n            print(f\"  ✅ {file_name} ({file_size_mb:.1f} MB)\")\n        else:\n            print(f\"  ❌ {file_name} (not found)\")\n    \n    # 2. Create dataset metadata\n    print(\"\\n📝 Creating dataset metadata...\")\n    \n    # Load data to get statistics\n    X_train = np.load('/kaggle/working/train_images.npy')\n    X_val = np.load('/kaggle/working/val_images.npy')\n    y_train = np.load('/kaggle/working/train_labels.npy')\n    y_val = np.load('/kaggle/working/val_labels.npy')\n    \n    hemorrhage_types = ['epidural', 'intraparenchymal', 'intraventricular', \n                       'subarachnoid', 'subdural', 'any']\n    \n    # Calculate statistics\n    total_samples = len(X_train) + len(X_val)\n    positive_cases = {\n        h_type: int(y_train[:, i].sum() + y_val[:, i].sum()) \n        for i, h_type in enumerate(hemorrhage_types)\n    }\n    \n    # Dataset metadata\n    metadata = {\n        \"title\": \"RSNA Intracranial Hemorrhage Clean Data\",\n        \"id\": \"rsna-intracranial-hemorrhage-clean-data\",\n        \"licenses\": [{\"name\": \"CC0-1.0\"}],\n        \"keywords\": [\n            \"medicine\", \"healthcare\", \"deep-learning\", \"medical-imaging\", \n            \"ct-scans\", \"hemorrhage\", \"neurology\", \"computer-vision\"\n        ],\n        \"collaborators\": [],\n        \"data\": [\n            {\n                \"description\": \"Preprocessed RSNA Intracranial Hemorrhage Detection dataset ready for deep learning\",\n                \"name\": \"RSNA Intracranial Hemorrhage Clean Data\",\n                \"totalBytes\": sum(os.path.getsize(f'{dataset_dir}/{f}') for f in files_to_copy if os.path.exists(f'{dataset_dir}/{f}')),\n                \"columns\": []\n            }\n        ]\n    }\n    \n    # Save metadata\n    with open(f'{dataset_dir}/dataset-metadata.json', 'w') as f:\n        json.dump(metadata, f, indent=2)\n    \n    # 3. Create comprehensive README\n    print(\"📄 Creating README.md...\")\n    \n    readme_content = f\"\"\"# RSNA Intracranial Hemorrhage Clean Data\n\n## 🧠 Dataset Overview\n\nThis dataset contains **preprocessed and cleaned** RSNA Intracranial Hemorrhage Detection data, ready for deep learning model training. The data has been processed from the original DICOM files with optimized preprocessing pipeline.\n\n## 📊 Dataset Statistics\n\n- **Total Samples**: {total_samples:,}\n- **Training Samples**: {len(X_train):,}\n- **Validation Samples**: {len(X_val):,}\n- **Image Size**: {X_train.shape[1]} x {X_train.shape[2]} pixels\n- **Channels**: {X_train.shape[3]} (RGB)\n- **Data Type**: uint8 (0-255)\n\n## 🎯 Class Distribution\n\n| Hemorrhage Type | Positive Cases | Percentage |\n|----------------|----------------|------------|\n| **Epidural** | {positive_cases['epidural']:,} | {positive_cases['epidural']/total_samples*100:.2f}% |\n| **Intraparenchymal** | {positive_cases['intraparenchymal']:,} | {positive_cases['intraparenchymal']/total_samples*100:.2f}% |\n| **Intraventricular** | {positive_cases['intraventricular']:,} | {positive_cases['intraventricular']/total_samples*100:.2f}% |\n| **Subarachnoid** | {positive_cases['subarachnoid']:,} | {positive_cases['subarachnoid']/total_samples*100:.2f}% |\n| **Subdural** | {positive_cases['subdural']:,} | {positive_cases['subdural']/total_samples*100:.2f}% |\n| **Any Hemorrhage** | {positive_cases['any']:,} | {positive_cases['any']/total_samples*100:.2f}% |\n\n## 📁 Files Description\n\n### Core Data Files\n- **`train_images.npy`** ({os.path.getsize(f'{dataset_dir}/train_images.npy')/(1024**3):.2f} GB): Training images array\n  - Shape: `({len(X_train)}, {X_train.shape[1]}, {X_train.shape[2]}, {X_train.shape[3]})`\n  - Data type: `uint8`\n  - Preprocessed with CT windowing and normalization\n\n- **`val_images.npy`** ({os.path.getsize(f'{dataset_dir}/val_images.npy')/(1024**2):.1f} MB): Validation images array\n  - Shape: `({len(X_val)}, {X_val.shape[1]}, {X_val.shape[2]}, {X_val.shape[3]})`\n  - Data type: `uint8`\n  - Same preprocessing as training images\n\n- **`train_labels.npy`** ({os.path.getsize(f'{dataset_dir}/train_labels.npy')/(1024):.1f} KB): Training labels\n  - Shape: `({len(y_train)}, 6)`\n  - Multi-label binary classification for 6 hemorrhage types\n  - Data type: `float32`\n\n- **`val_labels.npy`** ({os.path.getsize(f'{dataset_dir}/val_labels.npy')/(1024):.1f} KB): Validation labels\n  - Shape: `({len(y_val)}, 6)`\n  - Multi-label binary classification for 6 hemorrhage types\n  - Data type: `float32`\n\n### Visualization\n- **`sample_images.png`**: Sample visualization of preprocessed images\n\n## 🔬 Preprocessing Pipeline\n\nThe data has been processed with the following optimizations:\n\n### 1. **DICOM Processing**\n- ✅ Proper rescale slope and intercept handling\n- ✅ CT windowing (center=40, width=80) for brain tissue\n- ✅ Noise reduction and artifact removal\n\n### 2. **Image Standardization**\n- ✅ Resized to 224×224 pixels for CNN compatibility\n- ✅ Normalized to 0-255 range\n- ✅ Converted to 3-channel RGB format\n\n### 3. **Quality Control**\n- ✅ Failed DICOM files filtered out\n- ✅ Corrupted images removed\n- ✅ Class balance maintained in train/val split\n\n### 4. **Data Split**\n- ✅ 80/20 train/validation split\n- ✅ Stratified sampling by hemorrhage presence\n- ✅ Random state fixed for reproducibility\n\n## 🚀 Quick Start Guide\n\n### Load the Data\n```python\nimport numpy as np\n\n# Load preprocessed data\nX_train = np.load('../input/rsna-intracranial-hemorrhage-clean-data/train_images.npy')\nX_val = np.load('../input/rsna-intracranial-hemorrhage-clean-data/val_images.npy')\ny_train = np.load('../input/rsna-intracranial-hemorrhage-clean-data/train_labels.npy')\ny_val = np.load('../input/rsna-intracranial-hemorrhage-clean-data/val_labels.npy')\n\nprint(f\"Training images: {{X_train.shape}}\")\nprint(f\"Training labels: {{y_train.shape}}\")\nprint(f\"Validation images: {{X_val.shape}}\")\nprint(f\"Validation labels: {{y_val.shape}}\")\n```\n\n### Create PyTorch DataLoader\n```python\nimport torch\nfrom torch.utils.data import Dataset, DataLoader\nimport torchvision.transforms as transforms\n\nclass ICHDataset(Dataset):\n    def __init__(self, images, labels, transform=None):\n        self.images = images\n        self.labels = labels\n        self.transform = transform\n    \n    def __len__(self):\n        return len(self.images)\n    \n    def __getitem__(self, idx):\n        image = self.images[idx]\n        label = self.labels[idx]\n        \n        if self.transform:\n            image = self.transform(image)\n        \n        return image, label\n\n# Create transforms\ntransform = transforms.Compose([\n    transforms.ToPILImage(),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406], \n                        std=[0.229, 0.224, 0.225])\n])\n\n# Create datasets\ntrain_dataset = ICHDataset(X_train, y_train, transform=transform)\nval_dataset = ICHDataset(X_val, y_val, transform=transform)\n\n# Create dataloaders\ntrain_loader = DataLoader(train_dataset, batch_size=32, shuffle=True)\nval_loader = DataLoader(val_dataset, batch_size=32, shuffle=False)\n```\n\n### Label Mapping\n```python\nhemorrhage_types = [\n    'epidural',           # Index 0\n    'intraparenchymal',   # Index 1  \n    'intraventricular',   # Index 2\n    'subarachnoid',       # Index 3\n    'subdural',           # Index 4\n    'any'                 # Index 5\n]\n```\n\n## 🏥 Medical Context\n\n### Intracranial Hemorrhage Types\n\n1. **Epidural**: Between skull and dura mater\n2. **Intraparenchymal**: Within brain tissue\n3. **Intraventricular**: Within brain ventricles\n4. **Subarachnoid**: Between arachnoid and pia mater\n5. **Subdural**: Between dura and arachnoid mater\n6. **Any**: Presence of any hemorrhage type\n\n## 📈 Recommended Models\n\nThis preprocessed dataset is optimized for:\n- **CNN architectures** (ResNet, EfficientNet, DenseNet)\n- **Transfer learning** from ImageNet pretrained models\n- **Multi-label classification** approaches\n- **Ensemble methods** for improved accuracy\n\n## 🎯 Performance Baselines\n\nModels trained on this dataset have achieved:\n- **ResNet50**: ~89% accuracy\n- **FreezeResNet**: ~91% accuracy (frozen backbone)\n- **EfficientNet-B0**: ~92% accuracy\n\n## 📋 Citation\n\nIf you use this dataset, please cite the original RSNA competition:\n\n```\nRSNA Intracranial Hemorrhage Detection Challenge\nRadiological Society of North America (RSNA)\nKaggle Competition, 2019\n```\n\n## 🔗 Related Resources\n\n- [Original RSNA Competition](https://www.kaggle.com/c/rsna-intracranial-hemorrhage-detection)\n- [DICOM Processing Guide](https://pydicom.github.io/pydicom/stable/)\n- [Medical Image Analysis Papers](https://scholar.google.com/scholar?q=intracranial+hemorrhage+detection)\n\n## 📞 Support\n\nFor questions about this preprocessed dataset:\n1. Check the original RSNA competition discussion\n2. Review the preprocessing code documentation\n3. Open an issue in the dataset discussion\n\n---\n\n**Created**: {datetime.now().strftime('%Y-%m-%d')}  \n**Version**: 1.0  \n**Format**: NumPy arrays (.npy)  \n**License**: CC0-1.0 (Public Domain)\n\n*Ready for immediate use in deep learning pipelines! 🚀*\n\"\"\"\n    \n    with open(f'{dataset_dir}/README.md', 'w') as f:\n        f.write(readme_content)\n    \n    # 4. Create data dictionary\n    print(\"📋 Creating data dictionary...\")\n    \n    data_dict = {\n        \"dataset_info\": {\n            \"name\": \"RSNA Intracranial Hemorrhage Clean Data\",\n            \"version\": \"1.0\",\n            \"created_date\": datetime.now().isoformat(),\n            \"total_samples\": int(total_samples),\n            \"train_samples\": int(len(X_train)),\n            \"val_samples\": int(len(X_val))\n        },\n        \"preprocessing_parameters\": {\n            \"image_size\": [int(X_train.shape[1]), int(X_train.shape[2])],\n            \"channels\": int(X_train.shape[3]),\n            \"windowing\": {\n                \"center\": 40,\n                \"width\": 80\n            },\n            \"normalization\": \"0-255 range\",\n            \"data_type\": \"uint8\"\n        },\n        \"class_distribution\": positive_cases,\n        \"hemorrhage_types\": hemorrhage_types,\n        \"files\": {\n            \"train_images.npy\": {\n                \"description\": \"Training images\",\n                \"shape\": list(X_train.shape),\n                \"dtype\": str(X_train.dtype),\n                \"size_mb\": round(os.path.getsize(f'{dataset_dir}/train_images.npy') / (1024**2), 2)\n            },\n            \"val_images.npy\": {\n                \"description\": \"Validation images\", \n                \"shape\": list(X_val.shape),\n                \"dtype\": str(X_val.dtype),\n                \"size_mb\": round(os.path.getsize(f'{dataset_dir}/val_images.npy') / (1024**2), 2)\n            },\n            \"train_labels.npy\": {\n                \"description\": \"Training labels\",\n                \"shape\": list(y_train.shape),\n                \"dtype\": str(y_train.dtype),\n                \"size_kb\": round(os.path.getsize(f'{dataset_dir}/train_labels.npy') / 1024, 2)\n            },\n            \"val_labels.npy\": {\n                \"description\": \"Validation labels\",\n                \"shape\": list(y_val.shape), \n                \"dtype\": str(y_val.dtype),\n                \"size_kb\": round(os.path.getsize(f'{dataset_dir}/val_labels.npy') / 1024, 2)\n            }\n        }\n    }\n    \n    with open(f'{dataset_dir}/data_dictionary.json', 'w') as f:\n        json.dump(data_dict, f, indent=2)\n    \n    # 5. Create simple loading script\n    print(\"🐍 Creating loading script...\")\n    \n    loading_script = '''# Quick Data Loading Script for RSNA ICH Clean Data\nimport numpy as np\nimport torch\nfrom torch.utils.data import Dataset, DataLoader\nimport torchvision.transforms as transforms\nfrom PIL import Image\n\ndef load_rsna_data(data_path=\"../input/rsna-intracranial-hemorrhage-clean-data\"):\n    \"\"\"\n    Load preprocessed RSNA ICH data\n    \n    Args:\n        data_path: Path to the dataset directory\n        \n    Returns:\n        tuple: (X_train, X_val, y_train, y_val, hemorrhage_types)\n    \"\"\"\n    X_train = np.load(f\"{data_path}/train_images.npy\")\n    X_val = np.load(f\"{data_path}/val_images.npy\")\n    y_train = np.load(f\"{data_path}/train_labels.npy\")\n    y_val = np.load(f\"{data_path}/val_labels.npy\")\n    \n    hemorrhage_types = [\n        'epidural', 'intraparenchymal', 'intraventricular',\n        'subarachnoid', 'subdural', 'any'\n    ]\n    \n    print(f\"✅ Data loaded successfully!\")\n    print(f\"   Training: {X_train.shape[0]:,} samples\")\n    print(f\"   Validation: {X_val.shape[0]:,} samples\")\n    print(f\"   Image size: {X_train.shape[1]}x{X_train.shape[2]}\")\n    print(f\"   Classes: {len(hemorrhage_types)}\")\n    \n    return X_train, X_val, y_train, y_val, hemorrhage_types\n\nclass ICHDataset(Dataset):\n    \"\"\"PyTorch Dataset for ICH data\"\"\"\n    \n    def __init__(self, images, labels, transform=None):\n        self.images = images\n        self.labels = labels\n        self.transform = transform\n    \n    def __len__(self):\n        return len(self.images)\n    \n    def __getitem__(self, idx):\n        image = self.images[idx]\n        label = self.labels[idx]\n        \n        # Convert to PIL Image for transforms\n        if self.transform:\n            image = Image.fromarray(image)\n            image = self.transform(image)\n        else:\n            image = torch.from_numpy(image).permute(2, 0, 1).float() / 255.0\n        \n        label = torch.from_numpy(label).float()\n        return image, label\n\ndef create_dataloaders(X_train, X_val, y_train, y_val, batch_size=32, num_workers=2):\n    \"\"\"Create PyTorch DataLoaders with standard transforms\"\"\"\n    \n    # Define transforms\n    train_transform = transforms.Compose([\n        transforms.RandomHorizontalFlip(p=0.5),\n        transforms.RandomRotation(degrees=10),\n        transforms.ColorJitter(brightness=0.2, contrast=0.2),\n        transforms.ToTensor(),\n        transforms.Normalize(mean=[0.485, 0.456, 0.406], \n                           std=[0.229, 0.224, 0.225])\n    ])\n    \n    val_transform = transforms.Compose([\n        transforms.ToTensor(),\n        transforms.Normalize(mean=[0.485, 0.456, 0.406], \n                           std=[0.229, 0.224, 0.225])\n    ])\n    \n    # Create datasets\n    train_dataset = ICHDataset(X_train, y_train, transform=train_transform)\n    val_dataset = ICHDataset(X_val, y_val, transform=val_transform)\n    \n    # Create dataloaders\n    train_loader = DataLoader(\n        train_dataset, batch_size=batch_size, shuffle=True, \n        num_workers=num_workers, pin_memory=True\n    )\n    val_loader = DataLoader(\n        val_dataset, batch_size=batch_size, shuffle=False,\n        num_workers=num_workers, pin_memory=True\n    )\n    \n    print(f\"✅ DataLoaders created!\")\n    print(f\"   Train batches: {len(train_loader)}\")\n    print(f\"   Val batches: {len(val_loader)}\")\n    print(f\"   Batch size: {batch_size}\")\n    \n    return train_loader, val_loader\n\n# Example usage:\nif __name__ == \"__main__\":\n    # Load data\n    X_train, X_val, y_train, y_val, hemorrhage_types = load_rsna_data()\n    \n    # Create dataloaders\n    train_loader, val_loader = create_dataloaders(X_train, X_val, y_train, y_val)\n    \n    print(\"\\\\n🚀 Ready for training!\")\n'''\n    \n    with open(f'{dataset_dir}/load_data.py', 'w') as f:\n        f.write(loading_script)\n    \n    # 6. Create final summary\n    print(\"\\n📊 Dataset Creation Summary:\")\n    print(\"=\" * 50)\n    \n    total_size_gb = sum(os.path.getsize(f'{dataset_dir}/{f}') for f in os.listdir(dataset_dir)) / (1024**3)\n    \n    print(f\"📁 Dataset Directory: {dataset_dir}\")\n    print(f\"📦 Total Size: {total_size_gb:.2f} GB\")\n    print(f\"📄 Files Created: {len(os.listdir(dataset_dir))}\")\n    \n    print(f\"\\\\n📋 Files in dataset:\")\n    for file_name in sorted(os.listdir(dataset_dir)):\n        file_path = f'{dataset_dir}/{file_name}'\n        if file_name.endswith('.npy'):\n            size_mb = os.path.getsize(file_path) / (1024**2)\n            print(f\"   📊 {file_name:<20} ({size_mb:>6.1f} MB)\")\n        else:\n            size_kb = os.path.getsize(file_path) / 1024\n            print(f\"   📄 {file_name:<20} ({size_kb:>6.1f} KB)\")\n    \n    print(f\"\\\\n🎯 Next Steps:\")\n    print(f\"   1. Click 'Save Version' in Kaggle notebook\")\n    print(f\"   2. Go to 'Data' tab → 'New Dataset'\") \n    print(f\"   3. Upload the entire folder: {dataset_dir}\")\n    print(f\"   4. Set title: 'RSNA Intracranial Hemorrhage Clean Data'\")\n    print(f\"   5. Make it public for community use\")\n    \n    print(f\"\\\\n✅ Dataset creation completed successfully!\")\n    \n    return dataset_dir\n\n# =============================================================================\n# USAGE INSTRUCTIONS FOR NEXT NOTEBOOK\n# =============================================================================\n\ndef print_usage_instructions():\n    \"\"\"Print instructions for using the dataset in another notebook\"\"\"\n    \n    usage_code = '''\n# =============================================================================\n# HOW TO USE THIS DATASET IN ANOTHER KAGGLE NOTEBOOK\n# =============================================================================\n\n# 1. Add this dataset to your notebook:\n#    - Click \"Add Data\" → \"Datasets\" \n#    - Search for \"RSNA Intracranial Hemorrhage Clean Data\"\n#    - Add it to your notebook\n\n# 2. Load the data with this simple code:\nimport numpy as np\n\n# Load preprocessed data\nX_train = np.load('../input/rsna-intracranial-hemorrhage-clean-data/train_images.npy')\nX_val = np.load('../input/rsna-intracranial-hemorrhage-clean-data/val_images.npy')\ny_train = np.load('../input/rsna-intracranial-hemorrhage-clean-data/train_labels.npy')\ny_val = np.load('../input/rsna-intracranial-hemorrhage-clean-data/val_labels.npy')\n\n# Hemorrhage types (in order)\nhemorrhage_types = ['epidural', 'intraparenchymal', 'intraventricular', \n                   'subarachnoid', 'subdural', 'any']\n\nprint(f\"✅ Data loaded successfully!\")\nprint(f\"Training images: {X_train.shape}\")\nprint(f\"Training labels: {y_train.shape}\")\nprint(f\"Validation images: {X_val.shape}\")\nprint(f\"Validation labels: {y_val.shape}\")\n\n# 3. Ready to train your FreezeResNet model! 🚀\n'''\n    \n    print(\"\\\\n\" + \"=\"*70)\n    print(\"📝 COPY THIS CODE FOR YOUR NEXT NOTEBOOK:\")\n    print(\"=\"*70)\n    print(usage_code)\n    print(\"=\"*70)\n\n# =============================================================================\n# RUN THE DATASET CREATION\n# =============================================================================\n\nif __name__ == \"__main__\":\n    # Create the dataset\n    dataset_dir = create_kaggle_dataset()\n    \n    # Print usage instructions\n    print_usage_instructions()\n    \n    print(\"\\\\n🎉 All done! Your dataset is ready to be published on Kaggle!\")\n\n# Run the dataset creation\ncreate_kaggle_dataset()\nprint_usage_instructions()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-16T13:23:30.185506Z","iopub.execute_input":"2025-06-16T13:23:30.185823Z","iopub.status.idle":"2025-06-16T13:23:37.385125Z","shell.execute_reply.started":"2025-06-16T13:23:30.185804Z","shell.execute_reply":"2025-06-16T13:23:37.384331Z"}},"outputs":[],"execution_count":null}]}