{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":36363,"databundleVersionId":4050810,"isSourceIdPinned":false}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# %% [markdown]\n# # Phase 1: Data Preparation & EDA Pipeline\n# # Professional Research Code - RSNA 2022 Cervical Spine\n\n# %% [code]\n# Install dependencies (Run once)\n!pip install pydicom pylibjpeg pylibjpeg-libjpeg nibabel pandas matplotlib seaborn scikit-learn -q\n\nimport os\nimport logging\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport pydicom\nfrom pathlib import Path\nfrom sklearn.model_selection import train_test_split\nfrom typing import Tuple, List\n\n# Configure Logging\nlogging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s')\nlogger = logging.getLogger(__name__)\n\n# Configuration Class\nclass Config:\n    DATA_PATH = \"/kaggle/input/competitions/rsna-2022-cervical-spine-fracture-detection\"\n    TRAIN_IMAGES = os.path.join(DATA_PATH, \"train_images\")\n    TRAIN_CSV = os.path.join(DATA_PATH, \"train.csv\")\n    BBOX_CSV = os.path.join(DATA_PATH, \"train_bounding_boxes.csv\")\n    SEG_PATH = os.path.join(DATA_PATH, \"segmentations\")\n    \n    # Target Definition: C1 or C2 Fracture (Odontoid focus)\n    TARGET_COLS = ['C1', 'C2']\n    OUTPUT_DIR = \"/kaggle/working/phase1_output\"\n    \n    # Image Processing\n    HU_MIN = -1000\n    HU_MAX = 2000\n    IMG_SIZE = 224\n    USE_PRETRAINED = False\n    # Random State\n    SEED = 42\n\nos.makedirs(Config.OUTPUT_DIR, exist_ok=True)\nlogger.info(f\"Configuration loaded. Output dir: {Config.OUTPUT_DIR}\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-03-10T11:34:35.485284Z","iopub.execute_input":"2026-03-10T11:34:35.486146Z","iopub.status.idle":"2026-03-10T11:34:38.916957Z","shell.execute_reply.started":"2026-03-10T11:34:35.486104Z","shell.execute_reply":"2026-03-10T11:34:38.915862Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# %% [code]\nclass DICOMEngine:\n    \"\"\"\n    Professional DICOM handling with HU rescaling and compression support.\n    \"\"\"\n    @staticmethod\n    def load_patient_series(patient_uid: str) -> List[np.ndarray]:\n        \"\"\"\n        Loads all slices for a patient, sorts them, and applies HU rescaling.\n        Returns a list of numpy arrays (slices).\n        \"\"\"\n        patient_path = Path(Config.TRAIN_IMAGES) / patient_uid\n        if not patient_path.exists():\n            logger.warning(f\"Patient path not found: {patient_uid}\")\n            return []\n        \n        dicom_files = list(patient_path.glob(\"*.dcm\"))\n        if not dicom_files:\n            return []\n        \n        slices = []\n        for df in dicom_files:\n            try:\n                # Use force=True to handle missing headers, pylibjpeg handles compression\n                ds = pydicom.dcmread(df, force=True) \n                \n                # Handle Pixel Data\n                if hasattr(ds, 'PixelData'):\n                    pixel_array = ds.pixel_array.astype(np.float32)\n                    \n                    # HU Rescaling (Critical for CT)\n                    slope = float(ds.RescaleSlope) if hasattr(ds, 'RescaleSlope') else 1.0\n                    intercept = float(ds.RescaleIntercept) if hasattr(ds, 'RescaleIntercept') else 0.0\n                    hu_array = pixel_array * slope + intercept\n                    \n                    # Store with metadata for sorting\n                    slice_loc = float(ds.SliceLocation) if hasattr(ds, 'SliceLocation') else 0.0\n                    slices.append({'hu': hu_array, 'loc': slice_loc, 'instance': ds.InstanceNumber})\n            except Exception as e:\n                logger.error(f\"Error loading {df.name}: {str(e)}\")\n                continue\n        \n        # Sort by Slice Location (Anatomical order)\n        slices.sort(key=lambda x: x['loc'])\n        \n        return [s['hu'] for s in slices]\n\n    @staticmethod\n    def normalize_hu(image: np.ndarray) -> np.ndarray:\n        \"\"\"Clips HU to bone window and normalizes to [0, 1]\"\"\"\n        image = np.clip(image, Config.HU_MIN, Config.HU_MAX)\n        image = (image - Config.HU_MIN) / (Config.HU_MAX - Config.HU_MIN)\n        return image.astype(np.float32)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-10T11:34:38.919059Z","iopub.execute_input":"2026-03-10T11:34:38.919426Z","iopub.status.idle":"2026-03-10T11:34:38.928893Z","shell.execute_reply.started":"2026-03-10T11:34:38.919395Z","shell.execute_reply":"2026-03-10T11:34:38.927975Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# %% [code]\nclass EDAEngine:\n    def __init__(self, df: pd.DataFrame):\n        self.df = df\n        self.setup_targets()\n    \n    def setup_targets(self):\n        \"\"\"Create binary target for C1/C2 (Odontoid) fracture\"\"\"\n        # 1 if C1 OR C2 is fractured, else 0\n        self.df['target_c1_c2'] = (self.df['C1'] == 1) | (self.df['C2'] == 1)\n        self.df['target_c1_c2'] = self.df['target_c1_c2'].astype(int)\n        \n        # Count total fractures per patient\n        vertebrae_cols = ['C1', 'C2', 'C3', 'C4', 'C5', 'C6', 'C7']\n        self.df['fracture_count'] = self.df[vertebrae_cols].sum(axis=1)\n        \n    def generate_report(self) -> pd.DataFrame:\n        \"\"\"Generates statistical summary\"\"\"\n        total = len(self.df)\n        c1_c2_pos = self.df['target_c1_c2'].sum()\n        \n        report = {\n            'Total Patients': total,\n            'C1/C2 Fractures (Positive)': c1_c2_pos,\n            'C1/C2 Fractures (Negative)': total - c1_c2_pos,\n            'Positive Ratio (%)': round(c1_c2_pos / total * 100, 2),\n            'Avg Fractures per Patient': round(self.df['fracture_count'].mean(), 2)\n        }\n        \n        logger.info(\"=== EDA Report ===\")\n        for k, v in report.items():\n            logger.info(f\"{k}: {v}\")\n            \n        return pd.DataFrame([report])\n    \n    def plot_distributions(self, save_path: str):\n        \"\"\"Plots target distribution and fracture counts\"\"\"\n        fig, axes = plt.subplots(1, 2, figsize=(15, 5))\n        \n        # Plot 1: Target Balance\n        sns.countplot(data=self.df, x='target_c1_c2', ax=axes[0], palette='viridis')\n        axes[0].set_title('C1/C2 Fracture Distribution (Binary)')\n        axes[0].set_xlabel('Fracture (1=Yes, 0=No)')\n        \n        # Plot 2: Fracture Count Histogram\n        sns.histplot(data=self.df, x='fracture_count', bins=8, ax=axes[1], color='salmon')\n        axes[1].set_title('Number of Fractured Vertebrae per Patient')\n        \n        plt.tight_layout()\n        plt.savefig(save_path)\n        plt.show()\n        logger.info(f\"Distribution plots saved to {save_path}\")\n\n    def visualize_sample_scans(self, patient_ids: List[str], save_path: str):\n        \"\"\"Visualizes 3 random slices from specific patients\"\"\"\n        fig, axes = plt.subplots(len(patient_ids), 3, figsize=(15, 5 * len(patient_ids)))\n        if len(patient_ids) == 1: axes = np.array([axes]) # Handle single row case\n        \n        for i, pid in enumerate(patient_ids):\n            logger.info(f\"Loading scans for {pid}...\")\n            slices = DICOMEngine.load_patient_series(pid)\n            \n            if not slices:\n                continue\n                \n            # Select 3 representative slices (Top, Middle, Bottom)\n            n = len(slices)\n            indices = [0, n // 2, n - 1] if n >= 3 else list(range(n))\n            \n            for j, idx in enumerate(indices):\n                img = DICOMEngine.normalize_hu(slices[idx])\n                ax = axes[i, j] if len(patient_ids) > 1 else axes[j]\n                ax.imshow(img, cmap='bone')\n                ax.set_title(f'{pid} - Slice {idx}/{n}')\n                ax.axis('off')\n        \n        plt.tight_layout()\n        plt.savefig(save_path)\n        plt.show()\n        logger.info(f\"Sample scans saved to {save_path}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-10T11:34:38.930762Z","iopub.execute_input":"2026-03-10T11:34:38.931039Z","iopub.status.idle":"2026-03-10T11:34:38.949069Z","shell.execute_reply.started":"2026-03-10T11:34:38.931017Z","shell.execute_reply":"2026-03-10T11:34:38.948276Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# %% [code]\nclass DataSplitter:\n    def __init__(self, df: pd.DataFrame):\n        self.df = df\n    \n    def split(self, test_size=0.15, val_size=0.15) -> Tuple[pd.DataFrame, pd.DataFrame, pd.DataFrame]:\n        \"\"\"\n        Stratified split to maintain C1/C2 fracture ratio across sets.\n        \"\"\"\n        # First split: Test set\n        train_temp, test_df = train_test_split(\n            self.df, \n            test_size=test_size, \n            stratify=self.df['target_c1_c2'],\n            random_state=Config.SEED\n        )\n        \n        # Second split: Validation set from remaining training data\n        # Adjust val_size relative to the remaining data\n        actual_val_size = val_size / (1 - test_size)\n        \n        train_df, val_df = train_test_split(\n            train_temp,\n            test_size=actual_val_size,\n            stratify=train_temp['target_c1_c2'],\n            random_state=Config.SEED\n        )\n        \n        logger.info(f\"Split Complete -> Train: {len(train_df)}, Val: {len(val_df)}, Test: {len(test_df)}\")\n        return train_df, val_df, test_df\n\n    def save_metadata(self, train_df, val_df, test_df, output_dir):\n        \"\"\"Saves CSVs with file paths for easy loading in Phase 2\"\"\"\n        for name, df in [('train', train_df), ('val', val_df), ('test', test_df)]:\n            path = os.path.join(output_dir, f\"{name}_metadata.csv\")\n            df.to_csv(path, index=False)\n            logger.info(f\"Saved {name} metadata to {path}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-10T11:34:38.950922Z","iopub.execute_input":"2026-03-10T11:34:38.951172Z","iopub.status.idle":"2026-03-10T11:34:38.970524Z","shell.execute_reply.started":"2026-03-10T11:34:38.951151Z","shell.execute_reply":"2026-03-10T11:34:38.969659Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# %% [code]\ndef run_phase1():\n    logger.info(\"=== Starting Phase 1 Pipeline ===\")\n    \n    # 1. Load Data\n    df = pd.read_csv(Config.TRAIN_CSV)\n    logger.info(f\"Loaded {len(df)} records from train.csv\")\n    \n    # 2. Initialize EDA\n    eda = EDAEngine(df)\n    eda.generate_report()\n    eda.plot_distributions(os.path.join(Config.OUTPUT_DIR, \"distribution_plots.png\"))\n    \n    # 3. Visualize Samples (Pick 2 positive, 2 negative)\n    pos_samples = df[df['target_c1_c2'] == 1]['StudyInstanceUID'].head(2).tolist()\n    neg_samples = df[df['target_c1_c2'] == 0]['StudyInstanceUID'].head(2).tolist()\n    sample_ids = pos_samples + neg_samples\n    \n    eda.visualize_sample_scans(sample_ids, os.path.join(Config.OUTPUT_DIR, \"sample_scans.png\"))\n    \n    # 4. Split Data\n    splitter = DataSplitter(df)\n    train_df, val_df, test_df = splitter.split()\n    splitter.save_metadata(train_df, val_df, test_df, Config.OUTPUT_DIR)\n    \n    # 5. Test DICOM Loading on one patient\n    test_pid = train_df['StudyInstanceUID'].iloc[0]\n    slices = DICOMEngine.load_patient_series(test_pid)\n    logger.info(f\"Test Load Success: Patient {test_pid} has {len(slices)} slices.\")\n    \n    logger.info(\"=== Phase 1 Complete ===\")\n    return train_df, val_df, test_df\n\n# Execute Pipeline\ntrain_meta, val_meta, test_meta = run_phase1()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-10T11:34:38.971599Z","iopub.execute_input":"2026-03-10T11:34:38.971885Z","iopub.status.idle":"2026-03-10T11:35:20.160035Z","shell.execute_reply.started":"2026-03-10T11:34:38.97186Z","shell.execute_reply":"2026-03-10T11:35:20.159443Z"}},"outputs":[],"execution_count":null}]}