{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":99552,"databundleVersionId":13747926,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# import os\n# import shutil\n# from collections import defaultdict\n# import pandas as pd\n# import polars as pl\n# import pydicom\n# import numpy as np\n# import kaggle_evaluation.rsna_inference_server","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-14T16:32:33.31438Z","iopub.execute_input":"2025-09-14T16:32:33.315399Z","iopub.status.idle":"2025-09-14T16:32:33.32114Z","shell.execute_reply.started":"2025-09-14T16:32:33.315366Z","shell.execute_reply":"2025-09-14T16:32:33.319175Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ID_COL = 'SeriesInstanceUID'\n\n# LABEL_COLS = [\n#     'Left Infraclinoid Internal Carotid Artery',\n#     'Right Infraclinoid Internal Carotid Artery',\n#     'Left Supraclinoid Internal Carotid Artery',\n#     'Right Supraclinoid Internal Carotid Artery',\n#     'Left Middle Cerebral Artery',\n#     'Right Middle Cerebral Artery',\n#     'Anterior Communicating Artery',\n#     'Left Anterior Cerebral Artery',\n#     'Right Anterior Cerebral Artery',\n#     'Left Posterior Communicating Artery',\n#     'Right Posterior Communicating Artery',\n#     'Basilar Tip',\n#     'Other Posterior Circulation',\n#     'Aneurysm Present',\n# ]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-14T16:32:35.445049Z","iopub.execute_input":"2025-09-14T16:32:35.445367Z","iopub.status.idle":"2025-09-14T16:32:35.456278Z","shell.execute_reply.started":"2025-09-14T16:32:35.445348Z","shell.execute_reply":"2025-09-14T16:32:35.453719Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# # All tags (other than PixelData and SeriesInstanceUID) that may be in a test set dcm file\n\n\n# DICOM_TAG_ALLOWLIST = [\n#     'BitsAllocated',\n#     'BitsStored',\n#     'Columns',\n#     'FrameOfReferenceUID',\n#     'HighBit',\n#     'ImageOrientationPatient',\n#     'ImagePositionPatient',\n#     'InstanceNumber',\n#     'Modality',\n#     'PatientID',\n#     'PhotometricInterpretation',\n#     'PixelRepresentation',\n#     'PixelSpacing',\n#     'PlanarConfiguration',\n#     'RescaleIntercept',\n#     'RescaleSlope',\n#     'RescaleType',\n#     'Rows',\n#     'SOPClassUID',\n#     'SOPInstanceUID',\n#     'SamplesPerPixel',\n#     'SliceThickness',\n#     'SpacingBetweenSlices',\n#     'StudyInstanceUID',\n#     'TransferSyntaxUID',\n# ]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-14T16:32:38.057409Z","iopub.execute_input":"2025-09-14T16:32:38.057754Z","iopub.status.idle":"2025-09-14T16:32:38.064043Z","shell.execute_reply.started":"2025-09-14T16:32:38.057729Z","shell.execute_reply":"2025-09-14T16:32:38.062682Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def extract_features_from_series(series_path: str) -> dict:\n#     \"\"\"Extract basic features from DICOM series for rule-based prediction.\"\"\"\n#     all_filepaths = []\n#     for root, _, files in os.walk(series_path):\n#         for file in files:\n#             if file.endswith('.dcm'):\n#                 all_filepaths.append(os.path.join(root, file))\n#     all_filepaths.sort()\n    \n#     if not all_filepaths:\n#         return {'num_slices': 0, 'modality': 'UNKNOWN', 'slice_thickness': 1.0, \n#                 'pixel_spacing': [1.0, 1.0], 'age': 50, 'sex': 'U'}\n    \n#     # Read first DICOM to get series-level info\n#     try:\n#         ds = pydicom.dcmread(all_filepaths[0], force=True)\n#         features = {\n#             'num_slices': len(all_filepaths),\n#             'modality': getattr(ds, 'Modality', 'UNKNOWN'),\n#             'slice_thickness': float(getattr(ds, 'SliceThickness', 1.0)),\n#             'pixel_spacing': getattr(ds, 'PixelSpacing', [1.0, 1.0]),\n#             'age': int(getattr(ds, 'PatientAge', '050Y').replace('Y', '')),\n#             'sex': getattr(ds, 'PatientSex', 'U')\n#         }\n        \n#         # Convert pixel spacing to float if it exists\n#         if hasattr(features['pixel_spacing'], '__len__'):\n#             features['pixel_spacing'] = [float(x) for x in features['pixel_spacing']]\n        \n#     except Exception as e:\n#         print(f\"Error reading DICOM: {e}\")\n#         features = {'num_slices': len(all_filepaths), 'modality': 'UNKNOWN', \n#                    'slice_thickness': 1.0, 'pixel_spacing': [1.0, 1.0], \n#                    'age': 50, 'sex': 'U'}\n    \n#     return features\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-14T16:32:40.4134Z","iopub.execute_input":"2025-09-14T16:32:40.413751Z","iopub.status.idle":"2025-09-14T16:32:40.429069Z","shell.execute_reply.started":"2025-09-14T16:32:40.413726Z","shell.execute_reply":"2025-09-14T16:32:40.427364Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def rule_based_prediction(features: dict) -> dict:\n#     \"\"\"\n#     Simple rule-based prediction based on imaging characteristics.\n#     This is a basic approach for a first submission.\n#     \"\"\"\n    \n#     # Base probabilities (these are rough estimates based on medical literature)\n#     base_probs = {\n#         'Left Infraclinoid Internal Carotid Artery': 0.02,\n#         'Right Infraclinoid Internal Carotid Artery': 0.02,\n#         'Left Supraclinoid Internal Carotid Artery': 0.03,\n#         'Right Supraclinoid Internal Carotid Artery': 0.03,\n#         'Left Middle Cerebral Artery': 0.08,  # MCA aneurysms are more common\n#         'Right Middle Cerebral Artery': 0.08,\n#         'Anterior Communicating Artery': 0.12,  # AComm aneurysms are very common\n#         'Left Anterior Cerebral Artery': 0.02,\n#         'Right Anterior Cerebral Artery': 0.02,\n#         'Left Posterior Communicating Artery': 0.04,\n#         'Right Posterior Communicating Artery': 0.04,\n#         'Basilar Tip': 0.03,\n#         'Other Posterior Circulation': 0.02,\n#     }\n    \n#     # Adjust probabilities based on features\n#     modality = features.get('modality', 'UNKNOWN')\n#     age = features.get('age', 50)\n#     sex = features.get('sex', 'U')\n#     num_slices = features.get('num_slices', 50)\n    \n#     # Age factor: aneurysm risk increases with age\n#     age_factor = 1.0\n#     if age < 30:\n#         age_factor = 0.3\n#     elif age < 50:\n#         age_factor = 0.7\n#     elif age > 60:\n#         age_factor = 1.5\n#     elif age > 70:\n#         age_factor = 2.0\n    \n#     # Sex factor: women have slightly higher risk for some locations\n#     sex_factor = 1.2 if sex == 'F' else 1.0\n    \n#     # Modality factor: different modalities have different detection capabilities\n#     modality_factor = 1.0\n#     if modality in ['CTA', 'MRA']:  # Better for detecting aneurysms\n#         modality_factor = 1.3\n#     elif modality in ['MRI', 'MR']:\n#         modality_factor = 0.8\n    \n#     # Series size factor: more slices might indicate more detailed study\n#     slice_factor = min(1.5, max(0.5, num_slices / 100))\n    \n#     # Calculate adjusted probabilities\n#     adjusted_probs = {}\n#     for location, base_prob in base_probs.items():\n#         adjusted_prob = base_prob * age_factor * sex_factor * modality_factor * slice_factor\n#         # Add some random variation to avoid identical predictions\n#         noise = np.random.normal(0, 0.01)\n#         adjusted_prob = max(0.001, min(0.999, adjusted_prob + noise))\n#         adjusted_probs[location] = adjusted_prob\n    \n#     # Calculate overall aneurysm present probability\n#     # Use 1 - product of (1 - individual probabilities)\n#     prob_no_aneurysm = 1.0\n#     for prob in adjusted_probs.values():\n#         prob_no_aneurysm *= (1 - prob)\n    \n#     adjusted_probs['Aneurysm Present'] = 1 - prob_no_aneurysm\n    \n#     return adjusted_probs\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-14T16:32:42.784401Z","iopub.execute_input":"2025-09-14T16:32:42.785002Z","iopub.status.idle":"2025-09-14T16:32:42.79559Z","shell.execute_reply.started":"2025-09-14T16:32:42.784974Z","shell.execute_reply":"2025-09-14T16:32:42.794274Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n\n# def predict(series_path: str) -> pl.DataFrame | pd.DataFrame:\n#     \"\"\"Make a prediction using rule-based approach.\"\"\"\n    \n#     series_id = os.path.basename(series_path)\n    \n#     try:\n#         # Extract features from the DICOM series\n#         features = extract_features_from_series(series_path)\n        \n#         # Get rule-based predictions\n#         predictions_dict = rule_based_prediction(features)\n        \n#         # Create prediction row\n#         prediction_row = [series_id]\n#         for col in LABEL_COLS:\n#             prediction_row.append(predictions_dict.get(col, 0.1))\n        \n#         predictions = pl.DataFrame(\n#             data=[prediction_row],\n#             schema=[ID_COL] + LABEL_COLS,\n#             orient='row',\n#         )\n        \n#     except Exception as e:\n#         print(f\"Error in prediction for {series_id}: {e}\")\n#         # Fallback to conservative predictions\n#         predictions = pl.DataFrame(\n#             data=[[series_id] + [0.1] * len(LABEL_COLS)],\n#             schema=[ID_COL] + LABEL_COLS,\n#             orient='row',\n#         )\n    \n#     # Validation\n#     if isinstance(predictions, pl.DataFrame):\n#         assert predictions.columns == [ID_COL] + LABEL_COLS\n#     elif isinstance(predictions, pd.DataFrame):\n#         assert (predictions.columns == [ID_COL] + LABEL_COLS).all()\n#     else:\n#         raise TypeError('The predict function must return a DataFrame')\n    \n#     # IMPORTANT: Required cleanup to prevent disk space issues\n#     shutil.rmtree('/kaggle/shared', ignore_errors=True)\n    \n#     # Return predictions without the ID column as required\n#     return predictions.drop(ID_COL)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-14T16:32:46.067305Z","iopub.execute_input":"2025-09-14T16:32:46.067679Z","iopub.status.idle":"2025-09-14T16:32:46.079374Z","shell.execute_reply.started":"2025-09-14T16:32:46.067654Z","shell.execute_reply":"2025-09-14T16:32:46.077674Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# # Clean up shared directory before starting inference server\n# shutil.rmtree('/kaggle/shared', ignore_errors=True)\n\n# # Initialize and run the inference server\n# inference_server = kaggle_evaluation.rsna_inference_server.RSNAInferenceServer(predict)\n# if os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n#     inference_server.serve()\n# else:\n#     inference_server.run_local_gateway()\n#     display(pl.read_parquet('/kaggle/working/submission.parquet'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-14T16:32:48.596453Z","iopub.execute_input":"2025-09-14T16:32:48.5968Z","iopub.status.idle":"2025-09-14T16:32:52.87135Z","shell.execute_reply.started":"2025-09-14T16:32:48.596778Z","shell.execute_reply":"2025-09-14T16:32:52.870412Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport shutil\nfrom collections import defaultdict\nimport pandas as pd\nimport polars as pl\nimport pydicom\nimport numpy as np\nfrom scipy import ndimage\nfrom skimage import measure, filters, morphology\nimport kaggle_evaluation.rsna_inference_server","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T11:53:49.036036Z","iopub.execute_input":"2025-09-15T11:53:49.037212Z","iopub.status.idle":"2025-09-15T11:53:53.493679Z","shell.execute_reply.started":"2025-09-15T11:53:49.037166Z","shell.execute_reply":"2025-09-15T11:53:53.492212Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ID_COL = 'SeriesInstanceUID'\nLABEL_COLS = [\n    'Left Infraclinoid Internal Carotid Artery',\n    'Right Infraclinoid Internal Carotid Artery',\n    'Left Supraclinoid Internal Carotid Artery',\n    'Right Supraclinoid Internal Carotid Artery',\n    'Left Middle Cerebral Artery',\n    'Right Middle Cerebral Artery',\n    'Anterior Communicating Artery',\n    'Left Anterior Cerebral Artery',\n    'Right Anterior Cerebral Artery',\n    'Left Posterior Communicating Artery',\n    'Right Posterior Communicating Artery',\n    'Basilar Tip',\n    'Other Posterior Circulation',\n    'Aneurysm Present',\n]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T11:54:05.781486Z","iopub.execute_input":"2025-09-15T11:54:05.781836Z","iopub.status.idle":"2025-09-15T11:54:05.787469Z","shell.execute_reply.started":"2025-09-15T11:54:05.781787Z","shell.execute_reply":"2025-09-15T11:54:05.786325Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# All tags (other than PixelData and SeriesInstanceUID) that may be in a test set dcm file\nDICOM_TAG_ALLOWLIST = [\n    'BitsAllocated',\n    'BitsStored',\n    'Columns',\n    'FrameOfReferenceUID',\n    'HighBit',\n    'ImageOrientationPatient',\n    'ImagePositionPatient',\n    'InstanceNumber',\n    'Modality',\n    'PatientID',\n    'PhotometricInterpretation',\n    'PixelRepresentation',\n    'PixelSpacing',\n    'PlanarConfiguration',\n    'RescaleIntercept',\n    'RescaleSlope',\n    'RescaleType',\n    'Rows',\n    'SOPClassUID',\n    'SOPInstanceUID',\n    'SamplesPerPixel',\n    'SliceThickness',\n    'SpacingBetweenSlices',\n    'StudyInstanceUID',\n    'TransferSyntaxUID',\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T11:54:34.514741Z","iopub.execute_input":"2025-09-15T11:54:34.515146Z","iopub.status.idle":"2025-09-15T11:54:34.521998Z","shell.execute_reply.started":"2025-09-15T11:54:34.515112Z","shell.execute_reply":"2025-09-15T11:54:34.520519Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def normalize_image(image):\n    \"\"\"Normalize image intensities to 0-1 range\"\"\"\n    if image.max() == image.min():\n        return image\n    return (image - image.min()) / (image.max() - image.min())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T11:54:45.252252Z","iopub.execute_input":"2025-09-15T11:54:45.252698Z","iopub.status.idle":"2025-09-15T11:54:45.259378Z","shell.execute_reply.started":"2025-09-15T11:54:45.252666Z","shell.execute_reply":"2025-09-15T11:54:45.258232Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def extract_brain_mask(image, threshold_percentile=95):\n    \"\"\"Extract brain region from background\"\"\"\n    try:\n        # Threshold based on high percentile to focus on brain tissue\n        threshold = np.percentile(image[image > 0], threshold_percentile)\n        brain_mask = image > (threshold * 0.3)\n        \n        # Morphological operations to clean up the mask\n        brain_mask = morphology.binary_opening(brain_mask, structure=np.ones((3,3)))\n        brain_mask = morphology.binary_closing(brain_mask, structure=np.ones((5,5)))\n        \n        return brain_mask\n    except:\n        return np.ones_like(image, dtype=bool)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T11:54:54.037482Z","iopub.execute_input":"2025-09-15T11:54:54.037784Z","iopub.status.idle":"2025-09-15T11:54:54.043862Z","shell.execute_reply.started":"2025-09-15T11:54:54.037763Z","shell.execute_reply":"2025-09-15T11:54:54.04277Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def detect_vessel_like_structures(image):\n    \"\"\"Detect potential vessel-like structures using image processing\"\"\"\n    try:\n        # Normalize image\n        norm_image = normalize_image(image)\n        \n        # Apply Gaussian filter to reduce noise\n        smoothed = filters.gaussian(norm_image, sigma=1.0)\n        \n        # Detect edges using Sobel filter\n        edges = filters.sobel(smoothed)\n        \n        # Hessian-based vessel enhancement (simplified version)\n        # This helps highlight tubular structures like blood vessels\n        hessian_xx = filters.gaussian(smoothed, sigma=1.5, order=(2,0))\n        hessian_yy = filters.gaussian(smoothed, sigma=1.5, order=(0,2))\n        hessian_xy = filters.gaussian(smoothed, sigma=1.5, order=(1,1))\n        \n        # Eigenvalue analysis for vessel enhancement\n        det = hessian_xx * hessian_yy - hessian_xy**2\n        trace = hessian_xx + hessian_yy\n        \n        # Vessel probability based on eigenvalues\n        vessel_prob = np.where(trace < 0, np.abs(det), 0)\n        \n        return {\n            'vessel_probability': np.mean(vessel_prob),\n            'edge_strength': np.mean(edges),\n            'high_intensity_regions': np.sum(norm_image > 0.8) / norm_image.size,\n            'texture_variance': np.var(smoothed)\n        }\n    except:\n        return {\n            'vessel_probability': 0.1,\n            'edge_strength': 0.1,\n            'high_intensity_regions': 0.1,\n            'texture_variance': 0.1\n        }","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T11:55:09.224711Z","iopub.execute_input":"2025-09-15T11:55:09.22505Z","iopub.status.idle":"2025-09-15T11:55:09.233935Z","shell.execute_reply.started":"2025-09-15T11:55:09.225026Z","shell.execute_reply":"2025-09-15T11:55:09.232896Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def analyze_slice_connectivity(images):\n    \"\"\"Analyze connectivity patterns across slices\"\"\"\n    try:\n        if len(images) < 3:\n            return {'connectivity_score': 0.1, 'volume_consistency': 0.1}\n        \n        # Calculate inter-slice correlation\n        correlations = []\n        for i in range(len(images)-1):\n            corr = np.corrcoef(images[i].flatten(), images[i+1].flatten())[0,1]\n            if not np.isnan(corr):\n                correlations.append(corr)\n        \n        connectivity_score = np.mean(correlations) if correlations else 0.1\n        \n        # Volume consistency - look for consistent high-intensity regions\n        high_intensity_masks = [img > np.percentile(img, 90) for img in images]\n        volume_consistency = np.mean([np.sum(mask) for mask in high_intensity_masks]) / images[0].size\n        \n        return {\n            'connectivity_score': max(0, connectivity_score),\n            'volume_consistency': volume_consistency\n        }\n    except:\n        return {'connectivity_score': 0.1, 'volume_consistency': 0.1}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T11:55:23.494512Z","iopub.execute_input":"2025-09-15T11:55:23.494886Z","iopub.status.idle":"2025-09-15T11:55:23.503472Z","shell.execute_reply.started":"2025-09-15T11:55:23.494861Z","shell.execute_reply":"2025-09-15T11:55:23.50248Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def extract_advanced_features_from_series(series_path: str) -> dict:\n    \"\"\"Extract comprehensive features from DICOM series including image analysis.\"\"\"\n    all_filepaths = []\n    for root, _, files in os.walk(series_path):\n        for file in files:\n            if file.endswith('.dcm'):\n                all_filepaths.append(os.path.join(root, file))\n    all_filepaths.sort()\n    \n    if not all_filepaths:\n        return get_default_features()\n    \n    # Read metadata from first DICOM\n    try:\n        ds = pydicom.dcmread(all_filepaths[0], force=True)\n        features = extract_metadata_features(ds, len(all_filepaths))\n    except Exception as e:\n        print(f\"Error reading DICOM metadata: {e}\")\n        features = get_default_features()\n    \n    # Sample a subset of images for analysis (to manage computation time)\n    sample_indices = np.linspace(0, len(all_filepaths)-1, min(10, len(all_filepaths)), dtype=int)\n    images = []\n    \n    for idx in sample_indices:\n        try:\n            ds = pydicom.dcmread(all_filepaths[idx], force=True)\n            if hasattr(ds, 'pixel_array'):\n                image = ds.pixel_array.astype(np.float32)\n                \n                # Apply DICOM rescaling if available\n                if hasattr(ds, 'RescaleSlope') and hasattr(ds, 'RescaleIntercept'):\n                    image = image * ds.RescaleSlope + ds.RescaleIntercept\n                \n                images.append(image)\n        except Exception as e:\n            print(f\"Error reading image {idx}: {e}\")\n            continue\n    \n    if not images:\n        return features\n    \n    # Analyze image features\n    try:\n        # Analyze individual slices\n        vessel_features = []\n        for img in images[:5]:  # Analyze up to 5 slices\n            vessel_feat = detect_vessel_like_structures(img)\n            vessel_features.append(vessel_feat)\n        \n        # Aggregate vessel features\n        features['avg_vessel_probability'] = np.mean([f['vessel_probability'] for f in vessel_features])\n        features['avg_edge_strength'] = np.mean([f['edge_strength'] for f in vessel_features])\n        features['avg_high_intensity_regions'] = np.mean([f['high_intensity_regions'] for f in vessel_features])\n        features['avg_texture_variance'] = np.mean([f['texture_variance'] for f in vessel_features])\n        \n        # 3D connectivity analysis\n        connectivity_features = analyze_slice_connectivity(images)\n        features.update(connectivity_features)\n        \n        # Overall image statistics\n        all_intensities = np.concatenate([img.flatten() for img in images[:3]])\n        features['intensity_range'] = np.ptp(all_intensities)\n        features['intensity_std'] = np.std(all_intensities)\n        features['intensity_skew'] = float(np.mean([(np.mean(img) - np.median(img)) / (np.std(img) + 1e-8) for img in images]))\n        \n        # Series-level features\n        features['series_uniformity'] = calculate_series_uniformity(images)\n        features['potential_aneurysm_indicators'] = calculate_aneurysm_indicators(images, features)\n        \n    except Exception as e:\n        print(f\"Error in image analysis: {e}\")\n        # Add default image features\n        features.update(get_default_image_features())\n    \n    return features\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T11:55:40.534458Z","iopub.execute_input":"2025-09-15T11:55:40.534788Z","iopub.status.idle":"2025-09-15T11:55:40.556732Z","shell.execute_reply.started":"2025-09-15T11:55:40.534756Z","shell.execute_reply":"2025-09-15T11:55:40.555264Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def extract_metadata_features(ds, num_slices):\n    \"\"\"Extract features from DICOM metadata\"\"\"\n    features = {\n        'num_slices': num_slices,\n        'modality': getattr(ds, 'Modality', 'UNKNOWN'),\n        'slice_thickness': float(getattr(ds, 'SliceThickness', 1.0)),\n        'pixel_spacing': getattr(ds, 'PixelSpacing', [1.0, 1.0]),\n        'age': extract_age(ds),\n        'sex': getattr(ds, 'PatientSex', 'U'),\n        'rows': int(getattr(ds, 'Rows', 512)),\n        'columns': int(getattr(ds, 'Columns', 512)),\n        'bits_allocated': int(getattr(ds, 'BitsAllocated', 16)),\n        'spacing_between_slices': float(getattr(ds, 'SpacingBetweenSlices', 1.0)),\n    }\n    \n    # Convert pixel spacing to float\n    if hasattr(features['pixel_spacing'], '__len__'):\n        features['pixel_spacing'] = [float(x) for x in features['pixel_spacing']]\n        features['pixel_area'] = features['pixel_spacing'][0] * features['pixel_spacing'][1]\n    else:\n        features['pixel_area'] = 1.0\n    \n    return features\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T11:55:52.558857Z","iopub.execute_input":"2025-09-15T11:55:52.559237Z","iopub.status.idle":"2025-09-15T11:55:52.567989Z","shell.execute_reply.started":"2025-09-15T11:55:52.559209Z","shell.execute_reply":"2025-09-15T11:55:52.566308Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def extract_age(ds):\n    \"\"\"Extract age from patient age field\"\"\"\n    try:\n        patient_age = getattr(ds, 'PatientAge', '050Y')\n        if isinstance(patient_age, str):\n            age_str = patient_age.replace('Y', '').replace('M', '').replace('D', '')\n            return int(age_str) if age_str.isdigit() else 50\n        return int(patient_age)\n    except:\n        return 50\n\ndef calculate_series_uniformity(images):\n    \"\"\"Calculate how uniform the series is\"\"\"\n    try:\n        if len(images) < 2:\n            return 0.5\n        \n        # Calculate mean intensity for each slice\n        means = [np.mean(img) for img in images]\n        uniformity = 1.0 - (np.std(means) / (np.mean(means) + 1e-8))\n        return max(0, min(1, uniformity))\n    except:\n        return 0.5","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T11:56:02.187121Z","iopub.execute_input":"2025-09-15T11:56:02.187419Z","iopub.status.idle":"2025-09-15T11:56:02.1957Z","shell.execute_reply.started":"2025-09-15T11:56:02.187399Z","shell.execute_reply":"2025-09-15T11:56:02.194298Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def calculate_aneurysm_indicators(images, features):\n    \"\"\"Calculate features that might indicate aneurysms\"\"\"\n    try:\n        indicators = 0.0\n        \n        # High vessel probability\n        if features.get('avg_vessel_probability', 0) > 0.3:\n            indicators += 0.2\n        \n        # Strong edge features (could indicate vessel walls)\n        if features.get('avg_edge_strength', 0) > 0.2:\n            indicators += 0.15\n        \n        # High intensity regions (contrast enhancement)\n        if features.get('avg_high_intensity_regions', 0) > 0.1:\n            indicators += 0.25\n        \n        # Good 3D connectivity (suggests vascular structures)\n        if features.get('connectivity_score', 0) > 0.6:\n            indicators += 0.2\n        \n        # Texture variance (irregular vessel walls)\n        if features.get('avg_texture_variance', 0) > 0.05:\n            indicators += 0.1\n        \n        # Modality-specific bonuses\n        if features.get('modality') in ['CTA', 'MRA']:\n            indicators += 0.1\n        \n        return min(1.0, indicators)\n    except:\n        return 0.2\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T11:56:20.014058Z","iopub.execute_input":"2025-09-15T11:56:20.014382Z","iopub.status.idle":"2025-09-15T11:56:20.022135Z","shell.execute_reply.started":"2025-09-15T11:56:20.014362Z","shell.execute_reply":"2025-09-15T11:56:20.020989Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_default_features():\n    \"\"\"Return default features when analysis fails\"\"\"\n    return {\n        'num_slices': 50, 'modality': 'UNKNOWN', 'slice_thickness': 1.0,\n        'pixel_spacing': [1.0, 1.0], 'age': 50, 'sex': 'U', 'rows': 512,\n        'columns': 512, 'bits_allocated': 16, 'spacing_between_slices': 1.0,\n        'pixel_area': 1.0, **get_default_image_features()\n    }\n\ndef get_default_image_features():\n    \"\"\"Return default image features\"\"\"\n    return {\n        'avg_vessel_probability': 0.1, 'avg_edge_strength': 0.1,\n        'avg_high_intensity_regions': 0.1, 'avg_texture_variance': 0.1,\n        'connectivity_score': 0.1, 'volume_consistency': 0.1,\n        'intensity_range': 1000, 'intensity_std': 100, 'intensity_skew': 0.0,\n        'series_uniformity': 0.5, 'potential_aneurysm_indicators': 0.2\n    }","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T11:56:30.502707Z","iopub.execute_input":"2025-09-15T11:56:30.503043Z","iopub.status.idle":"2025-09-15T11:56:30.510613Z","shell.execute_reply.started":"2025-09-15T11:56:30.50302Z","shell.execute_reply":"2025-09-15T11:56:30.509311Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def enhanced_prediction_model(features: dict) -> dict:\n    \"\"\"Enhanced prediction model using comprehensive features.\"\"\"\n    \n    # Location-specific base probabilities (updated based on medical literature)\n    location_probs = {\n        'Left Infraclinoid Internal Carotid Artery': 0.015,\n        'Right Infraclinoid Internal Carotid Artery': 0.015,\n        'Left Supraclinoid Internal Carotid Artery': 0.025,\n        'Right Supraclinoid Internal Carotid Artery': 0.025,\n        'Left Middle Cerebral Artery': 0.12,  # Higher - MCA bifurcation common\n        'Right Middle Cerebral Artery': 0.12,\n        'Anterior Communicating Artery': 0.18,  # Highest - AComm most common\n        'Left Anterior Cerebral Artery': 0.015,\n        'Right Anterior Cerebral Artery': 0.015,\n        'Left Posterior Communicating Artery': 0.08,  # PComm relatively common\n        'Right Posterior Communicating Artery': 0.08,\n        'Basilar Tip': 0.06,  # Posterior circulation\n        'Other Posterior Circulation': 0.03,\n    }\n    \n    # Extract key features\n    age = features.get('age', 50)\n    sex = features.get('sex', 'U')\n    modality = features.get('modality', 'UNKNOWN')\n    num_slices = features.get('num_slices', 50)\n    vessel_prob = features.get('avg_vessel_probability', 0.1)\n    aneurysm_indicators = features.get('potential_aneurysm_indicators', 0.2)\n    edge_strength = features.get('avg_edge_strength', 0.1)\n    connectivity = features.get('connectivity_score', 0.1)\n    \n    # Age-based risk factor (aneurysms more common with age)\n    age_factor = calculate_age_factor(age)\n    \n    # Sex-based risk factor\n    sex_factor = 1.3 if sex == 'F' else 1.0  # Slightly higher risk in women\n    \n    # Modality effectiveness factor\n    modality_factor = calculate_modality_factor(modality)\n    \n    # Image quality and completeness factor\n    series_quality_factor = calculate_series_quality_factor(features)\n    \n    # Advanced image features factor\n    image_analysis_factor = calculate_image_analysis_factor(\n        vessel_prob, aneurysm_indicators, edge_strength, connectivity\n    )\n    \n    # Location-specific adjustments\n    location_adjustments = calculate_location_adjustments(features)\n    \n    # Calculate final probabilities\n    final_probs = {}\n    for location, base_prob in location_probs.items():\n        # Apply all factors\n        adjusted_prob = (base_prob * \n                        age_factor * \n                        sex_factor * \n                        modality_factor * \n                        series_quality_factor * \n                        image_analysis_factor *\n                        location_adjustments.get(location, 1.0))\n        \n        # Add controlled randomness to avoid identical predictions\n        noise = np.random.normal(0, 0.005)  # Reduced noise\n        adjusted_prob = max(0.001, min(0.99, adjusted_prob + noise))\n        \n        final_probs[location] = adjusted_prob\n    \n    # Calculate overall aneurysm presence probability\n    prob_no_aneurysm = 1.0\n    for prob in final_probs.values():\n        prob_no_aneurysm *= (1 - prob)\n    \n    final_probs['Aneurysm Present'] = 1 - prob_no_aneurysm\n    \n    return final_probs","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T11:56:48.386045Z","iopub.execute_input":"2025-09-15T11:56:48.386681Z","iopub.status.idle":"2025-09-15T11:56:48.400886Z","shell.execute_reply.started":"2025-09-15T11:56:48.386631Z","shell.execute_reply":"2025-09-15T11:56:48.399068Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def calculate_age_factor(age):\n    \"\"\"Calculate age-based risk multiplier\"\"\"\n    if age < 30:\n        return 0.2\n    elif age < 40:\n        return 0.5\n    elif age < 50:\n        return 0.8\n    elif age < 60:\n        return 1.2\n    elif age < 70:\n        return 1.8\n    else:\n        return 2.5\n\ndef calculate_modality_factor(modality):\n    \"\"\"Calculate modality-based detection capability\"\"\"\n    modality_factors = {\n        'CTA': 1.8,  # CT Angiography - excellent for aneurysms\n        'MRA': 1.6,  # MR Angiography - very good\n        'TOF': 1.5,  # Time of Flight MRA\n        'MR': 0.9,   # Regular MR\n        'MRI': 0.9,  # Regular MRI\n        'CT': 0.7,   # Non-contrast CT\n        'UNKNOWN': 1.0\n    }\n    return modality_factors.get(modality, 1.0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T11:57:02.877536Z","iopub.execute_input":"2025-09-15T11:57:02.877896Z","iopub.status.idle":"2025-09-15T11:57:02.885512Z","shell.execute_reply.started":"2025-09-15T11:57:02.877871Z","shell.execute_reply":"2025-09-15T11:57:02.884414Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def calculate_series_quality_factor(features):\n    \"\"\"Calculate factor based on series quality and completeness\"\"\"\n    factor = 1.0\n    \n    # Series completeness\n    num_slices = features.get('num_slices', 50)\n    if num_slices < 20:\n        factor *= 0.7\n    elif num_slices > 100:\n        factor *= 1.3\n    \n    # Resolution quality\n    pixel_area = features.get('pixel_area', 1.0)\n    if pixel_area < 0.5:  # High resolution\n        factor *= 1.2\n    elif pixel_area > 2.0:  # Low resolution\n        factor *= 0.8\n    \n    # Slice thickness\n    slice_thickness = features.get('slice_thickness', 1.0)\n    if slice_thickness <= 1.0:  # Thin slices\n        factor *= 1.3\n    elif slice_thickness > 3.0:  # Thick slices\n        factor *= 0.8\n    \n    return factor\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T11:57:13.220486Z","iopub.execute_input":"2025-09-15T11:57:13.220798Z","iopub.status.idle":"2025-09-15T11:57:13.227783Z","shell.execute_reply.started":"2025-09-15T11:57:13.220778Z","shell.execute_reply":"2025-09-15T11:57:13.226655Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def calculate_image_analysis_factor(vessel_prob, aneurysm_indicators, edge_strength, connectivity):\n    \"\"\"Calculate factor based on image analysis results\"\"\"\n    base_factor = 1.0\n    \n    # Vessel probability contribution\n    vessel_contribution = min(2.0, max(0.3, vessel_prob * 3))\n    \n    # Aneurysm indicators contribution\n    indicator_contribution = min(2.5, max(0.2, aneurysm_indicators * 4))\n    \n    # Edge strength contribution (moderate weight)\n    edge_contribution = min(1.5, max(0.5, 1 + edge_strength))\n    \n    # Connectivity contribution\n    connectivity_contribution = min(1.8, max(0.4, connectivity * 2))\n    \n    # Combine factors with appropriate weights\n    combined_factor = (\n        vessel_contribution * 0.3 +\n        indicator_contribution * 0.4 +\n        edge_contribution * 0.2 +\n        connectivity_contribution * 0.1\n    )\n    \n    return max(0.1, min(3.0, combined_factor))\n\ndef calculate_location_adjustments(features):\n    \"\"\"Calculate location-specific adjustments based on imaging characteristics\"\"\"\n    adjustments = {}\n    \n    # Base adjustments for all locations\n    base_adjustment = 1.0\n    \n    # Modality-specific location preferences\n    modality = features.get('modality', 'UNKNOWN')\n    \n    if modality in ['CTA', 'MRA']:\n        # These modalities are better for certain locations\n        adjustments.update({\n            'Anterior Communicating Artery': 1.4,  # AComm well visualized\n            'Left Middle Cerebral Artery': 1.3,\n            'Right Middle Cerebral Artery': 1.3,\n            'Basilar Tip': 1.3,  # Good posterior circulation visualization\n        })\n    \n    # Age-based location preferences\n    age = features.get('age', 50)\n    if age > 60:\n        # Certain locations more common in older patients\n        adjustments.update({\n            'Left Posterior Communicating Artery': adjustments.get('Left Posterior Communicating Artery', 1.0) * 1.2,\n            'Right Posterior Communicating Artery': adjustments.get('Right Posterior Communicating Artery', 1.0) * 1.2,\n            'Basilar Tip': adjustments.get('Basilar Tip', 1.0) * 1.3,\n        })\n    \n    # Fill in missing locations with base adjustment\n    for location in ['Left Infraclinoid Internal Carotid Artery', 'Right Infraclinoid Internal Carotid Artery',\n                    'Left Supraclinoid Internal Carotid Artery', 'Right Supraclinoid Internal Carotid Artery',\n                    'Left Middle Cerebral Artery', 'Right Middle Cerebral Artery',\n                    'Anterior Communicating Artery', 'Left Anterior Cerebral Artery',\n                    'Right Anterior Cerebral Artery', 'Left Posterior Communicating Artery',\n                    'Right Posterior Communicating Artery', 'Basilar Tip', 'Other Posterior Circulation']:\n        if location not in adjustments:\n            adjustments[location] = base_adjustment\n    \n    return adjustments\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T11:57:25.79536Z","iopub.execute_input":"2025-09-15T11:57:25.795739Z","iopub.status.idle":"2025-09-15T11:57:25.806033Z","shell.execute_reply.started":"2025-09-15T11:57:25.795715Z","shell.execute_reply":"2025-09-15T11:57:25.804843Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def predict(series_path: str) -> pl.DataFrame | pd.DataFrame:\n    \"\"\"Make prediction using enhanced image analysis and machine learning approach.\"\"\"\n    \n    series_id = os.path.basename(series_path)\n    \n    try:\n        # Extract comprehensive features from the DICOM series\n        features = extract_advanced_features_from_series(series_path)\n        \n        # Get enhanced predictions\n        predictions_dict = enhanced_prediction_model(features)\n        \n        # Create prediction row\n        prediction_row = [series_id]\n        for col in LABEL_COLS:\n            prediction_row.append(predictions_dict.get(col, 0.1))\n        \n        predictions = pl.DataFrame(\n            data=[prediction_row],\n            schema=[ID_COL] + LABEL_COLS,\n            orient='row',\n        )\n        \n    except Exception as e:\n        print(f\"Error in prediction for {series_id}: {e}\")\n        # Enhanced fallback predictions based on series_id patterns\n        predictions = create_fallback_predictions(series_id)\n    \n    # Validation\n    if isinstance(predictions, pl.DataFrame):\n        assert predictions.columns == [ID_COL] + LABEL_COLS\n    elif isinstance(predictions, pd.DataFrame):\n        assert (predictions.columns == [ID_COL] + LABEL_COLS).all()\n    else:\n        raise TypeError('The predict function must return a DataFrame')\n    \n    # IMPORTANT: Required cleanup to prevent disk space issues\n    shutil.rmtree('/kaggle/shared', ignore_errors=True)\n    \n    # Return predictions without the ID column as required\n    return predictions.drop(ID_COL)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T11:57:44.382346Z","iopub.execute_input":"2025-09-15T11:57:44.382685Z","iopub.status.idle":"2025-09-15T11:57:44.39152Z","shell.execute_reply.started":"2025-09-15T11:57:44.382663Z","shell.execute_reply":"2025-09-15T11:57:44.390048Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_fallback_predictions(series_id):\n    \"\"\"Create fallback predictions with some intelligence based on series ID\"\"\"\n    try:\n        # Extract some information from series ID if possible\n        # This is a backup when all else fails\n        base_probs = [0.08, 0.08, 0.12, 0.12, 0.15, 0.15, 0.20, 0.05, 0.05, 0.10, 0.10, 0.08, 0.04]  # Adjusted probabilities\n        \n        # Add controlled variation based on series_id hash\n        hash_val = hash(series_id) % 1000\n        variation = (hash_val / 1000 - 0.5) * 0.02  # Small variation\n        \n        adjusted_probs = [max(0.001, min(0.99, p + variation + np.random.normal(0, 0.005))) \n                         for p in base_probs]\n        \n        # Calculate aneurysm present probability\n        prob_no_aneurysm = 1.0\n        for prob in adjusted_probs:\n            prob_no_aneurysm *= (1 - prob)\n        adjusted_probs.append(1 - prob_no_aneurysm)\n        \n        predictions = pl.DataFrame(\n            data=[[series_id] + adjusted_probs],\n            schema=[ID_COL] + LABEL_COLS,\n            orient='row',\n        )\n        return predictions\n        \n    except:\n        # Ultimate fallback\n        return pl.DataFrame(\n            data=[[series_id] + [0.1] * len(LABEL_COLS)],\n            schema=[ID_COL] + LABEL_COLS,\n            orient='row',\n        )\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T11:58:01.486426Z","iopub.execute_input":"2025-09-15T11:58:01.48674Z","iopub.status.idle":"2025-09-15T11:58:01.495754Z","shell.execute_reply.started":"2025-09-15T11:58:01.486717Z","shell.execute_reply":"2025-09-15T11:58:01.494557Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Clean up shared directory before starting inference server\nshutil.rmtree('/kaggle/shared', ignore_errors=True)\n# Initialize and run the inference server\ninference_server = kaggle_evaluation.rsna_inference_server.RSNAInferenceServer(predict)\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway()\n    display(pl.read_parquet('/kaggle/working/submission.parquet'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T11:58:15.478345Z","iopub.execute_input":"2025-09-15T11:58:15.478891Z","iopub.status.idle":"2025-09-15T11:58:20.299959Z","shell.execute_reply.started":"2025-09-15T11:58:15.478843Z","shell.execute_reply":"2025-09-15T11:58:20.298857Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}