{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":99552,"databundleVersionId":13190393,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# ==========================================================\n# RSNA Intracranial Aneurysm Detection - Full Multi-Label Pipeline\n# Kaggle-Compatible (uses CSV + test DICOMs)\n# ==========================================================\n\n!pip install polars -q\n!pip install pydicom -q\n!pip install xgboost -q\n\nimport os\nimport shutil\nimport pydicom\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import roc_auc_score\nimport xgboost as xgb\nimport kaggle_evaluation.rsna_inference_server as rsna_server\n\n# ==========================================================\n# Step 1: Paths and Data\n# ==========================================================\ndata_path = \"/kaggle/input/rsna-intracranial-aneurysm-detection/\"\n\ntrain_csv = pd.read_csv(os.path.join(data_path, \"train.csv\"))\ntrain_localizers = pd.read_csv(os.path.join(data_path, \"train_localizers.csv\"))\n\nID_COL = \"SeriesInstanceUID\"\nLABEL_COLS = [\n    'Left Infraclinoid Internal Carotid Artery',\n    'Right Infraclinoid Internal Carotid Artery',\n    'Left Supraclinoid Internal Carotid Artery',\n    'Right Supraclinoid Internal Carotid Artery',\n    'Left Middle Cerebral Artery',\n    'Right Middle Cerebral Artery',\n    'Anterior Communicating Artery',\n    'Left Anterior Cerebral Artery',\n    'Right Anterior Cerebral Artery',\n    'Left Posterior Communicating Artery',\n    'Right Posterior Communicating Artery',\n    'Basilar Tip',\n    'Other Posterior Circulation',\n    'Aneurysm Present',\n]\n\n# ==========================================================\n# Step 2: Feature Engineering\n# ==========================================================\n# Feature: number of localizers per series\ntrain_features = (\n    train_localizers.groupby(ID_COL)\n    .size()\n    .reset_index(name=\"num_localizers\")\n)\n\ntrain_df = train_csv.merge(train_features, on=ID_COL, how=\"left\")\ntrain_df[\"num_localizers\"].fillna(0, inplace=True)\n\n# You can add more metadata features here\nX = train_df[[\"num_localizers\"]].values\ny = train_df[LABEL_COLS].values\n\n# ==========================================================\n# Step 3: Train Multi-Label Models\n# ==========================================================\nmodels = {}\nval_scores = {}\n\nfor i, col in enumerate(LABEL_COLS):\n    y_col = y[:, i]\n    X_train, X_val, y_train, y_val = train_test_split(\n        X, y_col, test_size=0.2, random_state=42\n    )\n\n    clf = xgb.XGBClassifier(\n        n_estimators=300,\n        max_depth=3,\n        learning_rate=0.05,\n        subsample=0.8,\n        colsample_bytree=0.8,\n        eval_metric=\"logloss\",\n        random_state=42\n    )\n    clf.fit(X_train, y_train)\n    models[col] = clf\n\n    # Validation AUC\n    try:\n        val_preds = clf.predict_proba(X_val)[:, 1]\n        auc = roc_auc_score(y_val, val_preds)\n        val_scores[col] = auc\n    except:\n        val_scores[col] = None\n\nprint(\"Validation AUCs:\")\nfor col, auc in val_scores.items():\n    print(f\"{col}: {auc}\")\n\n# ==========================================================\n# Step 4: Prediction Function for Kaggle\n# ==========================================================\ndef predict(series_path: str):\n    \"\"\"\n    Predict function for RSNA Aneurysm Detection.\n    Args:\n        series_path (str): Path to series folder.\n    Returns:\n        pd.DataFrame: Predictions for all 14 labels.\n    \"\"\"\n    series_id = os.path.basename(series_path)\n\n    # Feature: number of DICOM slices in test folder\n    try:\n        num_files = len(os.listdir(series_path))\n    except:\n        num_files = 0\n\n    X_test = np.array([[num_files]])\n\n    preds = {}\n    for col in LABEL_COLS:\n        if col in models:\n            preds[col] = float(models[col].predict_proba(X_test)[:, 1][0])\n        else:\n            preds[col] = 0.0\n\n    df = pd.DataFrame([[series_id, *preds.values()]], columns=[ID_COL, *LABEL_COLS])\n    return df.drop(columns=[ID_COL])\n\n# ==========================================================\n# Step 5: Inference Server (Kaggle)\n# ==========================================================\ninference_server = rsna_server.RSNAInferenceServer(predict)\n\nif os.getenv(\"KAGGLE_IS_COMPETITION_RERUN\"):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway()\n    display(pl.read_parquet(\"/kaggle/working/submission.parquet\").head())\n\n# Clean up\nshutil.rmtree(\"/kaggle/shared\", ignore_errors=True)\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-08-20T09:24:08.085942Z","iopub.execute_input":"2025-08-20T09:24:08.087151Z","iopub.status.idle":"2025-08-20T09:24:26.049415Z","shell.execute_reply.started":"2025-08-20T09:24:08.087099Z","shell.execute_reply":"2025-08-20T09:24:26.048336Z"}},"outputs":[],"execution_count":null}]}