{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":36363,"databundleVersionId":4050810,"isSourceIdPinned":false}],"dockerImageVersionId":31328,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-03-23T08:19:02.476071Z","iopub.execute_input":"2026-03-23T08:19:02.47686Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%pip install timm albumentations pydicom nibabel scikit-learn \\\n             matplotlib pandas numpy torch torchvision \\\n             wandb --quiet","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport timm\nimport pydicom\nimport nibabel\nimport albumentations\n \nprint(f\"PyTorch  : {torch.__version__}\")\nprint(f\"CUDA     : {torch.cuda.is_available()}\")\nprint(f\"timm     : {timm.__version__}\")\nprint(f\"pydicom  : {pydicom.__version__}\")\n \nif torch.cuda.is_available():\n    print(f\"GPU      : {torch.cuda.get_device_name(0)}\")\n    print(f\"VRAM     : {torch.cuda.get_device_properties(0).total_memory/1e9:.1f} GB\")\n \n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"SRC_DIR = \"/kaggle/input/competitions/rsna-2022-cervical-spine-fracture-detection/src\"   # adjust if needed\nsys.path.insert(0, SRC_DIR)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from eda import run_eda\nfrom pathlib import Path\n\neda_df = run_eda(\n    csv_path   = Path(\"/kaggle/input/rsna-2022-cervical-spine-fracture-detection/train.csv\"),\n    output_dir = Path(\"/kaggle/working/outputs\"),\n)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from config import CFG\n\n# Override defaults for Kaggle (smaller batch for T4 16 GB)\nCFG.device            = \"cuda\"\nCFG.amp               = True\nCFG.num_workers       = 4\n\n# Stage 1\nCFG.stage1.backbone   = \"efficientnet_b4\"  # change to _b5/_b6 for higher accuracy\nCFG.stage1.epochs     = 20\nCFG.stage1.batch_size = 8\nCFG.stage1.n_folds    = 5\n\n# Stage 2\nCFG.stage2.backbone   = \"efficientnet_b3\"\nCFG.stage2.epochs     = 20\nCFG.stage2.batch_size = 16\n\n# Cache preprocessed volumes (avoids re-reading DICOMs every epoch)\nCACHE_DIR = Path(\"/kaggle/working/volume_cache\")\n\nprint(\"Configuration ready.\")\nprint(f\"Stage 1: {CFG.stage1.backbone} × {CFG.stage1.epochs} epochs × {CFG.stage1.n_folds} folds\")\nprint(f\"Stage 2: {CFG.stage2.backbone} × {CFG.stage2.epochs} epochs\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from main import load_and_validate_data, create_folds, run_stage1\n\ndf = load_and_validate_data(CFG)\ndf = create_folds(df, n_folds=CFG.stage1.n_folds, seed=CFG.stage1.seed)\n\n# Train all 5 folds (comment out folds=[0] to run a single fold for speed)\ns1_results = run_stage1(df, CFG, folds=[0, 1, 2, 3, 4], cache_dir=CACHE_DIR)\n\nprint(f\"\\nStage 1 OOF AUC : {s1_results['oof_metrics']['auc']:.4f}\")\nprint(f\"Stage 1 OOF F1  : {s1_results['oof_metrics']['f1']:.4f}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from main import run_stage2\n\ns2_results = run_stage2(df, CFG, folds=[0, 1, 2, 3, 4], cache_dir=CACHE_DIR)\n\nprint(f\"\\nStage 2 OOF AUC : {s2_results['oof_metrics']['auc']:.4f}\")\nprint(f\"Stage 2 C2 Sens : {s2_results['oof_metrics']['sensitivity']:.4f}\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from metrics import print_final_report\n\npipeline_metrics = evaluate_pipeline(df, s1_results, s2_results, CFG)\n\nprint_final_report(\n    stage1_metrics   = s1_results[\"oof_metrics\"],\n    stage2_metrics   = s2_results[\"oof_metrics\"],\n    pipeline_metrics = pipeline_metrics,\n    output_path      = CFG.paths.output_dir / \"final_report.json\",\n)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nfrom IPython.display import Image, display\n\noutput_dir = Path(\"/kaggle/working/outputs\")\n\n# Show EDA report\nif (output_dir / \"eda_report.png\").exists():\n    display(Image(str(output_dir / \"eda_report.png\")))\n\n# Show training curves\nfor png in sorted(output_dir.glob(\"*curves.png\")):\n    display(Image(str(png)))\n\n# Show confusion matrices\nfor png in sorted(output_dir.glob(\"*confusion.png\")):\n    display(Image(str(png)))\n\n# Print final JSON report\nimport json\nreport_path = output_dir / \"final_report.json\"\nif report_path.exists():\n    with open(report_path) as f:\n        report = json.load(f)\n    print(json.dumps(report, indent=2))","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}