"""
Auto-generated Kaggle Notebook for competition: rsna-knee-abnormality-detection
Target column(s): ['acl_tear', 'meniscal_tear', 'cartilage_defect']
Generated by kaggle/src/build_notebook.py — do not edit by hand.
"""

# --- Cell 1: Environment detection and imports -----------------------------
import json
import os
import sys
from pathlib import Path

import numpy as np
import pandas as pd

SLUG = "rsna-knee-abnormality-detection"
INPUT_DIR = Path("/kaggle/input") / SLUG
WORKING_DIR = Path("/kaggle/working")
WORKING_DIR.mkdir(parents=True, exist_ok=True)

# Competition-specific modules are expected to be present as notebook inputs
# or copied into the same directory. Add the module directory to sys.path so
# import works both on Kaggle and in a local test run.
MODULE_DIR = Path(__file__).resolve().parent
sys.path.insert(0, str(MODULE_DIR))

CONFIG = {
  "competition_slug": "rsna-knee-abnormality-detection",
  "username": "adrianstanica",
  "target_cols": [
    "acl_tear",
    "meniscal_tear",
    "cartilage_defect"
  ],
  "id_col": "series_instance_uid",
  "execution_mode": "kaggle_notebook",
  "task": "multilabel_classification",
  "modality": "dicom_mri",
  "kaggle_notebook_path": "notebook.py"
}

# --- Cell 2: Load raw data --------------------------------------------------
def load_raw_data(config: dict):
    train_csv = INPUT_DIR / "train.csv"
    test_csv = INPUT_DIR / "test.csv"
    if not train_csv.exists():
        raise FileNotFoundError(f"train.csv not found at {train_csv}")
    if not test_csv.exists():
        raise FileNotFoundError(f"test.csv not found at {test_csv}")
    return pd.read_csv(train_csv), pd.read_csv(test_csv)


# --- Cell: features.py (competition-specific module) ---
features_py_source = r'''
"""Feature engineering for RSNA knee abnormality detection.

This module runs inside a Kaggle Notebook. It reads DICOM metadata and
precomputed CSVs from /kaggle/input/rsna-knee-abnormality-detection/ and
writes processed features to /kaggle/working/.

Because the competition does not provide public training labels, this baseline
uses only test-time metadata: the list of study/series UIDs and file paths.
"""

from pathlib import Path

import pandas as pd

SLUG = "rsna-knee-abnormality-detection"
INPUT_DIR = Path("/kaggle/input") / SLUG
WORKING_DIR = Path("/kaggle/working")


def _find_csv_candidates() -> list[Path]:
    """Return any CSV files present in the input directory."""
    if not INPUT_DIR.exists():
        return []
    return sorted(p for p in INPUT_DIR.rglob("*.csv") if p.is_file())


def _find_dicom_dirs() -> list[Path]:
    """Return directories that contain .dcm files."""
    if not INPUT_DIR.exists():
        return []
    return sorted({p.parent for p in INPUT_DIR.rglob("*.dcm")})


def engineer_features(config: dict) -> None:
    """Build a minimal feature table from available input files."""
    WORKING_DIR.mkdir(parents=True, exist_ok=True)

    csv_files = _find_csv_candidates()
    dicom_dirs = _find_dicom_dirs()

    rows = []
    for csv_path in csv_files:
        rows.append(
            {
                "source": str(csv_path.relative_to(INPUT_DIR)),
                "kind": "csv",
                "basename": csv_path.name,
                "size_bytes": csv_path.stat().st_size if csv_path.exists() else 0,
            }
        )

    for dicom_dir in dicom_dirs:
        dcm_count = len(list(dicom_dir.glob("*.dcm")))
        rows.append(
            {
                "source": str(dicom_dir.relative_to(INPUT_DIR)),
                "kind": "dicom_series",
                "basename": dicom_dir.name,
                "size_bytes": dcm_count,
            }
        )

    meta_df = pd.DataFrame(rows)
    meta_path = WORKING_DIR / "input_metadata.csv"
    meta_df.to_csv(meta_path, index=False)
    print(f"Wrote input metadata: {meta_path} ({len(meta_df)} rows)")
'''
exec(compile(features_py_source, 'features.py', 'exec'))

# --- Cell: train.py (competition-specific module) ---
train_py_source = r'''
"""Training stub for RSNA knee abnormality detection.

The competition does not provide public training labels, so there is nothing
to train. This module exists only to satisfy the notebook interface.
"""

from pathlib import Path

WORKING_DIR = Path("/kaggle/working")


def train_model(config: dict) -> Path:
    """Return a sentinel model path since no training data is available."""
    model_path = WORKING_DIR / "no_op_model.txt"
    model_path.write_text("No public training data available. Prediction uses a baseline.\n")
    print(f"No-op model path: {model_path}")
    return model_path
'''
exec(compile(train_py_source, 'train.py', 'exec'))

# --- Cell: predict.py (competition-specific module) ---
predict_py_source = r'''
"""Prediction stub for RSNA knee abnormality detection.

The competition does not provide public training data, so this baseline
generates a submission from sample_submission.csv if it exists. If not, it
raises a clear error explaining that the competition input is missing.
"""

from pathlib import Path

import pandas as pd

SLUG = "rsna-knee-abnormality-detection"
INPUT_DIR = Path("/kaggle/input") / SLUG
WORKING_DIR = Path("/kaggle/working")


def predict(config: dict, model_path: Path) -> Path:
    """Create submission.csv from sample_submission.csv or fail clearly."""
    WORKING_DIR.mkdir(parents=True, exist_ok=True)
    submission_path = WORKING_DIR / "submission.csv"

    sample_path = INPUT_DIR / "sample_submission.csv"
    if sample_path.exists():
        sub_df = pd.read_csv(sample_path)
        print(f"Loaded sample submission: {sub_df.shape}")
    else:
        # Fallback: build a minimal submission frame from any CSV with an id column.
        csv_files = sorted(p for p in INPUT_DIR.rglob("*.csv") if p.is_file())
        test_csvs = [p for p in csv_files if "test" in p.name.lower() and p.name.lower() != "sample_submission.csv"]
        if not test_csvs:
            raise FileNotFoundError(
                f"No test CSV or sample_submission.csv found in {INPUT_DIR}. "
                "Ensure the competition dataset is attached and the RSNA rules are accepted."
            )
        test_path = test_csvs[0]
        sub_df = pd.read_csv(test_path)
        id_col = config.get("id_col", sub_df.columns[0])
        if id_col not in sub_df.columns:
            id_col = sub_df.columns[0]
        sub_df = sub_df[[id_col]].copy()
        # No predictions possible without training labels; fill with 0.5 baseline.
        target_cols = config.get("target_cols", [])
        for col in target_cols:
            sub_df[col] = 0.5
        print(f"Built fallback submission from {test_path}: {sub_df.shape}")

    sub_df.to_csv(submission_path, index=False)
    print(f"Wrote submission: {submission_path}")
    return submission_path
'''
exec(compile(predict_py_source, 'predict.py', 'exec'))

# --- Cell: Main runner -----------------------------------------------------
def main():
    print(f"Input directory: {INPUT_DIR}")
    print(f"Working directory: {WORKING_DIR}")

    if INPUT_DIR.exists():
        print("Input directory contents:")
        for p in sorted(INPUT_DIR.iterdir()):
            print(f"  {p.name} {'(dir)' if p.is_dir() else p.stat().st_size}")
    else:
        print("WARNING: Input directory does not exist.")

    train_csv = INPUT_DIR / "train.csv"
    test_csv = INPUT_DIR / "test.csv"
    if train_csv.exists() and test_csv.exists():
        train_df, test_df = load_raw_data(CONFIG)
        print(f"Loaded train ({train_df.shape}) and test ({test_df.shape})")
    else:
        print("No train.csv/test.csv found; running test-only flow.")

    # Feature engineering (writes processed CSVs to working dir)
    engineer_features(CONFIG)

    # Training (writes model artifact to working dir)
    model_path = train_model(CONFIG)
    print(f"Model artifact: {model_path}")

    # Prediction / submission
    submission_path = predict(CONFIG, model_path)
    print(f"Submission written to: {submission_path}")
    print("Notebook run complete.")

if __name__ == "__main__":
    main()
