{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# NeurIPS Ariel Data Challenge 2025 — Stage 1: Ingest & CV Design\n**Track A (Classical Physics-Informed ML)**  \n**Scope:** Ingest metadata, verify contract shapes, extract ADC parameters, inspect `axis_info`, confirm 1:1 star-planet mapping, assign spectral classes (M/K/G/F), freeze 5-fold `StratifiedGroupKFold` CV with a 10% holdout vault, verify metric self-test, inspect label bounds, and confirm raw sensor cube dimensions on one target.","metadata":{}},{"cell_type":"code","source":"VERSION = \"1.1.0\"\nSEED = 42\n\nimport os\nimport sys\nimport hashlib\nfrom pathlib import Path\nimport numpy as np\nimport pandas as pd\nimport pyarrow as pa\nimport pyarrow.parquet as pq\nfrom sklearn.model_selection import StratifiedGroupKFold\n\nprint(f\"Notebook Version: {VERSION} | Seed: {SEED}\")\nprint(f\"Python:   {sys.version.split()[0]}\")\nprint(f\"NumPy:    {np.__version__}\")\nprint(f\"Pandas:   {pd.__version__}\")\nprint(f\"PyArrow:  {pa.__version__}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-29T17:33:36.050552Z","iopub.execute_input":"2026-09-29T17:33:36.050838Z","iopub.status.idle":"2026-09-29T17:33:36.060004Z","shell.execute_reply.started":"2026-09-29T17:33:36.050819Z","shell.execute_reply":"2026-09-29T17:33:36.05798Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def resolve_data_paths():\n    candidates = [\n        Path(\"/kaggle/input/competitions/ariel-data-challenge-2025\"),\n        Path(\"/kaggle/input/ariel-data-challenge-2025\"),\n        Path(\"./data/raw\"),\n        Path(\"../data/raw\")\n    ]\n    for cand in candidates:\n        if cand.exists() and (cand / \"train_star_info.csv\").exists():\n            return cand\n    for root in [Path(\"/kaggle/input\"), Path(\".\")]:\n        if root.exists():\n            for dirpath, _, filenames in os.walk(root):\n                if \"train_star_info.csv\" in filenames:\n                    return Path(dirpath)\n    raise FileNotFoundError(\"Unable to locate competition data directory containing train_star_info.csv\")\n\nDATA_DIR = resolve_data_paths()\nOUT_DIR = Path(\"/kaggle/working\") if Path(\"/kaggle/working\").exists() else Path(\"./data/cache\")\nOUT_DIR.mkdir(parents=True, exist_ok=True)\n\nprint(f\"DATA_DIR: {DATA_DIR}\")\nprint(f\"OUT_DIR:  {OUT_DIR}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-29T17:33:36.061627Z","iopub.execute_input":"2026-09-29T17:33:36.061932Z","iopub.status.idle":"2026-09-29T17:33:36.126488Z","shell.execute_reply.started":"2026-09-29T17:33:36.061905Z","shell.execute_reply":"2026-09-29T17:33:36.124622Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"star_df = pd.read_csv(DATA_DIR / \"train_star_info.csv\")\nlabels_df = pd.read_csv(DATA_DIR / \"train.csv\")\nadc_df = pd.read_csv(DATA_DIR / \"adc_info.csv\")\naxis_df = pd.read_parquet(DATA_DIR / \"axis_info.parquet\")\nsample_sub = pd.read_csv(DATA_DIR / \"sample_submission.csv\")\n\n# Contract shape assertions\nassert labels_df.shape == (1100, 284), f\"labels_df shape mismatch: expected (1100, 284), got {labels_df.shape}\"\nassert sample_sub.shape[1] == 567, f\"sample_submission.csv must have 567 columns (1 ID + 283 mu + 283 sigma), got {sample_sub.shape[1]}\"\nassert \"planet_id\" in sample_sub.columns, \"sample_submission.csv missing planet_id!\"\n\nprint(f\"train_star_info:   {star_df.shape}\")\nprint(f\"train labels:      {labels_df.shape} (asserted 1100 x 284)\")\nprint(f\"axis_info:         {axis_df.shape}\")\nprint(f\"sample_submission: {sample_sub.shape} (asserted 567 columns)\")\n\n# Strict ADC key extraction\nrequired_adc_cols = [\"FGS1_adc_gain\", \"FGS1_adc_offset\", \"AIRS-CH0_adc_gain\", \"AIRS-CH0_adc_offset\"]\nfor col in required_adc_cols:\n    assert col in adc_df.columns, f\"Required ADC column '{col}' missing from adc_info.csv!\"\n\nFGS1_ADC_GAIN = float(adc_df[\"FGS1_adc_gain\"].iloc[0])\nFGS1_ADC_OFFSET = float(adc_df[\"FGS1_adc_offset\"].iloc[0])\nAIRS_ADC_GAIN = float(adc_df[\"AIRS-CH0_adc_gain\"].iloc[0])\nAIRS_ADC_OFFSET = float(adc_df[\"AIRS-CH0_adc_offset\"].iloc[0])\n\nprint(\"\\nExtracted ADC Constants:\")\nprint(f\"  FGS1 ADC Gain:       {FGS1_ADC_GAIN:.4f} e-/ADU | Offset: {FGS1_ADC_OFFSET:.1f} ADU\")\nprint(f\"  AIRS-CH0 ADC Gain:   {AIRS_ADC_GAIN:.4f} e-/ADU | Offset: {AIRS_ADC_OFFSET:.1f} ADU\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-29T17:33:36.127667Z","iopub.execute_input":"2026-09-29T17:33:36.128263Z","iopub.status.idle":"2026-09-29T17:33:36.390379Z","shell.execute_reply.started":"2026-09-29T17:33:36.128235Z","shell.execute_reply":"2026-09-29T17:33:36.38854Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"wl_col = axis_df[\"AIRS-CH0-axis2-um\"].dropna()\nwl_min = float(wl_col.min())\nwl_max = float(wl_col.max())\nis_increasing = bool(wl_col.is_monotonic_increasing)\nis_decreasing = bool(wl_col.is_monotonic_decreasing)\n\nprint(\"=\" * 65)\nprint(\"AXIS_INFO WAVELENGTH GRID SUMMARY:\")\nprint(f\"  Raw Spectral Bins:   {len(wl_col)}\")\nprint(f\"  Wavelength Min:      {wl_min:.5f} um\")\nprint(f\"  Wavelength Max:      {wl_max:.5f} um\")\nprint(f\"  Monotonic Increasing:{is_increasing}\")\nprint(f\"  Monotonic Decreasing:{is_decreasing}\")\nprint(\"  Note: Crop window [37:319] and dispersion axis orientation are not frozen here; deferred to Stage 3.\")\nprint(\"=\" * 65)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-29T17:33:36.393413Z","iopub.execute_input":"2026-09-29T17:33:36.39376Z","iopub.status.idle":"2026-09-29T17:33:36.404894Z","shell.execute_reply.started":"2026-09-29T17:33:36.393735Z","shell.execute_reply":"2026-09-29T17:33:36.403817Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1:1 star-planet check\nn_planets = len(star_df)\nn_unique_planets = star_df[\"planet_id\"].nunique()\nn_unique_stars = star_df[[\"Ts\", \"Rs\", \"Ms\"]].drop_duplicates().shape[0]\n\nassert n_planets == n_unique_planets == n_unique_stars, \"Detected multi-planet systems or duplicate stars!\"\nprint(f\"✓ 1:1 host star to planet correspondence confirmed: {n_planets} unique systems.\")\n\nstar_df[\"star_id\"] = star_df[\"planet_id\"].astype(str)\n\n# Surface gravity log10(g) in cgs: log10(g_sun) ~ 4.4378\nstar_df[\"logg\"] = 4.4378 + np.log10(star_df[\"Ms\"] / (star_df[\"Rs\"] ** 2))\n\n# Astrophysical spectral class assignment\ndef get_spectral_class(ts):\n    if ts < 3700.0: return \"M\"\n    elif ts < 5200.0: return \"K\"\n    elif ts < 6000.0: return \"G\"\n    else: return \"F\"\n\nstar_df[\"spectral_class\"] = star_df[\"Ts\"].apply(get_spectral_class)\nprint(\"\\nSpectral class distribution:\")\nprint(star_df[\"spectral_class\"].value_counts().to_dict())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-29T17:33:36.406111Z","iopub.execute_input":"2026-09-29T17:33:36.406451Z","iopub.status.idle":"2026-09-29T17:33:36.463383Z","shell.execute_reply.started":"2026-09-29T17:33:36.406389Z","shell.execute_reply":"2026-09-29T17:33:36.461516Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1. Carve out a 10% stratified holdout vault (~112 planets)\nrng = np.random.default_rng(SEED)\nvault_indices = []\nfor s_class, group in star_df.groupby(\"spectral_class\"):\n    idx = group.index.values.copy()\n    rng.shuffle(idx)\n    n_vault = int(np.round(len(idx) * 0.102))\n    vault_indices.extend(idx[:n_vault])\n\nstar_df[\"is_vault\"] = False\nstar_df.loc[vault_indices, \"is_vault\"] = True\nnon_vault_df = star_df[~star_df[\"is_vault\"]].copy()\n\n# 2. 5-fold StratifiedGroupKFold on remaining non-vault stars\nsgkf = StratifiedGroupKFold(n_splits=5, shuffle=True, random_state=SEED)\nnon_vault_df[\"fold\"] = -1\n\nfor fold_id, (train_idx, val_idx) in enumerate(sgkf.split(\n    X=non_vault_df,\n    y=non_vault_df[\"spectral_class\"],\n    groups=non_vault_df[\"star_id\"]\n)):\n    val_planet_ids = non_vault_df.iloc[val_idx][\"planet_id\"]\n    non_vault_df.loc[non_vault_df[\"planet_id\"].isin(val_planet_ids), \"fold\"] = fold_id\n\nstar_df[\"fold\"] = -1\nstar_df.loc[non_vault_df.index, \"fold\"] = non_vault_df[\"fold\"]\n\n# 3. Zero-leakage assertion check\nfor f in range(5):\n    p_f = set(star_df[star_df[\"fold\"] == f][\"planet_id\"])\n    for f_other in range(f + 1, 5):\n        p_other = set(star_df[star_df[\"fold\"] == f_other][\"planet_id\"])\n        assert len(p_f.intersection(p_other)) == 0, f\"Leakage between fold {f} and {f_other}!\"\n\nvault_planets = set(star_df[star_df[\"is_vault\"]][\"planet_id\"])\nfor f in range(5):\n    p_f = set(star_df[star_df[\"fold\"] == f][\"planet_id\"])\n    assert len(vault_planets.intersection(p_f)) == 0, f\"Vault leaked into fold {f}!\"\n\nprint(f\"✓ Zero leakage confirmed across 5 folds and {len(vault_planets)} vault planets (10.2%).\")\ndisplay(pd.crosstab(star_df[\"fold\"], star_df[\"spectral_class\"]))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-29T17:33:36.465223Z","iopub.execute_input":"2026-09-29T17:33:36.465651Z","iopub.status.idle":"2026-09-29T17:33:36.834065Z","shell.execute_reply.started":"2026-09-29T17:33:36.465611Z","shell.execute_reply":"2026-09-29T17:33:36.832866Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"FGS1_WEIGHT = 57.846\nAIRS_WEIGHT = 1.0\nNUM_CHANNELS = 283\n\ndef get_official_channel_weights():\n    w = np.ones(NUM_CHANNELS, dtype=np.float64) * AIRS_WEIGHT\n    w[0] = FGS1_WEIGHT\n    return w\n\ndef compute_gll(y_true, mu_pred, sigma_pred):\n    weights = get_official_channel_weights()\n    y_true = np.asarray(y_true, dtype=np.float64)\n    mu_pred = np.asarray(mu_pred, dtype=np.float64)\n    sigma_pred = np.asarray(sigma_pred, dtype=np.float64)\n    \n    if np.any(np.isnan(y_true)) or np.any(np.isnan(mu_pred)) or np.any(np.isnan(sigma_pred)):\n        raise ValueError(\"NaN values detected in inputs or predictions!\")\n    if np.any(np.isinf(y_true)) or np.any(np.isinf(mu_pred)) or np.any(np.isinf(sigma_pred)):\n        raise ValueError(\"Infinite values detected in inputs or predictions!\")\n    if np.any(sigma_pred <= 0):\n        raise ValueError(\"sigma_pred must be strictly positive (sigma > 0)!\")\n        \n    var = sigma_pred ** 2\n    gll_per_channel = -0.5 * (np.log(2.0 * np.pi * var) + ((y_true - mu_pred) ** 2) / var)\n    return np.sum(weights * gll_per_channel, axis=-1) / np.sum(weights)\n\n# Self-test: FGS1 error penalty ratio\ny_mock = np.full((1, NUM_CHANNELS), 0.012)\nsig_mock = np.full((1, NUM_CHANNELS), 0.001)\nbase = compute_gll(y_mock, y_mock, sig_mock)[0]\n\nmu_fgs = y_mock.copy(); mu_fgs[0, 0] += 0.001\nmu_airs = y_mock.copy(); mu_airs[0, 1] += 0.001\n\nratio = (base - compute_gll(y_mock, mu_fgs, sig_mock)[0]) / (base - compute_gll(y_mock, mu_airs, sig_mock)[0])\nassert np.isclose(ratio, FGS1_WEIGHT, atol=1e-3)\nprint(f\"✓ GLL metric verified. FGS1 penalty ratio: {ratio:.4f} (expected {FGS1_WEIGHT:.4f})\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-29T17:33:36.835368Z","iopub.execute_input":"2026-09-29T17:33:36.835721Z","iopub.status.idle":"2026-09-29T17:33:36.857953Z","shell.execute_reply.started":"2026-09-29T17:33:36.835683Z","shell.execute_reply":"2026-09-29T17:33:36.856389Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_cols = [c for c in labels_df.columns if c != \"planet_id\"]\nfgs1_vals = labels_df[num_cols[0]].values\nairs_vals = labels_df[num_cols[1:]].values\n\nprint(\"=\" * 65)\nprint(\"GROUND TRUTH TRANSMISSION SPECTRUM SCALE (Rp/Rs)^2:\")\nprint(\"=\" * 65)\nprint(f\"FGS1 (Channel 0):\")\nprint(f\"  Mean:   {np.mean(fgs1_vals):.6f} ({np.mean(fgs1_vals)*1e6:.1f} ppm)\")\nprint(f\"  Std:    {np.std(fgs1_vals):.6f}\")\nprint(f\"  Min:    {np.min(fgs1_vals):.6f}\")\nprint(f\"  Max:    {np.max(fgs1_vals):.6f}\")\nprint(f\"\\nAIRS-CH0 (Channels 1-282):\")\nprint(f\"  Mean:   {np.mean(airs_vals):.6f} ({np.mean(airs_vals)*1e6:.1f} ppm)\")\nprint(f\"  Std:    {np.std(airs_vals):.6f}\")\nprint(f\"  Min:    {np.min(airs_vals):.6f}\")\nprint(f\"  Max:    {np.max(airs_vals):.6f}\")\n\nassert np.all(fgs1_vals >= 0.0) and np.all(airs_vals >= 0.0), \"Negative depths found!\"\nassert np.all(fgs1_vals <= 0.08), f\"FGS1 transit depth exceeds 8%: {np.max(fgs1_vals)}\"\nassert np.all(airs_vals <= 0.10), f\"AIRS transit depth exceeds physical bound of 10%: {np.max(airs_vals)}\"\nif np.max(airs_vals) > 0.08:\n    print(f\"  Notice: Maximum AIRS channel depth reaches {np.max(airs_vals):.6f} in deep molecular features (within <= 0.10 physical bound).\")\nprint(\"✓ Physical bounds verified: all depths lie strictly within valid exoplanetary range.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-29T17:33:36.85894Z","iopub.execute_input":"2026-09-29T17:33:36.85919Z","iopub.status.idle":"2026-09-29T17:33:36.928576Z","shell.execute_reply.started":"2026-09-29T17:33:36.85916Z","shell.execute_reply":"2026-09-29T17:33:36.926271Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dir = DATA_DIR / \"train\"\nsample_target = sorted([p for p in train_dir.iterdir() if p.is_dir()])[0]\n\nprint(f\"Target Planet: {sample_target.name}\")\n\n# Visit discovery helper\nfgs_signals = sorted(sample_target.glob(\"FGS1_signal_*.parquet\"))\nairs_signals = sorted(sample_target.glob(\"AIRS-CH0_signal_*.parquet\"))\nprint(f\"\\nVisit Discovery on Target {sample_target.name}:\")\nprint(f\"  FGS1 signal files ({len(fgs_signals)}):     {[f.name for f in fgs_signals]}\")\nprint(f\"  AIRS-CH0 signal files ({len(airs_signals)}): {[f.name for f in airs_signals]}\")\nif len(fgs_signals) > 1:\n    print(\"  -> Multi-visit structure detected on disk for this target.\")\nelse:\n    print(\"  -> Single-visit structure on disk for this target.\")\n\n# Load raw parquet signals and verify tensor shapes\nfgs_raw = pq.read_table(sample_target / \"FGS1_signal_0.parquet\").to_pandas().values\nairs_raw = pq.read_table(sample_target / \"AIRS-CH0_signal_0.parquet\").to_pandas().values\n\nprint(f\"\\nRaw FGS1 shape:     {fgs_raw.shape}, dtype: {fgs_raw.dtype}\")\nprint(f\"Raw AIRS-CH0 shape: {airs_raw.shape}, dtype: {airs_raw.dtype}\")\n\n# Verify 3D tensor reshape (no differencing or calibration applied in Stage 1)\nfgs_cube = fgs_raw.reshape(-1, 32, 32)\nairs_cube = airs_raw.reshape(-1, 32, 356)\n\nprint(f\"Reshaped FGS1 cube:     {fgs_cube.shape}\")\nprint(f\"Reshaped AIRS-CH0 cube: {airs_cube.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-29T17:33:36.929988Z","iopub.execute_input":"2026-09-29T17:33:36.930393Z","iopub.status.idle":"2026-09-29T17:33:38.946733Z","shell.execute_reply.started":"2026-09-29T17:33:36.930366Z","shell.execute_reply":"2026-09-29T17:33:38.945391Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"manifest_cols = [\n    \"star_id\", \"planet_id\", \"spectral_class\", \"Ts\", \"Rs\", \"Ms\", \"logg\", \"Mp\", \"e\", \"P\", \"sma\", \"i\", \"fold\", \"is_vault\"\n]\nmanifest_export = star_df[manifest_cols].copy()\n\n# Pre-save manifest schema checks\nfor col in manifest_cols:\n    assert col in manifest_export.columns, f\"Missing manifest column: {col}\"\nassert set(manifest_export[\"fold\"].unique()).issubset({-1, 0, 1, 2, 3, 4}), f\"Invalid folds: {manifest_export['fold'].unique()}\"\nassert np.isclose(manifest_export[\"is_vault\"].mean(), 0.102, atol=0.015), f\"Vault fraction anomaly: {manifest_export['is_vault'].mean()}\"\n\nmanifest_path = OUT_DIR / \"C01_manifest.csv\"\nmanifest_export.to_csv(manifest_path, index=False)\n\ncv_export = star_df[[\"star_id\", \"planet_id\", \"spectral_class\", \"fold\", \"is_vault\"]].copy()\ncv_path = OUT_DIR / \"C02_cv_folds.csv\"\ncv_export.to_csv(cv_path, index=False)\n\nvault_export = star_df[star_df[\"is_vault\"]][[\"star_id\", \"planet_id\", \"spectral_class\"]].copy()\nvault_path = OUT_DIR / \"vault_planets.csv\"\nvault_export.to_csv(vault_path, index=False)\n\ndef sha256(filepath):\n    h = hashlib.sha256()\n    with open(filepath, \"rb\") as f:\n        while chunk := f.read(65536):\n            h.update(chunk)\n    return h.hexdigest()\n\nprint(\"=\" * 70)\nprint(\"FROZEN STAGE 1 ARTIFACTS & SHA-256 HASHES:\")\nprint(\"=\" * 70)\nfor p in [manifest_path, cv_path, vault_path]:\n    print(f\"  {p.name:20s}: {sha256(p)}\")\nprint(\"=\" * 70)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-29T17:33:38.948587Z","iopub.execute_input":"2026-09-29T17:33:38.948846Z","iopub.status.idle":"2026-09-29T17:33:38.989619Z","shell.execute_reply.started":"2026-09-29T17:33:38.948825Z","shell.execute_reply":"2026-09-29T17:33:38.987687Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Stage 1 complete. Artifacts frozen. Proceed to Stage 2 for FGS1 detector calibration.","metadata":{}}]}