{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":101849,"databundleVersionId":13093295,"sourceType":"competition"}],"dockerImageVersionId":31090,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<div style=\"\n  background: linear-gradient(135deg, #0f172a, #1e293b);\n  border-radius: 18px;\n  padding: 36px;\n  color: #f1f5f9;\n  font-family: 'Segoe UI', Roboto, sans-serif;\n  box-shadow: 0 8px 24px rgba(0,0,0,0.45);\n  text-align: center;\n  line-height: 1.6;\n\">\n\n  <h1 style=\"font-size: 2.6rem; margin-bottom: 12px;\">\n    NeurIPS Ariel Data Challenge -Solution\n  </h1>\n\n\n  <div style=\"\n    display: inline-block;\n    text-align: left;\n    background: #1e293b;\n    padding: 20px 28px;\n    border-radius: 12px;\n    font-size: 1rem;\n    line-height: 1.8;\n    color: #e2e8f0;\n  \">\n    <p><strong style=\"color:#22c55e;\">🧠 Task:</strong> Atmospheric retrieval & spectral denoising</p>\n    <p><strong style=\"color:#38bdf8;\">📦 Dataset:</strong> Simulated transit spectra</p>\n    <p><strong style=\"color:#fbbf24;\">📏 Metric:</strong> RMSE + Wasserstein Distance</p>\n  </div>\n\n</div>\n","metadata":{"tags":[]}},{"cell_type":"code","source":"import pandas as pd\nimport polars as pl\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport seaborn as sns\nimport scipy.stats\nfrom tqdm import tqdm\nimport pickle\n\nfrom sklearn.model_selection import cross_val_predict\nfrom sklearn.linear_model import Ridge\nfrom sklearn.metrics import r2_score, mean_squared_error\n\nprint('All import are done')","metadata":{"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-08-14T17:34:59.709772Z","iopub.execute_input":"2025-08-14T17:34:59.710125Z","iopub.status.idle":"2025-08-14T17:34:59.715697Z","shell.execute_reply.started":"2025-08-14T17:34:59.710101Z","shell.execute_reply":"2025-08-14T17:34:59.715047Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<div style=\"\n  background: linear-gradient(135deg, #1e293b, #0f172a);\n  padding: 20px 30px;\n  border-radius: 14px;\n  color: #f8fafc;\n  font-family: 'Segoe UI', sans-serif;\n  box-shadow: 0 4px 16px rgba(0,0,0,0.4);\n  margin-bottom: 10px;\n\">\n\n  <h2 style=\"margin: 0; font-size: 1.8rem; font-weight: 600;\">\n    🔍 Imports & Data Inspection\n  </h2>\n  <p style=\"color: #94a3b8; margin: 6px 0 0; font-size: 1rem;\">\n    Loading libraries and exploring the dataset structure.\n  </p>\n\n</div>\n","metadata":{"tags":[]}},{"cell_type":"code","source":"import os\nimport pandas as pd\nfrom glob import glob\nfrom collections import defaultdict\nimport numpy as np\n\n# --- Load CSVs ---\ntrain_df = pd.read_csv('/kaggle/input/ariel-data-challenge-2025/train.csv')\nwavelengths_df = pd.read_csv('/kaggle/input/ariel-data-challenge-2025/wavelengths.csv')\ntrain_star_info = pd.read_csv('/kaggle/input/ariel-data-challenge-2025/train_star_info.csv')\n\nprint('All data loaded ....!!!')\n\n# --- 1. Basic Train Set Stats ---\nprint(\"🪐 Number of training planets:\", train_df.shape[0])\nprint(\"📈 Number of target labels (wavelengths):\", train_df.shape[1] - 1)\nprint(\"🔬 Length of wavelength grid:\", wavelengths_df.shape[0])\n\n# --- 2. Target Stats (per flux column) ---\ntarget_cols = [col for col in train_df.columns if col != 'planet_id']\nflux_summary = train_df[target_cols].agg(['min', 'max', 'mean', 'std']).T\nprint(\"\\n📊 Flux value summary (first 5 rows):\")\nprint(flux_summary.head())\n\n# --- 3. Unique Stars ---\nif 'planet_id' in train_star_info.columns:\n    num_stars = train_star_info.drop(columns='planet_id').drop_duplicates().shape[0]\nelse:\n    num_stars = train_star_info.drop_duplicates().shape[0]\nprint(\"\\n🌟 Number of unique stars in training:\", num_stars)\n\n# --- 4. Planets with Multiple Observations ---\nobs_counts = defaultdict(int)\ntrain_planets = os.listdir('/kaggle/input/ariel-data-challenge-2025/train')\n\nfor pid in train_planets:\n    air_obs = glob(f\"train/{pid}/AIRS-CH0_signal_*.parquet\")\n    obs_counts[pid] = len(air_obs)\n\nmulti_obs = {pid: count for pid, count in obs_counts.items() if count > 1}\nprint(\"\\n🔁 Planets with multiple observations:\", len(multi_obs))\n\n# --- 5. Check Calibration File Coverage ---\nmissing_calibs = []\nexpected = {\"dark\", \"dead\", \"flat\", \"linear_corr\", \"read\"}\n\nfor pid in train_planets:\n    for band in [\"AIRS-CH0\", \"FGS1\"]:\n        calib_path = f\"train/{pid}/{band}_calibration\"\n        calib_files = {os.path.splitext(f)[0] for f in os.listdir(calib_path)} if os.path.exists(calib_path) else set()\n        missing = expected - calib_files\n        if missing:\n            missing_calibs.append((pid, band, missing))\n\nprint(\"\\n🧪 Planets missing calibration files:\", len(missing_calibs))\nif missing_calibs:\n    print(\"   Example:\", missing_calibs[0])\n\n# --- 6. Optional: Distribution of Observations Per Planet ---\nobs_distribution = pd.Series(list(obs_counts.values())).value_counts().sort_index()\nprint(\"\\n🗂 Observation count distribution per planet (AIR-CH0):\")\nprint(obs_distribution)\n\n# --- 7. Planet-Star Uniqueness Check ---\nmerged = pd.merge(train_df[['planet_id']], train_star_info, on='planet_id', how='left')\nunique_links = merged[['planet_id'] + [col for col in train_star_info.columns if col != 'planet_id']].drop_duplicates()\nprint(\"\\n🔗 Unique planet-star mappings:\", unique_links.shape[0])\n","metadata":{"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-08-14T17:34:59.71697Z","iopub.execute_input":"2025-08-14T17:34:59.717224Z","iopub.status.idle":"2025-08-14T17:35:00.166781Z","shell.execute_reply.started":"2025-08-14T17:34:59.717205Z","shell.execute_reply":"2025-08-14T17:35:00.166092Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<div style=\"\n  background: linear-gradient(135deg, #111827, #0f172a);\n  border-radius: 18px;\n  padding: 36px 40px;\n  font-family: 'Segoe UI', sans-serif;\n  box-shadow: 0 8px 30px rgba(0,0,0,0.45);\n  color: #f1f5f9;\n  line-height: 1.7;\n\">\n\n  <h2 style=\"font-size: 2rem; color: #7dd3fc; margin-bottom: 20px;\">\n    📊 Key Dataset Insights — Ariel 2025\n  </h2>\n\n  <ul style=\"font-size: 1rem; padding-left: 20px; list-style-type: '▹ ';\">\n    <li><span style=\"color:#facc15;\">1100 training planets</span>, each linked to a <strong>unique star</strong>.</li>\n    <li>Each has <span style=\"color:#4ade80;\">283 spectral flux targets</span>.</li>\n    <li><strong>Single wavelength grid</strong> shared across all targets.</li>\n    <li>One-to-one <span style=\"color:#a78bfa;\">planet-star mapping</span> (no duplicates).</li>\n    <li><strong>No repeated observations</strong> per planet in training set.</li>\n    <li><span style=\"color:#60a5fa;\">2200 missing calibration sets</span> (e.g., AIRS-CH0 & FGS1 per planet).</li>\n    <li><code style=\"background:#1f2937; padding:3px 6px; border-radius:6px;\">Planet 1253730513</code> is missing all AIRS-CH0 calibration files.</li>\n    <li>Flux variability (first 5 wavelengths):\n      <ul style=\"list-style-type: '– '; margin-left: 20px;\">\n        <li>Mean ≈ <code>0.0146</code></li>\n        <li>Std Dev ≈ <code>0.0105</code></li>\n      </ul>\n    </li>\n    <li>All planets have <strong>0 signal observations</strong> in <code>AIR-CH0</code>.</li>\n  </ul>\n\n</div>\n","metadata":{"tags":[]}},{"cell_type":"code","source":"train_adc_info = pd.read_csv('/kaggle/input/ariel-data-challenge-2025/adc_info.csv')\n# test_adc_info = pd.read_csv('/kaggle/input/ariel-data-challenge-2024/test_adc_info.csv',\n#                            index_col='planet_id')\ntrain_labels = pd.read_csv('/kaggle/input/ariel-data-challenge-2025/train.csv',\n                           index_col='planet_id')\nwavelengths = pd.read_csv('/kaggle/input/ariel-data-challenge-2025/wavelengths.csv')\naxis_info = pd.read_parquet('/kaggle/input/ariel-data-challenge-2025/axis_info.parquet')\n","metadata":{"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-08-14T17:35:00.167948Z","iopub.execute_input":"2025-08-14T17:35:00.168257Z","iopub.status.idle":"2025-08-14T17:35:00.425876Z","shell.execute_reply.started":"2025-08-14T17:35:00.168228Z","shell.execute_reply":"2025-08-14T17:35:00.425132Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<div style=\"\n  background: linear-gradient(135deg, #2e1065, #1e1b4b);\n  border-radius: 16px;\n  padding: 32px 36px;\n  color: #fdf4ff;\n  font-family: 'Segoe UI', sans-serif;\n  box-shadow: 0 6px 22px rgba(0,0,0,0.5);\n  line-height: 1.75;\n\">\n\n  <h2 style=\"font-size: 1.8rem; color: #c084fc; margin-bottom: 18px;\">\n    🛰️ The FGS1 Data\n  </h2>\n\n  <p style=\"font-size: 1.05rem; color: #f3e8ff; margin-bottom: 14px;\">\n    After reviewing the metadata, we dive into the <strong>FGS1 (Fine Guidance System)</strong> data. \n    This includes one file per planet — precisely, \n    <span style=\"color:#facc15;\"><strong>673 files for 673 planets</strong></span> in the training dataset. \n    For now, calibration files are excluded.\n  </p>\n\n  <p style=\"font-size: 1.05rem; color: #f3e8ff; margin-bottom: 14px;\">\n    Each file contains <strong>135,000 rows</strong> of images captured at \n    <span style=\"color:#4ade80;\"><strong>0.1-second intervals</strong></span>. Each row holds a \n    <span style=\"color:#fb7185;\"><code>32 × 32</code></span> grayscale image representing a single wavelength.\n  </p>\n\n  <p style=\"font-size: 1.05rem; color: #f3e8ff;\">\n    Below, we inspect a sample file to understand the structure and form of FGS1 data.\n  </p>\n\n</div>\n","metadata":{"tags":[]}},{"cell_type":"code","source":"planet_id = 1010375142\nf_signal = pd.read_parquet(f'/kaggle/input/ariel-data-challenge-2025/train/{planet_id}/FGS1_signal_0.parquet')\nf_signal","metadata":{"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-08-14T17:35:00.427413Z","iopub.execute_input":"2025-08-14T17:35:00.427671Z","iopub.status.idle":"2025-08-14T17:35:01.972989Z","shell.execute_reply.started":"2025-08-14T17:35:00.427652Z","shell.execute_reply":"2025-08-14T17:35:01.972202Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<div style=\"\n  background: linear-gradient(135deg, #1c1917, #292524);\n  border-left: 6px solid #f59e0b;\n  border-radius: 16px;\n  padding: 28px 36px;\n  font-family: 'Segoe UI', sans-serif;\n  color: #fefce8;\n  box-shadow: 0 8px 26px rgba(0,0,0,0.5);\n  line-height: 1.7;\n\">\n\n  <h2 style=\"font-size: 1.7rem; color: #fbbf24; margin-bottom: 14px;\">\n    🖼️ Image Pairs in FGS1 Data\n  </h2>\n\n  <p style=\"font-size: 1.05rem; margin-bottom: 10px;\">\n    Each row in an FGS1 file represents a single image of the target star.\n    The images are captured in <strong style=\"color:#fde68a;\">pairs</strong>, where:\n  </p>\n\n  <ul style=\"margin-left: 24px; font-size: 1.05rem; color: #fcd34d;\">\n    <li><strong>First image</strong> — darker (baseline)</li>\n    <li><strong>Second image</strong> — lighter (flux increase)</li>\n  </ul>\n\n  <p style=\"font-size: 1.05rem; color: #fef9c3; margin-top: 10px;\">\n    This alternating pattern forms the basis for computing the net signal per planet.\n  </p>\n\n</div>\n","metadata":{"tags":[]}},{"cell_type":"code","source":"_, (ax1, ax2) = plt.subplots(1, 2, figsize=(12, 4))\nsns.heatmap(f_signal.iloc[0].values.reshape(32, 32), ax=ax1, vmin=0, vmax=52000)\nax1.set_aspect('equal')\nsns.heatmap(f_signal.iloc[1].values.reshape(32, 32), ax=ax2, vmin=0, vmax=52000)\nax2.set_aspect('equal')\nplt.suptitle('A pair of FGS1 images')\nplt.show()","metadata":{"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-08-14T17:35:01.973792Z","iopub.execute_input":"2025-08-14T17:35:01.974032Z","iopub.status.idle":"2025-08-14T17:35:02.604243Z","shell.execute_reply.started":"2025-08-14T17:35:01.974013Z","shell.execute_reply":"2025-08-14T17:35:02.603512Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<div style=\"\n  background: linear-gradient(135deg, #0f172a, #0a0f1c);\n  border-left: 6px solid #06b6d4;\n  border-radius: 16px;\n  padding: 36px 42px;\n  font-family: 'Segoe UI', sans-serif;\n  box-shadow: 0 10px 28px rgba(0,0,0,0.5);\n  color: #e0f2fe;\n  line-height: 1.75;\n\">\n\n  <h2 style=\"font-size: 1.85rem; margin-bottom: 14px; color: #22d3ee;\">\n    🌗 Net Signal Extraction from Time Series\n  </h2>\n\n  <p style=\"font-size: 1.05rem; margin-bottom: 14px;\">\n    To visualize the transit signal, we compute the \n    <strong style=\"color:#facc15;\">difference between even and odd frames</strong>, giving us a \n    <code style=\"background:#164e63; padding:3px 6px; border-radius: 6px;\">67500-step</code> \n    net signal per planet.\n  </p>\n\n  <p style=\"font-size: 1.05rem; margin-bottom: 14px;\">\n    We then <strong>average across all 1024 pixels</strong> to reduce dimensionality. \n    The signal is <span style=\"color:#f87171;\">noisy</span>, so we apply a \n    <strong style=\"color:#2dd4bf;\">moving average filter</strong> for smoothening.\n  </p>\n\n  <p style=\"font-size: 1.05rem; margin-bottom: 14px;\">\n    The result reveals the light curve. A clear dip in signal intensity is seen between time steps \n    <code style=\"background:#164e63; padding:3px 6px; border-radius: 6px;\">23500</code> and \n    <code style=\"background:#164e63; padding:3px 6px; border-radius: 6px;\">44000</code>, \n    as the planet transits its star.\n  </p>\n\n  <p style=\"font-size: 1.05rem;\">\n    Diagrams below illustrate the difference:\n    <ul style=\"margin-top: 8px; padding-left: 24px;\">\n      <li><strong>Left:</strong> Strong dip (high-opacity transit)</li>\n      <li><strong>Right:</strong> Weak dip (low-opacity transit)</li>\n    </ul>\n  </p>\n\n</div>\n","metadata":{"tags":[]}},{"cell_type":"code","source":"_, ((ax1, ax2), (ax3, ax4)) = plt.subplots(2, 2, sharex=True, figsize=(12, 4))\n\nplanet_id = 1048114509\nf_signal = pd.read_parquet(f'/kaggle/input/ariel-data-challenge-2025/train/{planet_id}/FGS1_signal_0.parquet')\n\nmean_signal = f_signal.values.mean(axis=1)\nnet_signal = mean_signal[1::2] - mean_signal[0::2]\ncum_signal = net_signal.cumsum()\nwindow=800\nsmooth_signal = (cum_signal[window:] - cum_signal[:-window]) / window\n\nax1.set_title('FGS1: time series of planet with strong signal')\nax1.plot(net_signal, label='raw signal')\nax1.legend()\nax3.plot(smooth_signal, color='c', label='smoothened signal')\nax3.legend()\nax3.set_xlabel('time step')\nfor time_step in [20500, 23500, 44000, 47000]:\n    ax3.axvline(time_step, color='gray')\n\nplanet_id = 1240764363\nf_signal = pd.read_parquet(f'/kaggle/input/ariel-data-challenge-2025/train/{planet_id}/FGS1_signal_0.parquet')\n\nmean_signal = f_signal.values.mean(axis=1)\nnet_signal = mean_signal[1::2] - mean_signal[0::2]\ncum_signal = net_signal.cumsum()\nwindow=800\nsmooth_signal = (cum_signal[window:] - cum_signal[:-window]) / window\n\nax2.set_title('FGS1: time series of planet with weak signal')\nax2.plot(net_signal, label='raw signal')\nax2.legend()\nax4.plot(smooth_signal, color='c', label='smoothened signal')\nax4.legend()\nax4.set_xlabel('time step')\nfor time_step in [20500, 23500, 44000, 47000]:\n    ax4.axvline(time_step, color='gray')\n\n# plt.suptitle('FGS1 time series', y=0.96)\nplt.show()","metadata":{"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-08-14T17:35:02.605061Z","iopub.execute_input":"2025-08-14T17:35:02.605355Z","iopub.status.idle":"2025-08-14T17:35:05.948598Z","shell.execute_reply.started":"2025-08-14T17:35:02.605326Z","shell.execute_reply":"2025-08-14T17:35:05.947774Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<div style=\"\n  background: linear-gradient(135deg, #1e3a8a, #0f172a);\n  border-radius: 16px;\n  padding: 32px 36px;\n  color: #f8fafc;\n  font-family: 'Segoe UI', sans-serif;\n  box-shadow: 0 6px 20px rgba(0,0,0,0.5);\n  line-height: 1.7;\n\">\n\n  <h2 style=\"font-size: 1.8rem; color: #93c5fd; margin-bottom: 18px;\">\n    🌌 The AIRS Data\n  </h2>\n\n  <p style=\"font-size: 1rem; color: #e2e8f0; margin-bottom: 14px;\">\n    <strong>AIRS</strong> is the second primary sensor aboard the Ariel satellite. Like FGS1,\n    it produces <strong>one file per planet</strong> in the dataset.\n    Each file provides <span style=\"color:#facc15;\"><strong>11,250 image rows</strong></span> captured\n    at uniform time intervals.\n  </p>\n\n  <p style=\"font-size: 1rem; color: #e2e8f0; margin-bottom: 14px;\">\n    Each image is originally of shape <code style=\"background:#1e293b;padding:2px 6px;border-radius:6px;\">32 × 356</code>,\n    but is stored in the files as a <span style=\"color:#34d399;\"><strong>flattened row with 11,392 columns</strong></span>.\n  </p>\n\n  <p style=\"font-size: 1rem; color: #e2e8f0;\">\n    Below, we will read a sample AIRS file to inspect its structure and begin understanding its format.\n  </p>\n\n</div>\n","metadata":{"tags":[]}},{"cell_type":"code","source":"planet_id = 1240764363\na_signal = pd.read_parquet(f'/kaggle/input/ariel-data-challenge-2025/train/{planet_id}/AIRS-CH0_signal_0.parquet')\na_signal","metadata":{"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-08-14T17:35:05.949246Z","iopub.execute_input":"2025-08-14T17:35:05.949436Z","iopub.status.idle":"2025-08-14T17:35:07.3296Z","shell.execute_reply.started":"2025-08-14T17:35:05.949421Z","shell.execute_reply":"2025-08-14T17:35:07.328812Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"a_signal = a_signal.values.reshape(11250, 32, 356)\n\nplt.figure(figsize=(10, 3))\nsns.heatmap(a_signal[1])\nplt.ylabel('spatial dimension')\nplt.xlabel('wavelength dimension')\nplt.show()","metadata":{"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-08-14T17:35:07.330403Z","iopub.execute_input":"2025-08-14T17:35:07.330618Z","iopub.status.idle":"2025-08-14T17:35:07.707448Z","shell.execute_reply.started":"2025-08-14T17:35:07.330602Z","shell.execute_reply":"2025-08-14T17:35:07.706746Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The data again is a time series, and we can see how the star is obscured while the planet is passing in front of it.","metadata":{"tags":[]}},{"cell_type":"code","source":"mean_signal = a_signal.mean(axis=2).mean(axis=1)\nnet_signal = mean_signal[1::2] - mean_signal[0::2]\ncum_signal = net_signal.cumsum()\nwindow=80\nsmooth_signal = (cum_signal[window:] - cum_signal[:-window]) / window\n\n_, (ax1, ax2) = plt.subplots(2, 1, sharex=True)\nax1.plot(net_signal, label='raw net signal')\nax1.legend()\nax2.plot(smooth_signal, color='c', label='smoothened net signal')\nax2.legend()\nax2.set_xlabel('time')\nfor time_step in [20500, 23500, 44000, 47000]:\n    ax2.axvline(time_step * 11250 // 135000, color='gray')\nplt.suptitle('AIRS-CH0 time series', y=0.96)\nplt.show()\n","metadata":{"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-08-14T17:35:07.708267Z","iopub.execute_input":"2025-08-14T17:35:07.708551Z","iopub.status.idle":"2025-08-14T17:35:08.143499Z","shell.execute_reply.started":"2025-08-14T17:35:07.708532Z","shell.execute_reply":"2025-08-14T17:35:08.142758Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<div style=\"\n  background: linear-gradient(135deg, #0f172a, #052e16);\n  border-left: 6px solid #10b981;\n  padding: 36px 42px;\n  border-radius: 16px;\n  font-family: 'Segoe UI', sans-serif;\n  color: #ecfdf5;\n  box-shadow: 0 8px 28px rgba(0,0,0,0.6);\n  line-height: 1.8;\n\">\n\n  <h2 style=\"font-size: 1.9rem; margin-bottom: 14px; color: #34d399;\">\n    📦 Reading the Data\n  </h2>\n\n  <p style=\"font-size: 1.08rem; margin-bottom: 12px;\">\n    We read signal data for all <strong style=\"color:#facc15;\">1100 training planets</strong>, combining streams from both onboard sensors:\n  </p>\n\n  <ul style=\"font-size: 1.05rem; padding-left: 24px; margin-bottom: 18px; color: #d1fae5;\">\n    <li><strong>FGS1:</strong> Extract a <code style=\"background:#064e3b; padding:3px 6px; border-radius:6px;\">67500-step</code> 1D time series</li>\n    <li><strong>AIRS-CH0:</strong> Extract a <code style=\"background:#064e3b; padding:3px 6px; border-radius:6px;\">5625-step</code> 1D time series</li>\n  </ul>\n\n  <p style=\"font-size: 1.05rem;\">\n    To ensure reproducibility during inference, we use \n    <code style=\"background:#064e3b; padding: 3px 6px; border-radius: 5px;\">%%writefile</code> \n    to export the logic and apply it to test data identically.\n  </p>\n\n</div>\n","metadata":{"tags":[]}},{"cell_type":"code","source":"%%writefile f_read_and_preprocess.py\n\ndef f_read_and_preprocess(dataset, adc_info, planet_ids):\n    \"\"\"Read the FGS1 files for all planet_ids and extract the time series.\n    \n    Parameters\n    dataset: 'train' or 'test'\n    adc_info: metadata dataframe, either train_adc_info or test_adc_info\n    planet_ids: list of planet ids\n    \n    Returns\n    dataframe with one row per planet_id and 67500 values per row\n    \n    \"\"\"\n    f_raw_train = np.full((len(planet_ids), 67500), np.nan, dtype=np.float32)\n    for i, planet_id in tqdm(list(enumerate(planet_ids))):\n        f_signal = pl.read_parquet(f'/kaggle/input/ariel-data-challenge-2025/{dataset}/{planet_id}/FGS1_signal_0.parquet')\n        mean_signal = f_signal.cast(pl.Int32).sum_horizontal().cast(pl.Float32).to_numpy() / 1024 # mean over the 32*32 pixels\n        net_signal = mean_signal[1::2] - mean_signal[0::2]\n        f_raw_train[i] = net_signal\n    return f_raw_train\n    ","metadata":{"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-08-14T17:35:08.145969Z","iopub.execute_input":"2025-08-14T17:35:08.146179Z","iopub.status.idle":"2025-08-14T17:35:08.151636Z","shell.execute_reply.started":"2025-08-14T17:35:08.146164Z","shell.execute_reply":"2025-08-14T17:35:08.151068Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nexec(open('f_read_and_preprocess.py', 'r').read())\nf_raw_train = f_read_and_preprocess('train', train_adc_info, train_labels.index)\nwith open('f_raw_train.pickle', 'wb') as f:\n    pickle.dump(f_raw_train, f)\n\n\nprint('Data Traiing completed !!!!')","metadata":{"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-08-14T17:35:08.152836Z","iopub.execute_input":"2025-08-14T17:35:08.153085Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile a_read_and_preprocess.py\ndef a_read_and_preprocess(dataset, adc_info, planet_ids):\n    \"\"\"Read the AIRS-CH0 files for all planet_ids and extract the time series.\n    \n    Parameters\n    dataset: 'train' or 'test'\n    adc_info: metadata dataframe, either train_adc_info or test_adc_info\n    planet_ids: list of planet ids\n    \n    Returns\n    dataframe with one row per planet_id and 5625 values per row\n    \n    \"\"\"\n    a_raw_train = np.full((len(planet_ids), 5625), np.nan, dtype=np.float32)\n    for i, planet_id in tqdm(list(enumerate(planet_ids))):\n        signal = pl.read_parquet(f'/kaggle/input/ariel-data-challenge-2025/{dataset}/{planet_id}/AIRS-CH0_signal_0.parquet')\n        mean_signal = signal.cast(pl.Int32).sum_horizontal().cast(pl.Float32).to_numpy() / (32*356) # mean over the 32*356 pixels\n        net_signal = mean_signal[1::2] - mean_signal[0::2]\n        a_raw_train[i] = net_signal\n    return a_raw_train\n    ","metadata":{"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nexec(open('a_read_and_preprocess.py', 'r').read())\na_raw_train = a_read_and_preprocess('train', train_adc_info, train_labels.index)\nwith open('a_raw_train.pickle', 'wb') as f:\n    pickle.dump(a_raw_train, f)\n","metadata":{"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"As a plausibility check, we plot the means of all time series:","metadata":{"tags":[]}},{"cell_type":"code","source":"plt.figure(figsize=(6, 2))\nplt.plot(f_raw_train.mean(axis=0))\nfor time_step in [20500, 23500, 44000, 47000]:\n    plt.axvline(time_step, color='gray')\nplt.xlabel('time step')\nplt.title('FGS1: Overall mean')\nplt.show()\n\nplt.figure(figsize=(6, 2))\nplt.plot(a_raw_train.mean(axis=0))\nfor time_step in [20500, 23500, 44000, 47000]:\n    plt.axvline(time_step * 11250 // 135000, color='gray')\nplt.xlabel('time step')\nplt.title('AIRS-CH0: Overall mean')\nplt.show()","metadata":{"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<div style=\"\n  background: #111827;\n  border: 1px solid #4b5563;\n  border-left: 6px solid #8b5cf6;\n  border-radius: 14px;\n  padding: 34px 40px;\n  font-family: 'Segoe UI', sans-serif;\n  color: #e5e7eb;\n  line-height: 1.75;\n  box-shadow: 0 6px 24px rgba(0,0,0,0.45);\n\">\n\n  <h2 style=\"font-size: 1.85rem; margin-bottom: 16px; color: #a78bfa;\">\n    🧠 Feature Engineering\n  </h2>\n\n  <p style=\"font-size: 1.05rem; margin-bottom: 14px;\">\n    To detect planetary transits, we analyze how much the star's brightness drops as the planet crosses in front.\n    This is captured as a <strong style=\"color:#facc15;\">drop in signal intensity</strong> over time.\n  </p>\n\n  <p style=\"font-size: 1.05rem; margin-bottom: 12px;\">\n    Time series diagrams above reveal that planets cause an average brightness reduction of \n    <strong style=\"color:#fb7185;\">~0.2%</strong>.\n  </p>\n\n  <div style=\"background:#1f2937; padding: 14px 20px; border-radius: 8px; font-size: 1rem; margin-bottom: 16px;\">\n    <ul style=\"margin: 0; padding-left: 20px;\">\n      <li>From <code style=\"background:#374151; padding: 2px 6px; border-radius: 4px;\">228.2</code> to <code style=\"background:#374151; padding: 2px 6px; border-radius: 4px;\">227.6</code></li>\n      <li>From <code style=\"background:#374151; padding: 2px 6px; border-radius: 4px;\">1371</code> to <code style=\"background:#374151; padding: 2px 6px; border-radius: 4px;\">1368</code></li>\n    </ul>\n  </div>\n\n  <p style=\"font-size: 1.05rem;\">\n    This value becomes a core feature for training models that predict spectral characteristics based on transit depth.\n  </p>\n\n</div>\n","metadata":{"tags":[]}},{"cell_type":"code","source":"%%writefile feature_engineering.py\n\ndef feature_engineering(f_raw, a_raw):\n    \"\"\"Create a dataframe with two features from the raw data.\n    \n    Parameters:\n    f_raw: ndarray of shape (n_planets, 67500)\n    a_raw: ndarray of shape (n_planets, 5625)\n    \n    Return value:\n    df: DataFrame of shape (n_planets, 2)\n    \"\"\"\n    obscured = f_raw[:, 23500:44000].mean(axis=1)\n    unobscured = (f_raw[:, :20500].mean(axis=1) + f_raw[:, 47000:].mean(axis=1)) / 2\n    f_relative_reduction = (unobscured - obscured) / unobscured\n    obscured = a_raw[:, 1958:3666].mean(axis=1)\n    unobscured = (a_raw[:, :1708].mean(axis=1) + a_raw[:, 3916:].mean(axis=1)) / 2\n    a_relative_reduction = (unobscured - obscured) / unobscured\n\n    df = pd.DataFrame({'a_relative_reduction': a_relative_reduction,\n                       'f_relative_reduction': f_relative_reduction})\n    \n    return df\n","metadata":{"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"exec(open('feature_engineering.py', 'r').read())\n\ntrain = feature_engineering(f_raw_train, a_raw_train)","metadata":{"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<div style=\"\n  background: linear-gradient(135deg, #0f172a, #1e293b);\n  border-left: 6px solid #3b82f6;\n  border-radius: 16px;\n  padding: 36px 42px;\n  font-family: 'Segoe UI', sans-serif;\n  box-shadow: 0 10px 28px rgba(0,0,0,0.55);\n  color: #e0f2fe;\n  line-height: 1.75;\n\">\n\n  <h2 style=\"font-size: 1.85rem; margin-bottom: 16px; color: #60a5fa;\">\n    🤖 Model & Cross-Validation Strategy\n  </h2>\n\n  <p style=\"font-size: 1.08rem; margin-bottom: 14px;\">\n    For interpretability and performance, we use \n    <strong style=\"color:#93c5fd;\">Ridge Regression</strong> to predict the target flux values.\n  </p>\n\n  <p style=\"font-size: 1.05rem; margin-bottom: 14px;\">\n    Three metrics are used to evaluate model quality:\n  </p>\n\n  <ol style=\"font-size: 1.05rem; padding-left: 24px; margin-bottom: 18px; color: #bae6fd;\">\n    <li><strong>R² Score</strong>: Exceeds <span style=\"color:#4ade80;\">0.9</span>, confirming strong correlation.</li>\n    <li><strong>RMSE</strong>: Root Mean Squared Error is used as an indicator of prediction uncertainty.</li>\n    <li><strong>Competition Metric</strong>: Reflects leaderboard performance but depends on \n        <code style=\"background:#1e40af; padding: 2px 6px; border-radius: 5px;\">sigma_true</code>, \n        which is unknown during training.</li>\n  </ol>\n\n  <p style=\"font-size: 1.05rem;\">\n    These metrics guide us in balancing generalization, error modeling, and competitive performance.\n  </p>\n\n</div>\n","metadata":{"tags":[]}},{"cell_type":"code","source":"model = Ridge(alpha=1e-12)\n\noof_pred = cross_val_predict(model, train, train_labels)\n\nprint(f\"# R2 score: {r2_score(train_labels, oof_pred):.3f}\")\nsigma_pred = mean_squared_error(train_labels, oof_pred, squared=False)\nprint(f\"# Root mean squared error: {sigma_pred:.6f}\")\n# R2 score: 0.971\n# Root mean squared error: 0.000293","metadata":{"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"col = 1\nplt.scatter(oof_pred[:,col], train_labels.iloc[:,col], s=15, c='lightgreen')\nplt.gca().set_aspect('equal')\nplt.xlabel('y_pred')\nplt.ylabel('y_true')\nplt.title('Comparing y_true and y_pred')\nplt.show()","metadata":{"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile competition_score.py\n# Adapted from https://www.kaggle.com/code/metric/ariel-gaussian-log-likelihood\n\n# Custom error for invalid submissions\nclass ParticipantVisibleError(Exception):\n    pass\n\ndef competition_score(\n    solution: pd.DataFrame,\n    submission: pd.DataFrame,\n    naive_mean: float,\n    naive_sigma: float,\n    sigma_true: float,\n    row_id_column_name='planet_id'\n) -> float:\n    '''\n    Computes a Gaussian Log Likelihood-based score.\n    '''\n    # Drop ID columns\n    solution = solution.drop(columns=[row_id_column_name], errors='ignore')\n    submission = submission.drop(columns=[row_id_column_name], errors='ignore')\n\n    # Validation checks\n    if submission.min().min() < 0:\n        raise ParticipantVisibleError('Negative values in the submission')\n\n    for col in submission.columns:\n        if not pd.api.types.is_numeric_dtype(submission[col]):\n            raise ParticipantVisibleError(f'Submission column {col} must be numeric: {col}')\n\n    n_wavelengths = len(solution.columns)\n    if len(submission.columns) != 2 * n_wavelengths:\n        raise ParticipantVisibleError('Submission must have 2x columns of the solution')\n\n    # Extract predictions and sigmas\n    y_pred = submission.iloc[:, :n_wavelengths].values\n    sigma_pred = np.clip(submission.iloc[:, n_wavelengths:].values, a_min=1e-15, a_max=None)\n    y_true = solution.values\n\n    # Compute log likelihoods\n    GLL_pred = np.sum(scipy.stats.norm.logpdf(y_true, loc=y_pred, scale=sigma_pred))\n    GLL_true = np.sum(scipy.stats.norm.logpdf(y_true, loc=y_true, scale=sigma_true))\n    GLL_mean = np.sum(scipy.stats.norm.logpdf(y_true, loc=naive_mean, scale=naive_sigma))\n\n    score = (GLL_pred - GLL_mean) / (GLL_true - GLL_mean)\n    return float(np.clip(score, 0.0, 1.0))","metadata":{"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n\ndef postprocessing(pred_array, index, sigma_pred, column_names=None):\n    \"\"\"\n    Creates a submission DataFrame with mean predictions and uncertainties.\n\n    Parameters:\n    - pred_array: ndarray of shape (n_samples, 283)\n    - index: pandas.Index of length n_samples\n    - sigma_pred: float or ndarray of shape (n_samples, 283)\n    - column_names: list of wavelength column names (optional)\n\n    Returns:\n    - df: DataFrame of shape (n_samples, 566)\n    \"\"\"\n    n_samples, n_waves = pred_array.shape\n\n    if column_names is None:\n        column_names = [f\"wl_{i+1}\" for i in range(n_waves)]\n\n    if np.isscalar(sigma_pred):\n        sigma_pred = np.full_like(pred_array, sigma_pred)\n\n    # Safety check\n    assert sigma_pred.shape == pred_array.shape, \"sigma_pred must match shape of pred_array\"\n    assert len(index) == n_samples, \"Index length must match number of rows\"\n\n    df_mean = pd.DataFrame(pred_array.clip(0, None), index=index, columns=column_names)\n    df_sigma = pd.DataFrame(sigma_pred, index=index, columns=[f\"sigma_{i+1}\" for i in range(n_waves)])\n\n    return pd.concat([df_mean, df_sigma], axis=1)","metadata":{"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"exec(open('competition_score.py', 'r').read())\n#exec(open('postprocessing.py', 'r').read())\n\noof_df = postprocessing(oof_pred, train_labels.index, sigma_pred)\ndisplay(oof_df)\n\ngll_score = competition_score(train_labels.copy().reset_index(),\n                              oof_df.copy().reset_index(),\n                              naive_mean=train_labels.values.mean(),\n                              naive_sigma=train_labels.values.std(),\n                              sigma_true=0.000003)\nprint(f\"# Estimated competition score: {gll_score:.3f}\")\n# Estimated competition score: 0.123","metadata":{"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<div style=\"\n  background: linear-gradient(135deg, #1f1f1f, #111827);\n  border-left: 6px solid #ef4444;\n  border-radius: 16px;\n  padding: 34px 40px;\n  font-family: 'Segoe UI', sans-serif;\n  color: #f3f4f6;\n  box-shadow: 0 10px 28px rgba(0,0,0,0.55);\n  line-height: 1.75;\n\">\n\n  <h2 style=\"font-size: 1.85rem; margin-bottom: 16px; color: #f87171;\">\n    💾 Refitting and Saving the Model\n  </h2>\n\n  <p style=\"font-size: 1.05rem; margin-bottom: 14px;\">\n    After evaluating our cross-validation performance, we retrain the model on the full training set to maximize learning from all available data.\n  </p>\n\n  <p style=\"font-size: 1.05rem; margin-bottom: 14px;\">\n    The final Ridge Regression model is then serialized and saved to disk, ensuring consistency for inference during submission.\n  </p>\n\n  <p style=\"font-size: 1.05rem;\">\n    This completes the modeling pipeline — the saved model will now be loaded in the test-time inference notebook.\n  </p>\n\n</div>\n","metadata":{"tags":[]}},{"cell_type":"code","source":"# Refit the model to the full dataset\nmodel.fit(train, train_labels)\nwith open('model.pickle', 'wb') as f:\n    pickle.dump(model, f)\nwith open('sigma_pred.pickle', 'wb') as f:\n    pickle.dump(sigma_pred, f)","metadata":{"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<div style=\"\n  background: linear-gradient(135deg, #1e293b, #111827);\n  border-left: 6px solid #fbbf24;\n  border-radius: 16px;\n  padding: 34px 42px;\n  font-family: 'Segoe UI', sans-serif;\n  color: #fefce8;\n  box-shadow: 0 10px 26px rgba(0,0,0,0.5);\n  line-height: 1.75;\n\">\n\n  <h2 style=\"font-size: 1.85rem; margin-bottom: 16px; color: #fde68a;\">\n    📤 Submission\n  </h2>\n\n  <p style=\"font-size: 1.05rem; margin-bottom: 14px;\">\n    We generate the final predictions using the refitted model and prepare the submission file in the required format.\n  </p>\n\n  <p style=\"font-size: 1.05rem; margin-bottom: 14px;\">\n    The output is saved as a CSV with shape \n    <code style=\"background:#fcd34d; color:#1f2937; padding: 2px 6px; border-radius: 6px;\">(1100, 283)</code>, \n    matching the competition requirements.\n  </p>\n\n  <p style=\"font-size: 1.05rem;\">\n    This file is now ready for upload to the competition leaderboard.\n  </p>\n\n</div>\n","metadata":{"tags":[]}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport pickle\n\n# 1. Load required files\ntest_adc_info = pd.read_csv('/kaggle/input/ariel-data-challenge-2025/test_star_info.csv', index_col='planet_id')\nsample_submission = pd.read_csv('/kaggle/input/ariel-data-challenge-2025/sample_submission.csv', index_col='planet_id')\nwavelengths = pd.read_csv('/kaggle/input/ariel-data-challenge-2025/wavelengths.csv')\n\n# 2. Load model and sigma\nwith open('model.pickle', 'rb') as f:\n    model = pickle.load(f)\n\nwith open('sigma_pred.pickle', 'rb') as f:\n    sigma_pred = pickle.load(f)\n\n# 3. Run your preprocessing + feature extraction on test set\n# These must be implemented in your own code — you can adapt from training logic\nf_raw_test = f_read_and_preprocess('test', test_adc_info, sample_submission.index)\na_raw_test = a_read_and_preprocess('test', test_adc_info, sample_submission.index)\ntest_features = feature_engineering(f_raw_test, a_raw_test)\n\n# 4. Predict\ntest_pred = model.predict(test_features)\n\n# 5. Postprocessing\ndef postprocessing(pred_array, index, sigma_pred, column_names):\n    \"\"\"\n    Convert predictions and uncertainty into final submission DataFrame.\n    \"\"\"\n    if np.isscalar(sigma_pred):\n        sigma_array = np.full_like(pred_array, sigma_pred)\n    else:\n        sigma_array = sigma_pred\n    df_pred = pd.DataFrame(pred_array.clip(0, None), index=index, columns=column_names)\n    df_sigma = pd.DataFrame(sigma_array, index=index, columns=[f\"sigma_{i}\" for i in range(1, len(column_names)+1)])\n    return pd.concat([df_pred, df_sigma], axis=1)\n\nsubmission_df = postprocessing(\n    pred_array=test_pred,\n    index=sample_submission.index,\n    sigma_pred=sigma_pred,\n    column_names=wavelengths.columns\n)\n\n# 6. Save\nsubmission_df.to_csv('submission.csv')\n\n# 7. Preview\nprint(submission_df.head())\n","metadata":{"tags":[],"trusted":true},"outputs":[],"execution_count":null}]}