{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":101849,"databundleVersionId":13093295,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('kaggle/input/neurlips-ariel-data-challenge-2025'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-08-07T13:42:11.003873Z","iopub.execute_input":"2025-08-07T13:42:11.004268Z","iopub.status.idle":"2025-08-07T13:42:11.414192Z","shell.execute_reply.started":"2025-08-07T13:42:11.004235Z","shell.execute_reply":"2025-08-07T13:42:11.413208Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# A look at the data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-07T13:42:11.415702Z","iopub.execute_input":"2025-08-07T13:42:11.4167Z","iopub.status.idle":"2025-08-07T13:42:11.420265Z","shell.execute_reply.started":"2025-08-07T13:42:11.416665Z","shell.execute_reply":"2025-08-07T13:42:11.419386Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nimport polars as pl\nimport matplotlib.pyplot as plt\n\nimport seaborn as sns\nimport scipy.stats\nfrom tqdm import tqdm\nimport pickle\nimport IPython\n\nfrom sklearn.model_selection import cross_val_predict\nfrom sklearn.linear_model import Ridge\nfrom sklearn.metrics import r2_score, mean_squared_error\nfrom sklearn.cross_decomposition import PLSRegression\nfrom sklearn.model_selection import GridSearchCV, cross_val_predict\n\n\nfrom sklearn.metrics import mean_squared_error\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-07T13:42:11.421166Z","iopub.execute_input":"2025-08-07T13:42:11.421507Z","iopub.status.idle":"2025-08-07T13:42:13.435287Z","shell.execute_reply.started":"2025-08-07T13:42:11.421485Z","shell.execute_reply":"2025-08-07T13:42:13.434369Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nfrom glob import glob\nfrom collections import defaultdict\nimport numpy as np\n\n# --- Load CSVs ---\ntrain_df = pd.read_csv('/kaggle/input/ariel-data-challenge-2025/train.csv')\nwavelengths_df = pd.read_csv('/kaggle/input/ariel-data-challenge-2025/wavelengths.csv')\ntrain_star_info = pd.read_csv('/kaggle/input/ariel-data-challenge-2025/train_star_info.csv')\n\n# --- 1. Basic Train Set Stats ---\nprint(\"🪐 Number of training planets:\", train_df.shape[0])\nprint(\"📈 Number of target labels (wavelengths):\", train_df.shape[1] - 1)\nprint(\"🔬 Length of wavelength grid:\", wavelengths_df.shape[0])\n\n# --- 2. Target Stats (per flux column) ---\ntarget_cols = [col for col in train_df.columns if col != 'planet_id']\nflux_summary = train_df[target_cols].agg(['min', 'max', 'mean', 'std']).T\nprint(\"\\n📊 Flux value summary (first 5 rows):\")\nprint(flux_summary.head())\n\n# --- 3. Unique Stars ---\nif 'planet_id' in train_star_info.columns:\n    num_stars = train_star_info.drop(columns='planet_id').drop_duplicates().shape[0]\nelse:\n    num_stars = train_star_info.drop_duplicates().shape[0]\nprint(\"\\n🌟 Number of unique stars in training:\", num_stars)\n\n# --- 4. Planets with Multiple Observations ---\nobs_counts = defaultdict(int)\ntrain_planets = os.listdir('/kaggle/input/ariel-data-challenge-2025/train')\n\nfor pid in train_planets:\n    air_obs = glob(f\"train/{pid}/AIRS-CH0_signal_*.parquet\")\n    obs_counts[pid] = len(air_obs)\n\nmulti_obs = {pid: count for pid, count in obs_counts.items() if count > 1}\nprint(\"\\n🔁 Planets with multiple observations:\", len(multi_obs))\n\n# --- 5. Check Calibration File Coverage ---\nmissing_calibs = []\nexpected = {\"dark\", \"dead\", \"flat\", \"linear_corr\", \"read\"}\n\nfor pid in train_planets:\n    for band in [\"AIRS-CH0\", \"FGS1\"]:\n        calib_path = f\"train/{pid}/{band}_calibration\"\n        calib_files = {os.path.splitext(f)[0] for f in os.listdir(calib_path)} if os.path.exists(calib_path) else set()\n        missing = expected - calib_files\n        if missing:\n            missing_calibs.append((pid, band, missing))\n\nprint(\"\\n🧪 Planets missing calibration files:\", len(missing_calibs))\nif missing_calibs:\n    print(\"   Example:\", missing_calibs[0])\n\n# --- 6. Optional: Distribution of Observations Per Planet ---\nobs_distribution = pd.Series(list(obs_counts.values())).value_counts().sort_index()\nprint(\"\\n🗂 Observation count distribution per planet (AIR-CH0):\")\nprint(obs_distribution)\n\n# --- 7. Planet-Star Uniqueness Check ---\nmerged = pd.merge(train_df[['planet_id']], train_star_info, on='planet_id', how='left')\nunique_links = merged[['planet_id'] + [col for col in train_star_info.columns if col != 'planet_id']].drop_duplicates()\nprint(\"\\n🔗 Unique planet-star mappings:\", unique_links.shape[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-07T13:42:13.436987Z","iopub.execute_input":"2025-08-07T13:42:13.437516Z","iopub.status.idle":"2025-08-07T13:42:13.95395Z","shell.execute_reply.started":"2025-08-07T13:42:13.43749Z","shell.execute_reply":"2025-08-07T13:42:13.952993Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Key Dataset Insights ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-07T13:42:13.954868Z","iopub.execute_input":"2025-08-07T13:42:13.955215Z","iopub.status.idle":"2025-08-07T13:42:13.959501Z","shell.execute_reply.started":"2025-08-07T13:42:13.955186Z","shell.execute_reply":"2025-08-07T13:42:13.958665Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_adc_info = pd.read_csv('/kaggle/input/ariel-data-challenge-2025/adc_info.csv')\n# test_adc_info = pd.read_csv('/kaggle/input/ariel-data-challenge-2024/test_adc_info.csv',\n#                            index_col='planet_id')\ntrain_labels = pd.read_csv('/kaggle/input/ariel-data-challenge-2025/train.csv',\n                           index_col='planet_id')\nwavelengths = pd.read_csv('/kaggle/input/ariel-data-challenge-2025/wavelengths.csv')\naxis_info = pd.read_parquet('/kaggle/input/ariel-data-challenge-2025/axis_info.parquet')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-07T13:42:13.960953Z","iopub.execute_input":"2025-08-07T13:42:13.961293Z","iopub.status.idle":"2025-08-07T13:42:14.307369Z","shell.execute_reply.started":"2025-08-07T13:42:13.961269Z","shell.execute_reply":"2025-08-07T13:42:14.306386Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# The FGS1 data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-07T13:42:14.308486Z","iopub.execute_input":"2025-08-07T13:42:14.308827Z","iopub.status.idle":"2025-08-07T13:42:14.312891Z","shell.execute_reply.started":"2025-08-07T13:42:14.308798Z","shell.execute_reply":"2025-08-07T13:42:14.311862Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"planet_id = 1010375142\nf_signal = pd.read_parquet(f'/kaggle/input/ariel-data-challenge-2025/train/{planet_id}/FGS1_signal_0.parquet')\nf_signal","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-07T13:42:14.313951Z","iopub.execute_input":"2025-08-07T13:42:14.314667Z","iopub.status.idle":"2025-08-07T13:42:16.322245Z","shell.execute_reply.started":"2025-08-07T13:42:14.314636Z","shell.execute_reply":"2025-08-07T13:42:16.321422Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"_, (ax1, ax2) = plt.subplots(1, 2, figsize=(12, 4))\nsns.heatmap(f_signal.iloc[0].values.reshape(32, 32), ax=ax1, vmin=0, vmax=52000)\nax1.set_aspect('equal')\nsns.heatmap(f_signal.iloc[1].values.reshape(32, 32), ax=ax2, vmin=0, vmax=52000)\nax2.set_aspect('equal')\nplt.suptitle('A pair of FGS1 images')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-07T13:42:16.323054Z","iopub.execute_input":"2025-08-07T13:42:16.323363Z","iopub.status.idle":"2025-08-07T13:42:17.06859Z","shell.execute_reply.started":"2025-08-07T13:42:16.323338Z","shell.execute_reply":"2025-08-07T13:42:17.067681Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# To see the time series, \n# we first have to compute the difference between the even \n# and the odd frames to get the net signal (67500 time steps). \n# We then take the mean over all 1024 pixels. \n# The net signal is very noisy, and we smoothen it by computing a moving average. \n# The plot of the smoothened signal clearly shows that the signal intensity is reduced \n# (i.e., the image gets darker) while the planet passes in front of the star (between time steps 23500 and 44000).\n\n# The left diagram shows a planet with a strong reduction of the signal intensity,\n# the right diagram shows a planet with a weak reduction:","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-07T13:42:17.071282Z","iopub.execute_input":"2025-08-07T13:42:17.071525Z","iopub.status.idle":"2025-08-07T13:42:17.075259Z","shell.execute_reply.started":"2025-08-07T13:42:17.071506Z","shell.execute_reply":"2025-08-07T13:42:17.074432Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"_, ((ax1, ax2), (ax3, ax4)) = plt.subplots(2, 2, sharex=True, figsize=(12, 4))\n\nplanet_id = 1048114509\nf_signal = pd.read_parquet(f'/kaggle/input/ariel-data-challenge-2025/train/{planet_id}/FGS1_signal_0.parquet')\n\nmean_signal = f_signal.values.mean(axis=1)\nnet_signal = mean_signal[1::2] - mean_signal[0::2]\ncum_signal = net_signal.cumsum()\nwindow=800\nsmooth_signal = (cum_signal[window:] - cum_signal[:-window]) / window\n\nax1.set_title('FGS1: time series of planet with strong signal')\nax1.plot(net_signal, label='raw signal')\nax1.legend()\nax3.plot(smooth_signal, color='c', label='smoothened signal')\nax3.legend()\nax3.set_xlabel('time step')\nfor time_step in [20500, 23500, 44000, 47000]:\n    ax3.axvline(time_step, color='gray')\n\nplanet_id = 1240764363\nf_signal = pd.read_parquet(f'/kaggle/input/ariel-data-challenge-2025/train/{planet_id}/FGS1_signal_0.parquet')\n\nmean_signal = f_signal.values.mean(axis=1)\nnet_signal = mean_signal[1::2] - mean_signal[0::2]\ncum_signal = net_signal.cumsum()\nwindow=800\nsmooth_signal = (cum_signal[window:] - cum_signal[:-window]) / window\n\nax2.set_title('FGS1: time series of planet with weak signal')\nax2.plot(net_signal, label='raw signal')\nax2.legend()\nax4.plot(smooth_signal, color='c', label='smoothened signal')\nax4.legend()\nax4.set_xlabel('time step')\nfor time_step in [20500, 23500, 44000, 47000]:\n    ax4.axvline(time_step, color='gray')\n\n# plt.suptitle('FGS1 time series', y=0.96)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-07T13:42:17.076074Z","iopub.execute_input":"2025-08-07T13:42:17.076338Z","iopub.status.idle":"2025-08-07T13:42:21.829055Z","shell.execute_reply.started":"2025-08-07T13:42:17.076314Z","shell.execute_reply":"2025-08-07T13:42:21.828054Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# The AIRS data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-07T13:42:21.830058Z","iopub.execute_input":"2025-08-07T13:42:21.830405Z","iopub.status.idle":"2025-08-07T13:42:21.834508Z","shell.execute_reply.started":"2025-08-07T13:42:21.830378Z","shell.execute_reply":"2025-08-07T13:42:21.833774Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"planet_id = 1240764363\na_signal = pd.read_parquet(f'/kaggle/input/ariel-data-challenge-2025/train/{planet_id}/AIRS-CH0_signal_0.parquet')\na_signal","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-07T13:42:21.835633Z","iopub.execute_input":"2025-08-07T13:42:21.835892Z","iopub.status.idle":"2025-08-07T13:42:23.961375Z","shell.execute_reply.started":"2025-08-07T13:42:21.835873Z","shell.execute_reply":"2025-08-07T13:42:23.960403Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"a_signal = a_signal.values.reshape(11250, 32, 356)\n\nplt.figure(figsize=(10, 3))\nsns.heatmap(a_signal[1])\nplt.ylabel('spatial dimension')\nplt.xlabel('wavelength dimension')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-07T13:42:23.962418Z","iopub.execute_input":"2025-08-07T13:42:23.962707Z","iopub.status.idle":"2025-08-07T13:42:24.371635Z","shell.execute_reply.started":"2025-08-07T13:42:23.962677Z","shell.execute_reply":"2025-08-07T13:42:24.370723Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# The data again is a time series, \n# and we can see how the star is obscured while the planet is passing in front of it.","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-07T13:42:24.372617Z","iopub.execute_input":"2025-08-07T13:42:24.373326Z","iopub.status.idle":"2025-08-07T13:42:24.376782Z","shell.execute_reply.started":"2025-08-07T13:42:24.3733Z","shell.execute_reply":"2025-08-07T13:42:24.375883Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mean_signal = a_signal.mean(axis=2).mean(axis=1)\nnet_signal = mean_signal[1::2] - mean_signal[0::2]\ncum_signal = net_signal.cumsum()\nwindow=80\nsmooth_signal = (cum_signal[window:] - cum_signal[:-window]) / window\n\n_, (ax1, ax2) = plt.subplots(2, 1, sharex=True)\nax1.plot(net_signal, label='raw net signal')\nax1.legend()\nax2.plot(smooth_signal, color='c', label='smoothened net signal')\nax2.legend()\nax2.set_xlabel('time')\nfor time_step in [20500, 23500, 44000, 47000]:\n    ax2.axvline(time_step * 11250 // 135000, color='gray')\nplt.suptitle('AIRS-CH0 time series', y=0.96)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-07T13:42:24.37777Z","iopub.execute_input":"2025-08-07T13:42:24.378116Z","iopub.status.idle":"2025-08-07T13:42:24.890828Z","shell.execute_reply.started":"2025-08-07T13:42:24.378067Z","shell.execute_reply":"2025-08-07T13:42:24.889874Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Reading the data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-07T13:42:24.891753Z","iopub.execute_input":"2025-08-07T13:42:24.892003Z","iopub.status.idle":"2025-08-07T13:42:24.895988Z","shell.execute_reply.started":"2025-08-07T13:42:24.891981Z","shell.execute_reply":"2025-08-07T13:42:24.895148Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile f_read_and_preprocess.py\n\ndef f_read_and_preprocess(dataset, adc_info, planet_ids):\n    \"\"\"Read the FGS1 files for all planet_ids and extract the time series.\n    \n    Parameters\n    dataset: 'train' or 'test'\n    adc_info: metadata dataframe, either train_adc_info or test_adc_info\n    planet_ids: list of planet ids\n    \n    Returns\n    dataframe with one row per planet_id and 67500 values per row\n    \n    \"\"\"\n    f_raw_train = np.full((len(planet_ids), 67500), np.nan, dtype=np.float32)\n    for i, planet_id in tqdm(list(enumerate(planet_ids))):\n        f_signal = pl.read_parquet(f'/kaggle/input/ariel-data-challenge-2025/{dataset}/{planet_id}/FGS1_signal_0.parquet')\n        mean_signal = f_signal.cast(pl.Int32).sum_horizontal().cast(pl.Float32).to_numpy() / 1024 # mean over the 32*32 pixels\n        net_signal = mean_signal[1::2] - mean_signal[0::2]\n        f_raw_train[i] = net_signal\n    return f_raw_train\n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-07T13:42:24.896917Z","iopub.execute_input":"2025-08-07T13:42:24.89732Z","iopub.status.idle":"2025-08-07T13:42:24.915336Z","shell.execute_reply.started":"2025-08-07T13:42:24.897298Z","shell.execute_reply":"2025-08-07T13:42:24.914154Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nexec(open('f_read_and_preprocess.py', 'r').read())\nf_raw_train = f_read_and_preprocess('train', train_adc_info, train_labels.index)\nwith open('f_raw_train.pickle', 'wb') as f:\n    pickle.dump(f_raw_train, f)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-07T13:42:24.916331Z","iopub.execute_input":"2025-08-07T13:42:24.916646Z","iopub.status.idle":"2025-08-07T13:57:35.442746Z","shell.execute_reply.started":"2025-08-07T13:42:24.916619Z","shell.execute_reply":"2025-08-07T13:57:35.441323Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile a_read_and_preprocess.py\n\ndef a_read_and_preprocess(dataset, adc_info, planet_ids):\n    \"\"\"Read the AIRS-CH0 files for all planet_ids and extract the time series.\n\n    Parameters\n    dataset: 'train' or 'test'\n    adc_info: metadata dataframe, either train_adc_info or test_adc_info\n    planet_ids: list of planet ids\n\n    Returns\n    ndarray with shape (n_planets, 5625)\n    \"\"\"\n    a_raw_train = np.full((len(planet_ids), 5625), np.nan, dtype=np.float32)\n\n    for i, planet_id in tqdm(list(enumerate(planet_ids)), desc=\"Processing AIRS-CH0\"):\n        # Read as LazyFrame\n        signal = pl.read_parquet(\n            f'/kaggle/input/ariel-data-challenge-2025/{dataset}/{planet_id}/AIRS-CH0_signal_0.parquet',\n            use_pyarrow=True\n        ).lazy()\n\n        # Compute mean signal over each row\n        mean_signal = (\n            signal\n            .select(\n                (pl.sum_horizontal(pl.all().cast(pl.Int32)) / (32 * 356)).alias(\"mean\")\n            )\n            .collect()\n            .get_column(\"mean\")\n            .to_numpy()\n            .astype(np.float32)\n        ).flatten()\n\n        # Tính net_signal: hiệu giữa cặp chẵn/lẻ\n        net_signal = mean_signal[1::2] - mean_signal[0::2]\n        a_raw_train[i] = net_signal\n\n    return a_raw_train\n\n\n\n\n\n\n\n\n# %%writefile a_read_and_preprocess.py\n# def a_read_and_preprocess(dataset, adc_info, planet_ids):\n#     \"\"\"Read the AIRS-CH0 files for all planet_ids and extract the time series.\n    \n#     Parameters\n#     dataset: 'train' or 'test'\n#     adc_info: metadata dataframe, either train_adc_info or test_adc_info\n#     planet_ids: list of planet ids\n    \n#     Returns\n#     dataframe with one row per planet_id and 5625 values per row\n    \n#     \"\"\"\n#     a_raw_train = np.full((len(planet_ids), 5625), np.nan, dtype=np.float32)\n#     for i, planet_id in tqdm(list(enumerate(planet_ids))):\n#         signal = pl.read_parquet(f'/kaggle/input/ariel-data-challenge-2025/{dataset}/{planet_id}/AIRS-CH0_signal_0.parquet')\n#         mean_signal = signal.cast(pl.Int32).sum_horizontal().cast(pl.Float32).to_numpy() / (32*356) # mean over the 32*356 pixels\n#         net_signal = mean_signal[1::2] - mean_signal[0::2]\n#         a_raw_train[i] = net_signal\n#     return a_raw_train\n\n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-07T13:57:35.444838Z","iopub.execute_input":"2025-08-07T13:57:35.4453Z","iopub.status.idle":"2025-08-07T13:57:35.457708Z","shell.execute_reply.started":"2025-08-07T13:57:35.445266Z","shell.execute_reply":"2025-08-07T13:57:35.456781Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nexec(open('a_read_and_preprocess.py', 'r').read())\na_raw_train = a_read_and_preprocess('train', train_adc_info, train_labels.index)\nwith open('a_raw_train.pickle', 'wb') as f:\n    pickle.dump(a_raw_train, f)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-07T13:57:35.459044Z","iopub.execute_input":"2025-08-07T13:57:35.459371Z","iopub.status.idle":"2025-08-07T14:45:24.432691Z","shell.execute_reply.started":"2025-08-07T13:57:35.459349Z","shell.execute_reply":"2025-08-07T14:45:24.431624Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(6, 2))\nplt.plot(f_raw_train.mean(axis=0))\nfor time_step in [20500, 23500, 44000, 47000]:\n    plt.axvline(time_step, color='gray')\nplt.xlabel('time step')\nplt.title('FGS1: Overall mean')\nplt.show()\n\nplt.figure(figsize=(6, 2))\nplt.plot(a_raw_train.mean(axis=0))\nfor time_step in [20500, 23500, 44000, 47000]:\n    plt.axvline(time_step * 11250 // 135000, color='gray')\nplt.xlabel('time step')\nplt.title('AIRS-CH0: Overall mean')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-07T14:45:24.434037Z","iopub.execute_input":"2025-08-07T14:45:24.434409Z","iopub.status.idle":"2025-08-07T14:45:24.862169Z","shell.execute_reply.started":"2025-08-07T14:45:24.434379Z","shell.execute_reply":"2025-08-07T14:45:24.861196Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"    # Feature engineering","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-07T14:45:24.863163Z","iopub.execute_input":"2025-08-07T14:45:24.863484Z","iopub.status.idle":"2025-08-07T14:45:24.867937Z","shell.execute_reply.started":"2025-08-07T14:45:24.863462Z","shell.execute_reply":"2025-08-07T14:45:24.867021Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile feature_engineering.py\n\ndef feature_engineering(f_raw, a_raw):\n    \"\"\"Create a dataframe with two features from the raw data.\n    \n    Parameters:\n    f_raw: ndarray of shape (n_planets, 67500)\n    a_raw: ndarray of shape (n_planets, 5625)\n    \n    Return value:\n    df: DataFrame of shape (n_planets, 2)\n    \"\"\"\n    obscured = f_raw[:, 23500:44000].mean(axis=1)\n    unobscured = (f_raw[:, :20500].mean(axis=1) + f_raw[:, 47000:].mean(axis=1)) / 2\n    f_relative_reduction = (unobscured - obscured) / unobscured\n    obscured = a_raw[:, 1958:3666].mean(axis=1)\n    unobscured = (a_raw[:, :1708].mean(axis=1) + a_raw[:, 3916:].mean(axis=1)) / 2\n    a_relative_reduction = (unobscured - obscured) / unobscured\n\n    df = pd.DataFrame({'a_relative_reduction': a_relative_reduction,\n                       'f_relative_reduction': f_relative_reduction})\n    \n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-07T14:45:24.869447Z","iopub.execute_input":"2025-08-07T14:45:24.870249Z","iopub.status.idle":"2025-08-07T14:45:24.890919Z","shell.execute_reply.started":"2025-08-07T14:45:24.870221Z","shell.execute_reply":"2025-08-07T14:45:24.889911Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"exec(open('feature_engineering.py', 'r').read())\n\ntrain = feature_engineering(f_raw_train, a_raw_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-07T14:45:24.892301Z","iopub.execute_input":"2025-08-07T14:45:24.892674Z","iopub.status.idle":"2025-08-07T14:45:24.953078Z","shell.execute_reply.started":"2025-08-07T14:45:24.89263Z","shell.execute_reply":"2025-08-07T14:45:24.95205Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# The model and the cross-validation","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-07T14:45:24.954184Z","iopub.execute_input":"2025-08-07T14:45:24.954506Z","iopub.status.idle":"2025-08-07T14:45:24.958434Z","shell.execute_reply.started":"2025-08-07T14:45:24.954478Z","shell.execute_reply":"2025-08-07T14:45:24.957694Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# model = Ridge(alpha=1e-12)\n\n# oof_pred = cross_val_predict(model, train, train_labels)\n\n# print(f\"# R2 score: {r2_score(train_labels, oof_pred):.3f}\")\n\n\n# sigma_pred = mean_squared_error(train_labels, oof_pred, squared=False)\n\n\n# sigma_pred = np.sqrt(mean_squared_error(train_labels, oof_pred))\n# print(f\"# Root mean squared error: {sigma_pred:.6f}\")\n\n\n# R2 score: 0.971\n# Root mean squared error: 0.000293\nfrom sklearn.cross_decomposition import PLSRegression\nfrom sklearn.model_selection import cross_val_predict\nfrom sklearn.metrics import r2_score, mean_squared_error\nimport numpy as np\n\nmodel = PLSRegression(n_components=2)\n\noof_pred = cross_val_predict(model, train, train_labels)\n\nprint(f\"# R2 score: {r2_score(train_labels, oof_pred):.3f}\")\nsigma_pred = np.sqrt(mean_squared_error(train_labels, oof_pred))\nprint(f\"# Root mean squared error: {sigma_pred:.6f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-07T14:45:24.959409Z","iopub.execute_input":"2025-08-07T14:45:24.959684Z","iopub.status.idle":"2025-08-07T14:45:25.21073Z","shell.execute_reply.started":"2025-08-07T14:45:24.959664Z","shell.execute_reply":"2025-08-07T14:45:25.209819Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"col = 1\nplt.scatter(oof_pred[:,col], train_labels.iloc[:,col], s=15, c='lightgreen')\nplt.gca().set_aspect('equal')\nplt.xlabel('y_pred')\nplt.ylabel('y_true')\nplt.title('Comparing y_true and y_pred')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-07T14:45:25.214993Z","iopub.execute_input":"2025-08-07T14:45:25.215377Z","iopub.status.idle":"2025-08-07T14:45:25.462359Z","shell.execute_reply.started":"2025-08-07T14:45:25.215353Z","shell.execute_reply":"2025-08-07T14:45:25.46148Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile competition_score.py\n# Adapted from https://www.kaggle.com/code/metric/ariel-gaussian-log-likelihood\n\n# Custom error for invalid submissions\nclass ParticipantVisibleError(Exception):\n    pass\n\ndef competition_score(\n    solution: pd.DataFrame,\n    submission: pd.DataFrame,\n    naive_mean: float,\n    naive_sigma: float,\n    sigma_true: float,\n    row_id_column_name='planet_id'\n) -> float:\n    '''\n    Computes a Gaussian Log Likelihood-based score.\n    '''\n    # Drop ID columns\n    solution = solution.drop(columns=[row_id_column_name], errors='ignore')\n    submission = submission.drop(columns=[row_id_column_name], errors='ignore')\n\n    # Validation checks\n    if submission.min().min() < 0:\n        raise ParticipantVisibleError('Negative values in the submission')\n\n    for col in submission.columns:\n        if not pd.api.types.is_numeric_dtype(submission[col]):\n            raise ParticipantVisibleError(f'Submission column {col} must be numeric: {col}')\n\n    n_wavelengths = len(solution.columns)\n    if len(submission.columns) != 2 * n_wavelengths:\n        raise ParticipantVisibleError('Submission must have 2x columns of the solution')\n\n    # Extract predictions and sigmas\n    y_pred = submission.iloc[:, :n_wavelengths].values\n    sigma_pred = np.clip(submission.iloc[:, n_wavelengths:].values, a_min=1e-15, a_max=None)\n    y_true = solution.values\n\n    # Compute log likelihoods\n    GLL_pred = np.sum(scipy.stats.norm.logpdf(y_true, loc=y_pred, scale=sigma_pred))\n    GLL_true = np.sum(scipy.stats.norm.logpdf(y_true, loc=y_true, scale=sigma_true))\n    GLL_mean = np.sum(scipy.stats.norm.logpdf(y_true, loc=naive_mean, scale=naive_sigma))\n\n    score = (GLL_pred - GLL_mean) / (GLL_true - GLL_mean)\n    return float(np.clip(score, 0.0, 1.0))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-07T14:45:25.463422Z","iopub.execute_input":"2025-08-07T14:45:25.463693Z","iopub.status.idle":"2025-08-07T14:45:25.470576Z","shell.execute_reply.started":"2025-08-07T14:45:25.463669Z","shell.execute_reply":"2025-08-07T14:45:25.469199Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def postprocessing(pred_array, index, sigma_pred, column_names=None):\n    \"\"\"\n    Creates a submission DataFrame with mean predictions and uncertainties.\n\n    Parameters:\n    - pred_array: ndarray of shape (n_samples, 283)\n    - index: pandas.Index of length n_samples\n    - sigma_pred: float or ndarray of shape (n_samples, 283)\n    - column_names: list of wavelength column names (optional)\n\n    Returns:\n    - df: DataFrame of shape (n_samples, 566)\n    \"\"\"\n    n_samples, n_waves = pred_array.shape\n\n    if column_names is None:\n        column_names = [f\"wl_{i+1}\" for i in range(n_waves)]\n\n    if np.isscalar(sigma_pred):\n        sigma_pred = np.full_like(pred_array, sigma_pred)\n\n    # Safety check\n    assert sigma_pred.shape == pred_array.shape, \"sigma_pred must match shape of pred_array\"\n    assert len(index) == n_samples, \"Index length must match number of rows\"\n\n    df_mean = pd.DataFrame(pred_array.clip(0, None), index=index, columns=column_names)\n    df_sigma = pd.DataFrame(sigma_pred, index=index, columns=[f\"sigma_{i+1}\" for i in range(n_waves)])\n\n    return pd.concat([df_mean, df_sigma], axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-07T14:45:25.471747Z","iopub.execute_input":"2025-08-07T14:45:25.472065Z","iopub.status.idle":"2025-08-07T14:45:25.493997Z","shell.execute_reply.started":"2025-08-07T14:45:25.472043Z","shell.execute_reply":"2025-08-07T14:45:25.493074Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"exec(open('competition_score.py', 'r').read())\n#exec(open('postprocessing.py', 'r').read())\n\noof_df = postprocessing(oof_pred, train_labels.index, sigma_pred)\ndisplay(oof_df)\n\ngll_score = competition_score(train_labels.copy().reset_index(),\n                              oof_df.copy().reset_index(),\n                              naive_mean=train_labels.values.mean(),\n                              naive_sigma=train_labels.values.std(),\n                              sigma_true=0.000003)\nprint(f\"# Estimated competition score: {gll_score:.3f}\")\n# Estimated competition score: 0.123","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-07T14:45:25.495046Z","iopub.execute_input":"2025-08-07T14:45:25.495367Z","iopub.status.idle":"2025-08-07T14:45:25.630741Z","shell.execute_reply.started":"2025-08-07T14:45:25.495343Z","shell.execute_reply":"2025-08-07T14:45:25.629963Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Refitting and saving the model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-07T14:45:25.631804Z","iopub.execute_input":"2025-08-07T14:45:25.632143Z","iopub.status.idle":"2025-08-07T14:45:25.636253Z","shell.execute_reply.started":"2025-08-07T14:45:25.632113Z","shell.execute_reply":"2025-08-07T14:45:25.635257Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Refit the model to the full dataset\nmodel.fit(train, train_labels)\nwith open('model.pickle', 'wb') as f:\n    pickle.dump(model, f)\nwith open('sigma_pred.pickle', 'wb') as f:\n    pickle.dump(sigma_pred, f)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-07T14:45:25.636995Z","iopub.execute_input":"2025-08-07T14:45:25.637307Z","iopub.status.idle":"2025-08-07T14:45:25.678397Z","shell.execute_reply.started":"2025-08-07T14:45:25.637281Z","shell.execute_reply":"2025-08-07T14:45:25.677432Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-07T14:45:25.679387Z","iopub.execute_input":"2025-08-07T14:45:25.679706Z","iopub.status.idle":"2025-08-07T14:45:25.685345Z","shell.execute_reply.started":"2025-08-07T14:45:25.679676Z","shell.execute_reply":"2025-08-07T14:45:25.684276Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport pickle\n\n# 1. Load required files\ntest_adc_info = pd.read_csv('/kaggle/input/ariel-data-challenge-2025/test_star_info.csv', index_col='planet_id')\nsample_submission = pd.read_csv('/kaggle/input/ariel-data-challenge-2025/sample_submission.csv', index_col='planet_id')\nwavelengths = pd.read_csv('/kaggle/input/ariel-data-challenge-2025/wavelengths.csv')\n\n# 2. Load model and sigma\nwith open('model.pickle', 'rb') as f:\n    model = pickle.load(f)\n\nwith open('sigma_pred.pickle', 'rb') as f:\n    sigma_pred = pickle.load(f)\n\n# 3. Run your preprocessing + feature extraction on test set\n# These must be implemented in your own code — you can adapt from training logic\nf_raw_test = f_read_and_preprocess('test', test_adc_info, sample_submission.index)\na_raw_test = a_read_and_preprocess('test', test_adc_info, sample_submission.index)\ntest_features = feature_engineering(f_raw_test, a_raw_test)\n\n# 4. Predict\ntest_pred = model.predict(test_features)\n\n# 5. Postprocessing\ndef postprocessing(pred_array, index, sigma_pred, column_names):\n    \"\"\"\n    Convert predictions and uncertainty into final submission DataFrame.\n    \"\"\"\n    if np.isscalar(sigma_pred):\n        sigma_array = np.full_like(pred_array, sigma_pred)\n    else:\n        sigma_array = sigma_pred\n    df_pred = pd.DataFrame(pred_array.clip(0, None), index=index, columns=column_names)\n    df_sigma = pd.DataFrame(sigma_array, index=index, columns=[f\"sigma_{i}\" for i in range(1, len(column_names)+1)])\n    return pd.concat([df_pred, df_sigma], axis=1)\n\nsubmission_df = postprocessing(\n    pred_array=test_pred,\n    index=sample_submission.index,\n    sigma_pred=sigma_pred,\n    column_names=wavelengths.columns\n)\n\n# 6. Save\nsubmission_df.to_csv('submission.csv')\n\n# 7. Preview\n!head submission.csv","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-07T14:45:25.68639Z","iopub.execute_input":"2025-08-07T14:45:25.686686Z","iopub.status.idle":"2025-08-07T14:45:30.011879Z","shell.execute_reply.started":"2025-08-07T14:45:25.686662Z","shell.execute_reply":"2025-08-07T14:45:30.01074Z"}},"outputs":[],"execution_count":null}]}