{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"papermill":{"default_parameters":{},"duration":484.304573,"end_time":"2023-08-17T16:29:30.442356","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2023-08-17T16:21:26.137783","version":"2.4.0"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":101849,"databundleVersionId":13093295,"sourceType":"competition"},{"sourceId":13155462,"sourceType":"datasetVersion","datasetId":8335336}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nimport matplotlib.pyplot as plt\nimport pickle","metadata":{"_kg_hide-output":true,"papermill":{"duration":13.193202,"end_time":"2023-08-17T16:21:53.762495","exception":false,"start_time":"2023-08-17T16:21:40.569293","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2025-09-26T05:52:35.218664Z","iopub.execute_input":"2025-09-26T05:52:35.218967Z","iopub.status.idle":"2025-09-26T05:52:35.541219Z","shell.execute_reply.started":"2025-09-26T05:52:35.218943Z","shell.execute_reply":"2025-09-26T05:52:35.540344Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## <span style=\"color:grey; font-size:80%;\">get raw data</span>","metadata":{}},{"cell_type":"markdown","source":"Thankyou - https://www.kaggle.com/code/ambrosm/adc24-intro-training\n\nFor FGS1 data - Reads data per planet, computes summed pixel values, and stores net signals (odd-even difference) into a (n_planets × 67500) array.\n\nThe same for AIRS data, along normalizing by detector size (32×356) and returning a (n_planets × 5625) array.","metadata":{}},{"cell_type":"code","source":"def load_fgs1_signals(dataset, planet_ids):\n    \"\"\"Return FGS1 net signals for given planet IDs (shape: n_planets × 67500).\"\"\"\n    arr = np.full((len(planet_ids), 67500), np.nan, dtype=np.float32)\n    for i, pid in tqdm(enumerate(planet_ids), total=len(planet_ids)):\n        path = f'/kaggle/input/ariel-data-challenge-2025/{dataset}/{pid}/FGS1_signal_0.parquet'\n        sig = pl.read_parquet(path).cast(pl.Int32).sum_horizontal().to_numpy() / 1024\n        arr[i] = sig[1::2] - sig[0::2]\n    return arr\n\ndef load_airs_signals(dataset, planet_ids):\n    \"\"\"Return AIRS-CH0 net signals for given planet IDs (shape: n_planets × 5625).\"\"\"\n    arr = np.full((len(planet_ids), 5625), np.nan, dtype=np.float32)\n    for i, pid in tqdm(enumerate(planet_ids), total=len(planet_ids)):\n        path = f'/kaggle/input/ariel-data-challenge-2025/{dataset}/{pid}/AIRS-CH0_signal_0.parquet'\n        sig = pl.read_parquet(path).cast(pl.Int32).sum_horizontal().to_numpy() / (32 * 356)\n        arr[i] = sig[1::2] - sig[0::2]\n    return arr\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-26T05:52:35.542893Z","iopub.execute_input":"2025-09-26T05:52:35.543274Z","iopub.status.idle":"2025-09-26T05:52:35.551896Z","shell.execute_reply.started":"2025-09-26T05:52:35.543252Z","shell.execute_reply":"2025-09-26T05:52:35.550592Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"file_path = \"/kaggle/input/nips-25-ariel-raw-data/nd_a_raw_train.pickle\"\n\nwith open(file_path, \"rb\") as f:\n    a_raw = pickle.load(f)\n\nfile_path = \"/kaggle/input/nips-25-ariel-raw-data/nd_f_raw_train.pickle\"\n\nwith open(file_path, \"rb\") as f:\n    f_raw = pickle.load(f)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-26T05:52:35.552774Z","iopub.execute_input":"2025-09-26T05:52:35.55308Z","iopub.status.idle":"2025-09-26T05:52:38.783519Z","shell.execute_reply.started":"2025-09-26T05:52:35.55305Z","shell.execute_reply":"2025-09-26T05:52:38.782613Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## <span style=\"color:grey; font-size:80%;\">winsorize</span>","metadata":{}},{"cell_type":"code","source":"def winsorize_array(arr, lower=0.0001, upper=0.0001, axis=1):\n    \"\"\"\n    Winsorize a 2D array along the specified axis.\n    \"\"\"\n    q_low = np.quantile(arr, lower, axis=axis, keepdims=True)\n    q_high = np.quantile(arr, 1 - upper, axis=axis, keepdims=True)\n    return np.clip(arr, q_low, q_high)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-26T05:52:38.784388Z","iopub.execute_input":"2025-09-26T05:52:38.784619Z","iopub.status.idle":"2025-09-26T05:52:38.789635Z","shell.execute_reply.started":"2025-09-26T05:52:38.7846Z","shell.execute_reply":"2025-09-26T05:52:38.788833Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"f_data = winsorize_array(f_raw,lower=0.0001, upper=0.0001)\na_data = winsorize_array(a_raw,lower=0.01, upper=0.01)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-26T05:52:38.792356Z","iopub.execute_input":"2025-09-26T05:52:38.792689Z","iopub.status.idle":"2025-09-26T05:52:41.534689Z","shell.execute_reply.started":"2025-09-26T05:52:38.792657Z","shell.execute_reply":"2025-09-26T05:52:41.53385Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## <span style=\"color:grey; font-size:80%;\">generate features/visualize</span>","metadata":{}},{"cell_type":"markdown","source":"The function computes Relative Reduction (RR) features from both FGS1 and AIRS signals.Input arrays are split into equal-sized segments, where each segment represents a portion of the signal across time/wavelengths.For each segment, the mean signal is compared against the mean of the \"unobscured\" parts (everything outside the segment).\n\nOnly the middle segments are used to calculate features, while the first and last segments are skipped.rr_dip → relative reduction when the segment mean is lower than outside mean (dip strength).rr_dome → relative increase when the segment mean is higher than outside mean (dome strength).","metadata":{}},{"cell_type":"code","source":"def generate_rr_features_multi_segments(f_raw, a_raw, segment_list=[50]):\n    import numpy as np\n    import pandas as pd\n\n    def compute_features(array, name_prefix, n_segments):\n        n_planets, n_cols = array.shape\n        segment_size = n_cols // n_segments  # integer segments; remainder naturally sits in last segment (which we skip)\n\n        # Only build feature names for middle segments (skip 1st and last)\n        features = {\n            f\"{name_prefix}_{stat}_{n_segments}_{i+1}\": []\n            for stat in ['rr_dip', 'rr_dome']\n            for i in range(1, n_segments - 1)  # skip first (0) and last (n_segments-1)\n        }\n\n        for i in range(1, n_segments - 1):  # loop only middle segments\n            start = i * segment_size\n            end = (i + 1) * segment_size if i < n_segments - 1 else n_cols\n\n            if end > start:\n                segment = array[:, start:end]\n                segment_mean = segment.mean(axis=1)\n\n                if start > 0 and end < n_cols:\n                    left = array[:, :start]\n                    right = array[:, end:]\n                    unobscured_mean = (left.mean(axis=1) + right.mean(axis=1)) / 2\n                elif start == 0:\n                    right = array[:, end:]\n                    unobscured_mean = right.mean(axis=1)\n                else:\n                    left = array[:, :start]\n                    unobscured_mean = left.mean(axis=1)\n\n                with np.errstate(divide='ignore', invalid='ignore'):\n                    rr_dip = np.where(unobscured_mean != 0,\n                                  (segment_mean - unobscured_mean) / unobscured_mean, 0.0)\n                    rr_dome = np.where(unobscured_mean != 0,\n                                       (unobscured_mean - segment_mean) / unobscured_mean, 0.0)\n\n                # vectorized append\n                features[f\"{name_prefix}_rr_dip_{n_segments}_{i+1}\"].extend(rr_dip.tolist())\n                features[f\"{name_prefix}_rr_dome_{n_segments}_{i+1}\"].extend(rr_dome.tolist())\n            else:\n                features[f\"{name_prefix}_rr_dip_{n_segments}_{i+1}\"].extend([0.0] * n_planets)\n                features[f\"{name_prefix}_rr_dome_{n_segments}_{i+1}\"].extend([0.0] * n_planets)\n\n        return pd.DataFrame(features)\n\n    dfs = []\n    for n_segments in segment_list:\n        dfs.append(compute_features(f_raw, \"f\", n_segments))\n        dfs.append(compute_features(a_raw, \"a\", n_segments))\n\n    return pd.concat(dfs, axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-26T05:52:41.535323Z","iopub.execute_input":"2025-09-26T05:52:41.535527Z","iopub.status.idle":"2025-09-26T05:52:41.546822Z","shell.execute_reply.started":"2025-09-26T05:52:41.53551Z","shell.execute_reply":"2025-09-26T05:52:41.546129Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_labels = pd.read_csv('/kaggle/input/ariel-data-challenge-2025/train.csv', index_col='planet_id')\nsegment_counts = [75]\ntrain_seg = generate_rr_features_multi_segments(f_data, a_data, segment_list=segment_counts)\ntrain_seg.index = train_labels.index\ntrain_seg","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-26T05:52:41.547942Z","iopub.execute_input":"2025-09-26T05:52:41.548611Z","iopub.status.idle":"2025-09-26T05:52:47.158077Z","shell.execute_reply.started":"2025-09-26T05:52:41.548587Z","shell.execute_reply":"2025-09-26T05:52:47.15726Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## <span style=\"color:grey; font-size:80%;\">segmentation/visualization</span>","metadata":{}},{"cell_type":"code","source":"segment_counts = [75]\ntrain_seg = generate_rr_features_multi_segments(f_data, a_data, segment_list=segment_counts)\ntrain_seg.index = train_labels.index\ntrain_seg","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-26T05:52:47.158891Z","iopub.execute_input":"2025-09-26T05:52:47.159213Z","iopub.status.idle":"2025-09-26T05:52:52.553194Z","shell.execute_reply.started":"2025-09-26T05:52:47.15919Z","shell.execute_reply":"2025-09-26T05:52:52.552344Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"planet_ids = [34983, 1873185,3849793,8456603,905997089,1338107575,1738121950,1783354373]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-26T05:52:52.554112Z","iopub.execute_input":"2025-09-26T05:52:52.554503Z","iopub.status.idle":"2025-09-26T05:52:52.558781Z","shell.execute_reply.started":"2025-09-26T05:52:52.554474Z","shell.execute_reply":"2025-09-26T05:52:52.558056Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nn_rows = 4\nn_cols = 2\n\nfig, axes = plt.subplots(n_rows, n_cols, figsize=(10, 15), sharex=True, sharey=True)\naxes = axes.flatten()\n\nfor i, pid in enumerate(planet_ids):\n    ax = axes[i]\n    if pid not in train_seg.index:\n        continue\n    row = train_seg.loc[pid]\n    rr_cols = sorted(\n        [col for col in row.index if col.startswith(\"a_rr_dip\")],\n        key=lambda x: int(x.split(\"_\")[-1])\n    )\n    rr_vals = row[rr_cols].values\n    x = range(1, len(rr_vals) + 1)\n    ax.plot(x, rr_vals, marker='o', label=f\"Planet {pid}\")\n    ax.set_title(f\"Planet ID {pid}\")\n    ax.grid(True)\n\nfig.suptitle(\"a_rr_dip Profiles\", fontsize=16)\nplt.tight_layout(rect=[0, 0.03, 1, 0.95])\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-26T05:52:52.559648Z","iopub.execute_input":"2025-09-26T05:52:52.559952Z","iopub.status.idle":"2025-09-26T05:52:54.37904Z","shell.execute_reply.started":"2025-09-26T05:52:52.559927Z","shell.execute_reply":"2025-09-26T05:52:54.378162Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"n_rows = 4\nn_cols = 2\n\nfig, axes = plt.subplots(n_rows, n_cols, figsize=(10, 15), sharex=True, sharey=True)\naxes = axes.flatten()\n\nfor i, pid in enumerate(planet_ids):\n    ax = axes[i]\n    if pid not in train_seg.index:\n        continue\n    row = train_seg.loc[pid]\n    rr_cols = sorted(\n        [col for col in row.index if col.startswith(\"f_rr_dip\")],\n        key=lambda x: int(x.split(\"_\")[-1])\n    )\n    rr_vals = row[rr_cols].values\n    x = range(1, len(rr_vals) + 1)\n    ax.plot(x, rr_vals, marker='o', label=f\"Planet {pid}\")\n    ax.set_title(f\"Planet ID {pid}\")\n    ax.grid(True)\n\nfig.suptitle(\"f_rr_dip Profiles\", fontsize=16)\nplt.tight_layout(rect=[0, 0.03, 1, 0.95])\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-26T05:52:54.37997Z","iopub.execute_input":"2025-09-26T05:52:54.380225Z","iopub.status.idle":"2025-09-26T05:52:55.961405Z","shell.execute_reply.started":"2025-09-26T05:52:54.380205Z","shell.execute_reply":"2025-09-26T05:52:55.960545Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"n_rows = 4\nn_cols = 2\n\nfig, axes = plt.subplots(n_rows, n_cols, figsize=(10, 15), sharex=True, sharey=True)\naxes = axes.flatten()\n\nfor i, pid in enumerate(planet_ids):\n    ax = axes[i]\n    if pid not in train_seg.index:\n        continue\n    \n    row = train_seg.loc[pid]\n    \n    # dip features\n    rr_dip_cols = sorted(\n        [col for col in row.index if col.startswith(\"a_rr_dip\")],\n        key=lambda x: int(x.split(\"_\")[-1])\n    )\n    rr_dip_vals = row[rr_dip_cols].values\n    \n    # dome features (same segment numbers)\n    rr_dome_cols = sorted(\n        [col for col in row.index if col.startswith(\"a_rr_dome\")],\n        key=lambda x: int(x.split(\"_\")[-1])\n    )\n    rr_dome_vals = row[rr_dome_cols].values\n    \n    x = range(1, len(rr_dip_vals) + 1)\n    \n    ax.plot(x, rr_dip_vals, marker='o', label=\"Dip\", color=\"tab:blue\")\n    ax.plot(x, rr_dome_vals, marker='s', label=\"Dome\", color=\"tab:orange\")\n    \n    ax.set_title(f\"Planet ID {pid}\")\n    ax.grid(True)\n    ax.legend()\n\nfig.suptitle(\"a_rr_dip vs a_rr_dome Profiles\", fontsize=16)\nplt.tight_layout(rect=[0, 0.03, 1, 0.95])\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-26T05:52:55.962326Z","iopub.execute_input":"2025-09-26T05:52:55.962596Z","iopub.status.idle":"2025-09-26T05:52:57.767398Z","shell.execute_reply.started":"2025-09-26T05:52:55.962576Z","shell.execute_reply":"2025-09-26T05:52:57.766593Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## <span style=\"color:grey; font-size:80%;\">multi-segments/visualization</span>","metadata":{}},{"cell_type":"code","source":"segment_counts = [130, 40, 50]\ntrain_seg = generate_rr_features_multi_segments(f_data, a_data, segment_list=segment_counts)\ntrain_seg.index = train_labels.index\ntrain_seg","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-26T05:52:57.768354Z","iopub.execute_input":"2025-09-26T05:52:57.76896Z","iopub.status.idle":"2025-09-26T05:53:13.438Z","shell.execute_reply.started":"2025-09-26T05:52:57.768929Z","shell.execute_reply":"2025-09-26T05:53:13.437155Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_rr_multi_segments(train_seg, planet_ids, segment_list=(30,40,50), prefix=\"a\", kind=\"both\",\n                           n_rows=4, n_cols=2):\n    \"\"\"\n    kind: \"dip\", \"dome\", or \"both\"\n    prefix: \"a\" for AIRS, \"f\" for FGS1\n    \"\"\"\n    import matplotlib.pyplot as plt\n    import numpy as np\n\n    fig, axes = plt.subplots(n_rows, n_cols, figsize=(11, 16), sharex=True, sharey=True)\n    axes = axes.flatten()\n\n    def _collect(rr_kind, nseg, row):       \n        cols = [c for c in row.index if c.startswith(f\"{prefix}_rr_{rr_kind}_{nseg}_\")]\n        # sort by the trailing segment index\n        cols = sorted(cols, key=lambda x: int(x.split(\"_\")[-1]))\n        return row[cols].values if len(cols) else np.array([])\n\n    for i, pid in enumerate(planet_ids[:n_rows*n_cols]):\n        ax = axes[i]\n        if pid not in train_seg.index:\n            ax.set_title(f\"Planet {pid} (missing)\")\n            ax.axis(\"off\")\n            continue\n\n        row = train_seg.loc[pid]\n        for nseg in segment_list:\n            x = range(1, nseg-1)  \n            if kind in (\"dip\", \"both\"):\n                y_dip = _collect(\"dip\", nseg, row)\n                if y_dip.size:\n                    ax.plot(x, y_dip, marker=\"o\", linestyle=\"-\", label=f\"{nseg} seg · dip\")\n            if kind in (\"dome\", \"both\"):\n                y_dome = _collect(\"dome\", nseg, row)\n                if y_dome.size:\n                    ax.plot(x, y_dome, marker=\"s\", linestyle=\"--\", label=f\"{nseg} seg · dome\")\n\n        ax.set_title(f\"Planet {pid}\")\n        ax.grid(True)\n        ax.set_xlabel(\"Segment index\")\n        ax.set_ylabel(\"RR value\")\n        ax.legend(fontsize=8, ncol=2)\n\n    fig.suptitle(f\"{prefix.upper()} RR profiles across multiple segment counts\", fontsize=16)\n    plt.tight_layout(rect=[0, 0.03, 1, 0.95])\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-26T05:53:13.44056Z","iopub.execute_input":"2025-09-26T05:53:13.441344Z","iopub.status.idle":"2025-09-26T05:53:13.451124Z","shell.execute_reply.started":"2025-09-26T05:53:13.441313Z","shell.execute_reply":"2025-09-26T05:53:13.450258Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_rr_multi_segments(train_seg, planet_ids, segment_list=(30,40,50), prefix=\"a\", kind=\"both\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-26T05:53:13.45189Z","iopub.execute_input":"2025-09-26T05:53:13.452139Z","iopub.status.idle":"2025-09-26T05:53:15.570202Z","shell.execute_reply.started":"2025-09-26T05:53:13.452113Z","shell.execute_reply":"2025-09-26T05:53:15.569358Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_rr_multi_segments(train_seg, planet_ids, segment_list=(30,40,50), prefix=\"f\", kind=\"both\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-26T05:53:15.571247Z","iopub.execute_input":"2025-09-26T05:53:15.571536Z","iopub.status.idle":"2025-09-26T05:53:17.634639Z","shell.execute_reply.started":"2025-09-26T05:53:15.57151Z","shell.execute_reply":"2025-09-26T05:53:17.63363Z"}},"outputs":[],"execution_count":null}]}