{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":101849,"databundleVersionId":13093295,"sourceType":"competition"},{"sourceId":9629432,"sourceType":"datasetVersion","datasetId":5846888}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-09-18T13:35:07.616877Z","iopub.execute_input":"2025-09-18T13:35:07.617175Z","iopub.status.idle":"2025-09-18T13:35:28.8625Z","shell.execute_reply.started":"2025-09-18T13:35:07.617146Z","shell.execute_reply":"2025-09-18T13:35:28.861358Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Hello World\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-18T13:35:28.863496Z","iopub.execute_input":"2025-09-18T13:35:28.863939Z","iopub.status.idle":"2025-09-18T13:35:28.869761Z","shell.execute_reply.started":"2025-09-18T13:35:28.863916Z","shell.execute_reply":"2025-09-18T13:35:28.868733Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install --no-index --find-links=/kaggle/input/ariel-2024-pqdm pqdm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-18T13:35:28.872155Z","iopub.execute_input":"2025-09-18T13:35:28.872476Z","iopub.status.idle":"2025-09-18T13:35:34.894524Z","shell.execute_reply.started":"2025-09-18T13:35:28.872451Z","shell.execute_reply":"2025-09-18T13:35:34.892992Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Import libraries","metadata":{}},{"cell_type":"code","source":"import os\nimport time\nimport itertools\nimport numpy as np\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\n\nfrom tqdm import tqdm\nfrom pqdm.threads import pqdm\nfrom astropy.stats import sigma_clip\nfrom scipy.optimize import minimize\nfrom torch.utils.data import DataLoader, TensorDataset, random_split\nfrom sklearn.preprocessing import StandardScaler\nfrom scipy.signal import savgol_filter\nfrom sklearn.metrics import mean_squared_error\nimport matplotlib.pyplot as plt\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-18T13:35:34.900428Z","iopub.execute_input":"2025-09-18T13:35:34.90072Z","iopub.status.idle":"2025-09-18T13:35:42.845831Z","shell.execute_reply.started":"2025-09-18T13:35:34.900686Z","shell.execute_reply":"2025-09-18T13:35:42.844465Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ROOT_PATH = \"/kaggle/input/ariel-data-challenge-2025\"\nMODE = \"test\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-18T13:35:42.846898Z","iopub.execute_input":"2025-09-18T13:35:42.847348Z","iopub.status.idle":"2025-09-18T13:35:42.85227Z","shell.execute_reply.started":"2025-09-18T13:35:42.847323Z","shell.execute_reply":"2025-09-18T13:35:42.851348Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Configuration","metadata":{}},{"cell_type":"code","source":"class Config:\n    DATA_PATH = \"/kaggle/input/ariel-data-challenge-2025\"\n    DATASET = \"train\"\n\n    SCALE = 0.96\n    SIGMA = 0.00055\n\n    CUT_INF = 39\n    CUT_SUP = 321\n\n    SENSOR_CONFIG = {\n        \"AIRS-CH0\": {\n            \"raw_shape\": [11250, 32, 356],\n            \"calibrated_shape\": [1, 32, CUT_SUP - CUT_INF],\n            \"linear_corr_shape\": (6, 32, 356),\n            \"dt_pattern\": (0.1, 4.5), \n            \"binning\": 30\n        },\n        \"FGS1\": {\n            \"raw_shape\": [135000, 32, 32],\n            \"calibrated_shape\": [1, 32, 32],\n            \"linear_corr_shape\": (6, 32, 32),\n            \"dt_pattern\": (0.1, 0.1),\n            \"binning\": 30 * 12\n        }\n    }\n\n    MODEL_PHASE_DETECTION_SLICE = slice(30, 140)\n    MODEL_OPTIMIZATION_DELTA = 11\n    MODEL_POLYNOMIAL_DEGREE = 3\n    \n    N_JOBS = 6 ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-18T13:35:42.853443Z","iopub.execute_input":"2025-09-18T13:35:42.85385Z","iopub.status.idle":"2025-09-18T13:35:42.882679Z","shell.execute_reply.started":"2025-09-18T13:35:42.853815Z","shell.execute_reply":"2025-09-18T13:35:42.881491Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Generating Dataset","metadata":{}},{"cell_type":"markdown","source":"### How to Use the Preprocessing Toggles (Recommended Practice)¶\n\n- DO_MASK: Recommended True. Even masking dead/hot pixels alone reduces noise significantly.\n\n- DO_NL_CORR: Start with False. Once learning/optimization is stable, test with True.\n\n- DO_DARK: Recommended True. Key tip: scale by integration time before subtraction.\n\n- DO_FLAT: Recommended True. Apply after binning for better numerical stability and smoother missing-data propagation.","metadata":{}},{"cell_type":"markdown","source":"We need to prove that the using the ROI is better than applying to all the rows","metadata":{}},{"cell_type":"code","source":"#Data for the FGS1 \nclass SignalProcessor:\n    def __init__(self, config):\n        self.cfg = config\n        self.adc_info = pd.read_csv(f\"{self.cfg.DATA_PATH}/adc_info.csv\")\n        self.planet_ids = pd.read_csv(f\"{self.cfg.DATA_PATH}/{self.cfg.DATASET}_star_info.csv\",\n                                     index_col = \"planet_id\").index.astype(int)\n\n    def _apply_linear_corr(self, linear_corr, signal):\n        # Using the Homer's method for polynomial correction\n        coeffs = np.flip(linear_corr, axis = 0)\n        x = signal.astype(np.float64, copy = False)\n\n        out = np.empty_like(x, dtype = np.float64)\n        out[...] = coeffs[0]\n\n        for k in range(1, coeffs.shape[0]):\n            out = np.multiply(out, signal)\n            out += coeffs[k]\n\n        return out.astype(signal.dtype, copy = False)\n\n\n    def _calibrate_single_signal(self, planet_id, sensor):\n        sensor_cfg = self.cfg.SENSOR_CONFIG[sensor]\n    \n        signal = pd.read_parquet(\n            f\"{self.cfg.DATA_PATH}/{self.cfg.DATASET}/{planet_id}/{sensor}_signal_0.parquet\"\n        ).to_numpy()\n        \n        dark = pd.read_parquet(\n            f\"{self.cfg.DATA_PATH}/{self.cfg.DATASET}/{planet_id}/{sensor}_calibration_0/dark.parquet\"\n        ).to_numpy()\n        \n        dead = pd.read_parquet(\n            f\"{self.cfg.DATA_PATH}/{self.cfg.DATASET}/{planet_id}/{sensor}_calibration_0/dead.parquet\"\n        ).to_numpy()\n        \n        flat = pd.read_parquet(\n            f\"{self.cfg.DATA_PATH}/{self.cfg.DATASET}/{planet_id}/{sensor}_calibration_0/flat.parquet\"\n        ).to_numpy()\n        \n        linear_corr = pd.read_parquet(\n            f\"{self.cfg.DATA_PATH}/{self.cfg.DATASET}/{planet_id}/{sensor}_calibration_0/linear_corr.parquet\"\n        ).values.astype(np.float64).reshape(sensor_cfg[\"linear_corr_shape\"])\n\n        # ADC Conversion\n        signal = signal.reshape(sensor_cfg[\"raw_shape\"])\n        gain = self.adc_info[f\"{sensor}_adc_gain\"].iloc[0]\n        offset = self.adc_info[f\"{sensor}_adc_offset\"].iloc[0]\n        signal = signal / gain + offset\n    \n        hot = sigma_clip(dark, sigma=5, maxiters=5).mask\n    \n        if sensor == \"AIRS-CH0\":\n            signal = signal[:, :, self.cfg.CUT_INF : self.cfg.CUT_SUP]\n            linear_corr = linear_corr[:, :, self.cfg.CUT_INF : self.cfg.CUT_SUP]\n            dark = dark[:, self.cfg.CUT_INF : self.cfg.CUT_SUP]\n            dead = dead[:, self.cfg.CUT_INF : self.cfg.CUT_SUP]\n            flat = flat[:, self.cfg.CUT_INF : self.cfg.CUT_SUP]\n            hot = hot[:, self.cfg.CUT_INF : self.cfg.CUT_SUP]\n    \n        if sensor == \"FGS1\":\n            y0, y1, x0, x1 = 10, 22, 10, 22\n            signal = signal[:, y0:y1, x0:x1]\n            dark   = dark[y0:y1, x0:x1]\n            dead   = dead[y0:y1, x0:x1]\n            flat   = flat[y0:y1, x0:x1]\n            linear_corr = linear_corr[:, y0:y1, x0:x1]\n            hot    = hot[y0:y1, x0:x1]\n    \n        np.maximum(signal, 0, out=signal)\n\n        # Using linear correction \n        if sensor == \"FGS1\":\n            signal = self._apply_linear_corr(linear_corr, signal)\n        elif sensor == \"AIRS-CH0\":\n            # Use ROI to reduce the risk of noise amplification\n            sl = (slice(None), slice(10, 22), slice(None))\n            signal[sl] = self._apply_linear_corr(linear_corr[:, 10:22, :], signal[sl])\n        else:\n            signal = self._apply_linear_corr(linear_corr, signal)\n\n        \n        # Dark subtraction step\n        base_dt, increment = sensor_cfg[\"dt_pattern\"]\n        even_scale = base_dt\n        odd_scale  = base_dt + increment\n        signal[::2]  -= dark * even_scale\n        signal[1::2] -= dark * odd_scale\n    \n        if sensor == \"FGS1\":\n            flat_roi = flat.astype(signal.dtype, copy=False).copy()\n            bad = (dead) | ~np.isfinite(flat_roi) | (flat_roi == 0)\n            flat_roi[bad] = np.nan\n            signal /= flat_roi\n    \n        elif sensor == \"AIRS-CH0\":\n            y0, y1 = 10, 22\n            flat_roi = flat[y0:y1, :].astype(signal.dtype, copy=False).copy()\n            bad = (dead[y0:y1, :]) | ~np.isfinite(flat_roi) | (flat_roi == 0)\n            flat_roi[bad] = np.nan\n            signal[:, y0:y1, :] /= flat_roi\n    \n        else:\n            flat2 = flat.astype(signal.dtype, copy=False).copy()\n            bad2 = (dead) | ~np.isfinite(flat2) | (flat2 == 0)\n            flat2[bad2] = np.nan\n            signal /= flat2\n\n        return signal\n\n    def _preprocess_calibrated_signal(self, calibrated_signal, sensor):\n        sensor_cfg = self.cfg.SENSOR_CONFIG[sensor]\n        binning = sensor_cfg[\"binning\"]\n\n        if sensor == \"AIRS-CH0\":\n            signal_roi = calibrated_signal[:, 10:22, :]\n        elif sensor == \"FGS1\":\n            signal_roi = calibrated_signal[:, :, :]\n            signal_roi = signal_roi.reshape(signal_roi.shape[0], -1)\n\n        # Averages the signal over the spatial dimension (axis=1) \n        # to produce a 1D time series per feature (wavelength for AIRS-CH0, flattened pixels for FGS1).\n        mean_signal = np.nanmean(signal_roi, axis=1)\n        cds_signal = mean_signal[1::2] - mean_signal[0::2]\n\n        n_bins = cds_signal.shape[0] // binning\n        binned = np.array([\n            cds_signal[j*binning : (j+1)*binning].mean(axis=0) \n            for j in range(n_bins)\n        ])\n\n        # Removing the outliers\n        if sensor == \"AIRS-CH0\":\n            # Just keeping the values with low percentile (5%) and high percentile (95%)\n            q_lo = np.nanpercentile(binned, 5.0, axis=1, keepdims=True)\n            q_hi = np.nanpercentile(binned, 95.0, axis=1, keepdims=True)\n            np.clip(binned, q_lo, q_hi, out=binned)\n\n        if sensor == \"FGS1\":\n            binned = binned.reshape((binned.shape[0], 1))\n\n        if sensor == \"AIRS-CH0\":\n            var = np.nanvar(binned, axis=0, ddof=1)\n            med = np.nanmedian(var)\n            safe_var = np.where(~np.isfinite(var) | (var <= 0), med if (np.isfinite(med) and med > 0) else 1.0, var)\n            w = 1.0 / safe_var\n\n            lo, hi = np.nanpercentile(w, 5.0), np.nanpercentile(w, 95.0)\n            if np.isfinite(lo) and np.isfinite(hi) and lo < hi:\n                w = np.clip(w, lo, hi)\n\n            M = binned.shape[1]\n            s = np.nansum(w)\n            if np.isfinite(s) and s > 0:\n                w = w * (M / s)\n            else:\n                w = np.ones_like(w)\n\n            binned *= w[None, :]\n\n        return binned\n\n    def _process_planet_sensor(self, args):\n        planet_id, sensor = args['planet_id'], args['sensor']\n        calibrated = self._calibrate_single_signal(planet_id, sensor)\n        preprocessed = self._preprocess_calibrated_signal(calibrated, sensor)\n        return preprocessed\n\n    def process_all_data(self):\n\n        args_airs_ch0 = [dict(planet_id=planet_id, sensor=\"AIRS-CH0\") for planet_id in self.planet_ids]\n        preprocessed_airs_ch0 = pqdm(args_airs_ch0, self._process_planet_sensor, n_jobs=self.cfg.N_JOBS)\n        print(preprocessed_airs_ch0[0].shape)\n        \n        args_fgs1 = [dict(planet_id=planet_id, sensor=\"FGS1\") for planet_id in self.planet_ids]\n        preprocessed_fgs1 = pqdm(args_fgs1, self._process_planet_sensor, n_jobs=self.cfg.N_JOBS)\n        print(preprocessed_fgs1[0].shape)\n        \n\n        preprocessed_signal = np.concatenate(\n            [np.stack(preprocessed_fgs1), np.stack(preprocessed_airs_ch0)], axis=2\n        )\n        return preprocessed_signal\n\n\nconfig = Config()\nprint(config.SENSOR_CONFIG['FGS1']['binning'])\n\nsignal_process = SignalProcessor(config)\npreprocess_data = signal_process.process_all_data()\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-18T13:35:42.883776Z","iopub.execute_input":"2025-09-18T13:35:42.884037Z","iopub.status.idle":"2025-09-18T13:36:35.251608Z","shell.execute_reply.started":"2025-09-18T13:35:42.884017Z","shell.execute_reply":"2025-09-18T13:36:35.250056Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"preprocess_data.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-18T13:36:35.255412Z","iopub.execute_input":"2025-09-18T13:36:35.255787Z","iopub.status.idle":"2025-09-18T13:36:35.263909Z","shell.execute_reply.started":"2025-09-18T13:36:35.255751Z","shell.execute_reply":"2025-09-18T13:36:35.26266Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data1 = preprocess_data[0]\nfgs1 = data1[:, 0]\nplt.plot(fgs1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-18T13:36:35.264922Z","iopub.execute_input":"2025-09-18T13:36:35.265314Z","iopub.status.idle":"2025-09-18T13:36:35.597761Z","shell.execute_reply.started":"2025-09-18T13:36:35.265279Z","shell.execute_reply":"2025-09-18T13:36:35.596657Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"link_save = '/kaggle/working/preprocess_data.npy'\n\nnp.save(link_save, preprocess_data)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-18T13:36:35.598812Z","iopub.execute_input":"2025-09-18T13:36:35.599151Z","iopub.status.idle":"2025-09-18T13:36:35.610378Z","shell.execute_reply.started":"2025-09-18T13:36:35.599113Z","shell.execute_reply":"2025-09-18T13:36:35.609088Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}