{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":101849,"databundleVersionId":12846694,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# This notebook is a kind of answer of [Calibrating and binning ariel data](https://www.kaggle.com/code/gordonyip/calibrating-and-binning-ariel-data)","metadata":{}},{"cell_type":"markdown","source":"<h2 style=\"color: #0c0a0b; font-size: 3.5rem; text-align: center;padding:0.5rem;border-radius:5rem; background-color:#8db084; border-bottom: 1.5rem solid #8A26BD\"> 0.3) Create my custom palette</h2>","metadata":{}},{"cell_type":"code","source":"%%time\nimport matplotlib.pyplot as plt # Inne method plt\nimport seaborn as sns \nfrom matplotlib.colors import ListedColormap\n\n\n# Custom colors\nmy_coolors = [\"#407e31\", \"#71b12c\", \"#8A26BD\", \"#C7283A\",\"#b5ff46\", \"#858585\",\"#8db084\", \"#0c0a0b\", \"#edf2f8\"]\nclass mainK:\n    c = '\\033[1m' + '\\033[1;38;2;138;38;189m'\n    f = '\\033[0m'\nclass secK:\n    c = '\\033[1m' + '\\033[1;38;2;181;255;70m'\n    f = '\\033[0m'  \nclass terK:\n    c = '\\033[1m' + '\\033[1;38;2;199;40;58m'\n    f = '\\033[0m'      \n\npltKolors = ListedColormap(my_coolors)\nsns.palplot(sns.color_palette(my_coolors))\nprint(mainK.c+\"Main color Palette:\"+mainK.f,\"\\n\")\nprint(secK.c+\"Second color Palette:\"+secK.f,\"\\n\")\nprint(terK.c+\"Third color Palette:\"+terK.f,\"\\n\")\nplt.show()","metadata":{"trusted":true,"_kg_hide-input":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h2 style=\"color: #0c0a0b; font-size: 3.5rem; text-align: center;padding:0.5rem;border-radius:5rem; background-color:#8db084; ; border-bottom: 1.5rem solid #8A26BD\"> 0.4) Prepare CPU for the job on Dask </h2>","metadata":{}},{"cell_type":"code","source":"%%time\nimport os\nimport dask\nfrom dask.distributed import LocalCluster, Client, SSHCluster\nimport concurrent.futures\nimport dask.dataframe as dd\nimport dask.array as da\nimport traceback\nimport cProfile\n\ndask_client = None  # Global reference to Dask client\n\nclass DaskClusterManager:\n    \"\"\"Manages the Dask cluster and client.\"\"\"\n    def __init__(self, remote_addresses=None):\n        global dask_client\n        self.remote_addresses = remote_addresses if remote_addresses else []\n        self.client = dask_client  # Use global reference\n\n    def create_cluster(self):\n        \"\"\"Creates and returns a Dask client.\"\"\"\n        global dask_client\n        try:\n            num_local_workers = os.cpu_count()\n            local_cluster = LocalCluster(n_workers=num_local_workers)\n            add_remote_machines = len(self.remote_addresses) > 0\n\n            if add_remote_machines:\n                remote_cluster = SSHCluster(self.remote_addresses)\n                combined_cluster = local_cluster + remote_cluster\n            else:\n                combined_cluster = local_cluster\n\n            self.client = Client(combined_cluster)\n            dask_client = self.client  # Assign to global variable\n\n                        \n            print(mainK.c + f'The link to the dashboard is  >>>>>>>>>>' + mainK.f,\n                  secK.c + f'{self.client.dashboard_link}' + secK.f,\n                  terK.c+f\"\\n🧑‍💻 Using client as 🧑‍💻 >>>>>>>>>> {self.client}\\n\"+terK.f)            \n            return self.client\n        except Exception as e:\n            print(secK.c + f\" \\u2620\\ufe0f An error occurred while creating the Dask cluster \\u2620\\ufe0f: {e}\" + secK.f)\n            traceback.print_exc()\n            return None\n\n    def get_client(self):\n        \"\"\"Returns the existing client or creates a new one if it doesn't exist.\"\"\"\n        global dask_client\n        if self.client is None:\n            self.client = self.create_cluster()\n            dask_client = self.client\n        return self.client\n\n    def close_client(self):\n        \"\"\"Closes the dask client\"\"\"\n        global dask_client\n        if self.client:\n            self.client.close()\n            dask_client = None\n\n    def get_workers_id(self):\n        \"\"\"Returns a dictionary with worker IDs and their addresses.\"\"\"\n        if self.client is None:\n            print(\"No client available. Please create a cluster first.\")\n            return {}\n        scheduler_info = self.client.scheduler_info()\n        workers = scheduler_info.get('workers', {})\n        worker_info = {}\n        for worker_id, worker_details in workers.items():\n            worker_info[worker_id] = worker_details['host']\n        return worker_info\n\n\nif __name__ == \"__main__\":\n    cluster_manager = DaskClusterManager()  # Create the Dask manager\n    dask_client = cluster_manager.get_client()  # Get the dask client\n\n    if dask_client:\n        # Get worker IDs and addresses\n        workers = cluster_manager.get_workers_id()\n        for worker_id, address in workers.items():\n            print(secK.c+f\"⚙️ Worker ID ⚙️: {worker_id}, Address: {address}\"+secK.f)\n\n        # Optional profiling or cleanup\n        # cProfile.run('dask_client')\n        # cluster_manager.close_client()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-04T00:42:08.314393Z","iopub.execute_input":"2025-07-04T00:42:08.314826Z","iopub.status.idle":"2025-07-04T00:42:16.221119Z","shell.execute_reply.started":"2025-07-04T00:42:08.314793Z","shell.execute_reply":"2025-07-04T00:42:16.220332Z"},"_kg_hide-input":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h1 style=\"color:#0c0a0b;background-color:#71b12c;font-size: 7.5rem; text-align: center;padding:0.5rem;border-radius:5rem; border-bottom: 1.5rem solid #C7283A\"> 1️⃣ ETL 1️⃣</h1>","metadata":{}},{"cell_type":"markdown","source":"<h2 style=\"color:#8A26BD;font-size: 2rem; border-bottom: .5rem solid #71b12c;\">Dataset Description</h2>\n<p style=\"font-size: 1.6rem; line-height: 1.6;\">\nCharacterizing the chemistry of exoplanets is one of the great active projects in astronomy. The European Space Agency's Ariel mission will gather data on roughly 1,000 exoplanets by observing them while they transit in front of their host stars. Even with the powerful instruments on board Ariel, the resulting data will be based on a limited number of photons and include a fair amount of noise. Your challenge in this competition is to extract the chemical spectrum of the atmospheres of exoplanets using simulated Ariel data.\n</p>\n\n\n<h3 style=\"color:#71b12c;font-size: 2.2rem;\">Metadata Files</h3>\n<table border=\"1\" style=\"width:100%; border-collapse: collapse;\">\n    <tr>\n        <th style=\"padding: 8px; text-align: left; background-color: #71b12c;\">File</th>\n        <th style=\"padding: 8px; text-align: left; background-color: #71b12c;\">Description</th>\n    </tr>\n    <tr>\n        <td style=\"padding: 8px;\">train.csv</td>\n        <td style=\"padding: 8px;\">Ground truth spectra.</td>\n    </tr>\n    <tr>\n        <td style=\"padding: 8px;\">wavelengths.csv</td>\n        <td style=\"padding: 8px;\">The wavelength grid for each ground truth spectrum in the dataset.</td>\n    </tr>\n    <tr>\n        <td style=\"padding: 8px;\">axis_info.parquet</td>\n        <td style=\"padding: 8px;\">Axis information for both instruments (AIRS-CH0 and FGS1).</td>\n    </tr>\n    <tr>\n        <td style=\"padding: 8px;\">adc_info.csv</td>\n        <td style=\"padding: 8px;\">Analog-to-digital (ADC) conversion parameters (gain and offset) for restoring the original dynamic range of the data. Unlike last year's challenge, all planets use the same adc_info.</td>\n    </tr>\n    <tr>\n        <td style=\"padding: 8px;\">[train/test]_star_info.csv</td>\n        <td style=\"padding: 8px;\">\n            <ul>\n                <li><strong>planet_id:</strong> Unique identifier for the star-planet system.</li>\n                <li><strong>Rs:</strong> Stellar radius in solar radii (R☉).</li>\n                <li><strong>Ms:</strong> Stellar mass in solar masses (M☉).</li>\n                <li><strong>Ts:</strong> Stellar effective temperature in Kelvin.</li>\n                <li><strong>Mp:</strong> Planetary mass in Earth masses (M⊕).</li>\n                <li><strong>e:</strong> Orbital eccentricity (dimensionless).</li>\n                <li><strong>P:</strong> Orbital period in days.</li>\n                <li><strong>sma:</strong> Semi-major axis in stellar radii (Rs), showing the orbital distance relative to the stellar radii.</li>\n                <li><strong>i:</strong> Orbital inclination in degrees.</li>\n            </ul>\n        </td>\n    </tr>\n    <tr>\n        <td style=\"padding: 8px;\">sample_submission.csv</td>\n        <td style=\"padding: 8px;\">A sample submission in the correct format.</td>\n    </tr>\n</table>\n\n<h3 style=\"color:#71b12c;font-size: 2.2rem;\">Signal Files</h3>\n<table border=\"1\" style=\"width:100%; border-collapse: collapse;\">\n    <tr>\n        <th style=\"padding: 8px; text-align: left; background-color: #71b12c;\">File</th>\n        <th style=\"padding: 8px; text-align: left; background-color: #71b12c;\">Description</th>\n    </tr>\n    <tr>\n        <td style=\"padding: 8px;\">[train/test]/[planet_id]/AIRS-CH0_signal_[observation_count].parquet</td>\n        <td style=\"padding: 8px;\">Signal data from the AIRS-CH0 instrument. Each file contains 11,250 rows of images captured at constant time steps noted in axis_info.parquet file for details of the time steps. Each 32 x 356 image has been flattened into 11392 columns. You can un-flatten the data with numpy.reshape(11250, 32, 356). The instruments generate data as uint16. To restore the full dynamic range you must multiply the data by the matching gain value from adc_info.csv and then add the offset value, also from adc_info.csv.</td>\n    </tr>\n    <tr>\n        <td style=\"padding: 8px;\">[train/test]/[planet_id]/FGS1_signal_[observation_count].parquet</td>\n        <td style=\"padding: 8px;\">Signal data from the FGS1 instrument. Each file contains 135,000 rows of images at 0.1 second time steps. Each 32x32 image has been flattened into 1024 columns. You can un-flatten the data with numpy.reshape(135000, 32, 32). Similar to AIR-CH0, the data is generated in uint16. To restore its original dynamic range you must multiply the data by the matching gain value from adc_info.csv and then add the offset value, also from adc_info.csv.</td>\n    </tr>\n</table>\n\n<h3 style=\"color:#71b12c;font-size: 2.2rem;\">Calibration Files</h3>\n<table border=\"1\" style=\"width:100%; border-collapse: collapse;\">\n    <tr>\n        <th style=\"padding: 8px; text-align: left; background-color: #71b12c;\">File</th>\n        <th style=\"padding: 8px; text-align: left; background-color: #71b12c;\">Description</th>\n    </tr>\n    <tr>\n        <td style=\"padding: 8px;\">[train/test]/[planet_id]/[AIRS-CH0/FGS1]_calibration/dark.parquet</td>\n        <td style=\"padding: 8px;\">Dark frames are exposures taken with the shutter closed, capturing the thermal noise and bias level of the sensor. These are used to subtract the dark current from science images.</td>\n    </tr>\n    <tr>\n        <td style=\"padding: 8px;\">[train/test]/[planet_id]/[AIRS-CH0/FGS1]_calibration/dead.parquet</td>\n        <td style=\"padding: 8px;\">Identifies dead or hot pixels on the sensor. Dead pixels do not respond to light, while hot pixels consistently produce high signal levels regardless of incoming light.</td>\n    </tr>\n    <tr>\n        <td style=\"padding: 8px;\">[train/test]/[planet_id]/[AIRS-CH0/FGS1]_calibration/flat.parquet</td>\n        <td style=\"padding: 8px;\">Flat field frames are created by imaging a uniformly illuminated surface. They are used to correct for variations in pixel-to-pixel sensitivity and optical system irregularities.</td>\n    </tr>\n    <tr>\n        <td style=\"padding: 8px;\">[train/test]/[planet_id]/[AIRS-CH0/FGS1]_calibration/linear_corr.parquet</td>\n        <td style=\"padding: 8px;\">Information about the linearity correction of the sensor. The response of the pixels in the detector becomes less linear as they fill with electrons, approaching the point of saturation, where the pixel can no longer collect additional electrons and its response to light becomes flat. For an accurate estimate of the signal, the instrument's response as a function of the received charge is calibrated, and the correction is calculated using a polynomial of degree n. This polynomial allows for the conversion of the number of electrons collected/measured by the pixel into the number of electrons that the detector would have generated with a linear response.</td>\n    </tr>\n    <tr>\n        <td style=\"padding: 8px;\">[train/test]/[planet_id]/[AIRS-CH0/FGS1]_calibration/read.parquet</td>\n        <td style=\"padding: 8px;\">Read noise frames capture the electronic noise introduced during the readout process of the sensor. This noise is present even when no light falls on the detector.</td>\n    </tr>\n</table>\n","metadata":{}},{"cell_type":"code","source":"%%time\n\"\"\"How the metadata comes\"\"\"\n\nimport polars as pl\n\n# Define the dataset\nMetadataSet = [\n    \"/kaggle/input/ariel-data-challenge-2025/adc_info.csv\",\n    \"/kaggle/input/ariel-data-challenge-2025/axis_info.parquet\",\n    \"/kaggle/input/ariel-data-challenge-2025/sample_submission.csv\",\n    \"/kaggle/input/ariel-data-challenge-2025/test_star_info.csv\",\n    \"/kaggle/input/ariel-data-challenge-2025/train.csv\",\n    \"/kaggle/input/ariel-data-challenge-2025/train_star_info.csv\",\n    \"/kaggle/input/ariel-data-challenge-2025/wavelengths.csv\"\n]\n\ndef analyze_file(file_path):\n    \"\"\"Analyze a single file and print its characteristics\"\"\"\n    print(f\"\\n=== Analyzing: {file_path.split('/')[-1]} ===\")\n    \n    try:\n        # Read file based on extension\n        if file_path.endswith('.csv'):\n            df = pl.read_csv(file_path)\n        elif file_path.endswith('.parquet'):\n            df = pl.read_parquet(file_path)\n        else:\n            print(\"Unsupported file format\")\n            return\n        \n        # Display basic information\n        print(\"\\n\\033[1mFirst 2 rows:\\033[0m\")\n        print(df.head(2))\n        \n        print(\"\\n\\033[1mData types:\\033[0m\")\n        for name, dtype in zip(df.columns, df.dtypes):\n            print(f\"{name}: {dtype}\")\n        \n        print(\"\\n\\033[1mDataframe Shape:\\033[0m\")\n        print(f\"Rows: {df.height:,}, Columns: {df.width}\")\n        \n        # Display statistics for numeric columns\n        numeric_cols = [col for col, dtype in zip(df.columns, df.dtypes) if dtype in (pl.Int64, pl.Float64)]\n        if numeric_cols:\n            print(\"\\n\\033[1mBasic statistics:\\033[0m\")\n            print(df.select(numeric_cols).describe())\n        else:\n            print(\"\\nNo numeric columns for statistics\")\n            \n    except Exception as e:\n        print(f\"\\nError processing file: {str(e)}\")\n\n\nif __name__ == \"__main__\":\n    # Analyze all files\n    for file_path in MetadataSet:\n        analyze_file(file_path)\n        print(\"\\n\" + \"=\"*60)","metadata":{"trusted":true,"_kg_hide-input":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### DataExplorer\n**Purpose**: Directory structure analysis.  \n**Time Complexity**:  \n- **O(D + F)**, where:  \n  - `D`: Subdirectories  \n  - `F`: Files  \n**Key Notes**:  \n- Recursive traversal with pattern generalization.  \n- Plotly visualizations add **O(1)** post-processing.  ","metadata":{}},{"cell_type":"code","source":"%%time\n\"\"\"Folder explorer + plot\"\"\"\n\nimport os\nimport re\nfrom collections import defaultdict\nimport plotly.graph_objects as go\nfrom plotly.subplots import make_subplots\n\nclass DataExplorer:\n    def __init__(self, root_path):\n        self.root_path = root_path\n        self.structure = {}\n        self.patterns = {\n            'folder_patterns': defaultdict(int),\n            'file_patterns': defaultdict(int),\n            'extensions': defaultdict(int)\n        }\n    \n    def explore(self):\n        \"\"\"Explora recursivamente la estructura del directorio.\"\"\"\n        self._explore_recursive(self.root_path, level=0)\n    \n    def _explore_recursive(self, path, level):\n        \"\"\"Función recursiva para mapear carpetas y archivos.\"\"\"\n        if level not in self.structure:\n            self.structure[level] = {'folders': [], 'files': []}\n        \n        try:\n            for item in os.listdir(path):\n                full_path = os.path.join(path, item)\n                if os.path.isdir(full_path):\n                    self.structure[level]['folders'].append(item)\n                    pattern = self._generalize_name(item)\n                    self.patterns['folder_patterns'][pattern] += 1\n                    self._explore_recursive(full_path, level + 1)\n                else:\n                    self.structure[level]['files'].append(item)\n                    ext = os.path.splitext(item)[1]\n                    pattern = self._generalize_name(item)\n                    self.patterns['file_patterns'][pattern] += 1\n                    self.patterns['extensions'][ext] += 1\n        except PermissionError:\n            print(f\"[!] No se puede acceder a {path}: permiso denegado.\")\n    \n    def _generalize_name(self, name):\n        \"\"\"\n        Convierte nombres específicos en patrones generalizados.\n        Ejemplo: P001 -> ID_*, batch_001 -> batch_*\n        \"\"\"\n        # Reemplaza números con *\n        generalized = re.sub(r'\\d+', '*', name)\n        # Reemplaza IDs tipo P001 -> ID_*\n        generalized = re.sub(r'[pP]\\d+', 'ID_*', generalized)\n        # Reemplaza fechas tipo 20250401 -> DATE_*\n        generalized = re.sub(r'\\b\\d{8}\\b', 'DATE_*', generalized)\n        # Reemplaza UUIDs u otros patrones alfanuméricos largos\n        generalized = re.sub(r'^[a-zA-Z0-9]{6,}$', 'CODE_*', generalized)\n        return generalized\n    \n    def summary(self):\n        \"\"\"Imprime un resumen de la estructura y patrones encontrados.\"\"\"\n        print(\"\\n=== Directory structure ===\")\n        for level, content in self.structure.items():\n            print(f\"\\nLevel {level}:\")\n            if content['folders']:\n                print(\"  Folders:\", ', '.join(content['folders'][:5]) + ('...' if len(content['folders']) > 5 else ''))\n            if content['files']:\n                print(\"  Files:\", ', '.join(content['files'][:5]) + ('...' if len(content['files']) > 5 else ''))\n        \n        print(\"\\n=== Generalized patterns ===\")\n        print(\"\\nFolder patterns:\")\n        for pattern, count in sorted(self.patterns['folder_patterns'].items(), key=lambda x: -x[1]):\n            print(f\"  {pattern}: {count}\")\n        \n        print(\"\\nFile patterns:\")\n        for pattern, count in sorted(self.patterns['file_patterns'].items(), key=lambda x: -x[1]):\n            print(f\"  {pattern}: {count}\")\n        \n        print(\"\\nFrequency of extensions:\")\n        for ext, count in sorted(self.patterns['extensions'].items(), key=lambda x: -x[1]):\n            print(f\"  {ext}: {count}\")\n    \n    def plot_results(self, top_n=10, output_dir=\"plots\"):\n        \"\"\"\n        Genera gráficos interactivos con Plotly y los guarda como HTML.\n        \n        Args:\n            top_n (int): Número máximo de elementos a mostrar en cada gráfico\n            output_dir (str): Directorio donde guardar los gráficos HTML\n        \"\"\"\n        if not os.path.exists(output_dir):\n            os.makedirs(output_dir)\n        \n        # Gráfico para patrones de carpetas\n        folder_patterns = sorted(self.patterns['folder_patterns'].items(), key=lambda x: -x[1])[:top_n]\n        fig1 = go.Figure()\n        fig1.add_trace(go.Bar(\n            x=[p[0] for p in folder_patterns],\n            y=[p[1] for p in folder_patterns],\n            name=\"Folder patterns\"\n        ))\n        fig1.update_layout(\n            title=\"Folder naming patterns (Top {})\".format(top_n),\n            xaxis_title=\"Pattern\",\n            yaxis_title=\"Frecuency\"\n        )\n        fig1.write_html(os.path.join(output_dir, \"folder_patternsTest.html\"))\n        \n        # Gráfico para patrones de archivos\n        file_patterns = sorted(self.patterns['file_patterns'].items(), key=lambda x: -x[1])[:top_n]\n        fig2 = go.Figure()\n        fig2.add_trace(go.Bar(\n            x=[p[0] for p in file_patterns],\n            y=[p[1] for p in file_patterns],\n            name=\"File patterns\"\n        ))\n        fig2.update_layout(\n            title=\"Folder naming patterns (Top {})\".format(top_n),\n            xaxis_title=\"Pattern\",\n            yaxis_title=\"Frecuency\"\n        )\n        fig2.write_html(os.path.join(output_dir, \"file_patternsTrain.html\"))\n        \n        # Gráfico para extensiones de archivos\n        extensions = sorted(self.patterns['extensions'].items(), key=lambda x: -x[1])[:top_n]\n        fig3 = go.Figure()\n        fig3.add_trace(go.Bar(\n            x=[p[0] for p in extensions],\n            y=[p[1] for p in extensions],\n            name=\"File extensions\"\n        ))\n        fig3.update_layout(\n            title=\"Most common file extensions (Top {})\".format(top_n),\n            xaxis_title=\"Extension\",\n            yaxis_title=\"Frecuency\"\n        )\n        fig3.write_html(os.path.join(output_dir, \"file_extensionsTest.html\"))\n        \n        # Gráfico combinado para resumen\n        fig_combined = make_subplots(\n            rows=3, cols=1,\n            subplot_titles=(\"Folder patterns\", \"Files patterns\", \"File extensions\")\n        )\n        \n        fig_combined.add_trace(\n            go.Bar(\n                x=[p[0] for p in folder_patterns],\n                y=[p[1] for p in folder_patterns],\n                name=\"Folders\"\n            ),\n            row=1, col=1\n        )\n        \n        fig_combined.add_trace(\n            go.Bar(\n                x=[p[0] for p in file_patterns],\n                y=[p[1] for p in file_patterns],\n                name=\"File\"\n            ),\n            row=2, col=1\n        )\n        \n        fig_combined.add_trace(\n            go.Bar(\n                x=[p[0] for p in extensions],\n                y=[p[1] for p in extensions],\n                name=\"Extensions\"\n            ),\n            row=3, col=1\n        )\n        \n        fig_combined.update_layout(\n            height=1200,\n            title_text=\"Summary of patterns in the directory structure for Test\",\n            showlegend=False\n        )\n        \n        fig_combined.write_html(os.path.join(output_dir, \"combined_summaryTest.html\"))\n        \n        print(f\"\\nGráficos guardados en el directorio '{output_dir}' como archivos HTML.\")\n\nif __name__ == \"__main__\":\n    # Ruta a tu carpeta principal\n    ROOT_PATH = \"/kaggle/input/ariel-data-challenge-2025/test\"\n    \n    # Crear instancia del explorador\n    explorer = DataExplorer(ROOT_PATH)\n    \n    # Explorar la estructura\n    explorer.explore()\n    \n    # Mostrar resumen\n    explorer.summary()\n    \n    # Generar gráficos\n    explorer.plot_results(top_n=15)","metadata":{"trusted":true,"jupyter":{"source_hidden":true},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### ArielUltraLightMapper\n**Purpose**: Maps directory structure to a minimal JSON index.  \n**Time Complexity**:  \n- **O(P × (S + C))**, where:  \n  - `P`: Number of planet directories  \n  - `S`: Average signal files per planet  \n  - `C`: Average calibration files per planet  \n**Key Notes**:  \n- Linear traversal of files/directories.  \n- Optimized with `defaultdict` for fast lookups. ","metadata":{}},{"cell_type":"code","source":"%%time\n\n\"\"\"ULTRA FILTRO + plot\"\"\"\n\nimport json\nfrom pathlib import Path\nfrom collections import defaultdict\nimport os\nimport plotly.express as px\nimport pandas as pd\n\n\nclass ArielUltraLightMapper:\n    def __init__(self, root_path):\n        self.root_path = Path(root_path).absolute()\n        self.data = {\n            '_meta': {\n                'root_path': str(self.root_path),\n                'file_patterns': {\n                    'signal': '{instrument}_signal_{obs_type}.parquet',\n                    'calibration': '{instrument}_calibration_{set_num}/{calib_type}.parquet'\n                },\n                'stats': defaultdict(int)\n            },\n            'index': {\n                'planets': set(),  # Usamos set para planetas únicos\n                'instruments': defaultdict(list),  # instrument: [planet_ids]\n                'signals': defaultdict(lambda: defaultdict(list)),  # instrument: {type: [keys]}\n                'calibrations': defaultdict(lambda: defaultdict(dict))  # instrument: {type: {set_num: key}}\n            },\n            'paths': {}  # key: relative_path\n        }\n\n    def build_mapping(self):\n        \"\"\"Construye un mapeo minimalista con índices cruzados\"\"\"\n        for planet_dir in self.root_path.iterdir():\n            if not planet_dir.is_dir():\n                continue\n            planet_id = planet_dir.name\n            self.data['index']['planets'].add(planet_id)\n            self.data['_meta']['stats']['planets'] += 1\n\n            # Procesar señales\n            for signal_file in planet_dir.glob('*_signal_*.parquet'):\n                instrument, obs_type = self._parse_signal(signal_file.name)\n                if not instrument:\n                    continue\n                rel_path = str(signal_file.relative_to(self.root_path))\n                key = f\"signal_{instrument}_{obs_type}\"\n                self.data['paths'][key] = rel_path\n\n                self.data['index']['signals'][instrument][obs_type].append(key)\n                if planet_id not in self.data['index']['instruments'][instrument]:\n                    self.data['index']['instruments'][instrument].append(planet_id)\n                self.data['_meta']['stats']['signals'] += 1\n\n            # Procesar calibraciones\n            for calib_dir in planet_dir.glob('*_calibration*'):\n                instrument, calib_set = self._parse_calib(calib_dir.name)\n                if not instrument:\n                    continue\n                for calib_file in calib_dir.glob('*.parquet'):\n                    calib_type = calib_file.stem\n                    rel_path = str(calib_file.relative_to(self.root_path))\n                    key = f\"calib_{instrument}_{calib_type}_{calib_set}\"\n                    self.data['paths'][key] = rel_path\n\n                    self.data['index']['calibrations'][instrument][calib_type][calib_set] = key\n                    if planet_id not in self.data['index']['instruments'][instrument]:\n                        self.data['index']['instruments'][instrument].append(planet_id)\n                    self.data['_meta']['stats']['calibrations'] += 1\n\n        self._optimize_structure()\n        return self.data\n\n    def _parse_signal(self, filename):\n        parts = filename.rsplit('_', 2)\n        if len(parts) == 3 and parts[-1].split('.')[0] in ('0', '1'):\n            return parts[0], parts[-1].split('.')[0]\n        return None, None\n\n    def _parse_calib(self, dirname):\n        parts = dirname.split('_')\n        if len(parts) >= 3:\n            return parts[0], parts[-1]\n        return None, None\n\n    def _optimize_structure(self):\n        self.data['index']['planets'] = sorted(self.data['index']['planets'])\n        self.data['index']['instruments'] = dict(self.data['index']['instruments'])\n        self.data['index']['signals'] = {\n            instr: dict(types)\n            for instr, types in self.data['index']['signals'].items()\n        }\n        self.data['index']['calibrations'] = {\n            instr: dict(types)\n            for instr, types in self.data['index']['calibrations'].items()\n        }\n\n    def save_json(self, output_file=\"ariel_ultralight.json\"):\n        with open(output_file, 'w') as f:\n            json.dump(self.data, f, separators=(',', ':'), indent=2 if __debug__ else None)\n\n        size_mb = os.path.getsize(output_file) / (1024 * 1024)\n        print(f\"✅ JSON ultra-ligero guardado ({size_mb:.2f} MB)\")\n        print(\"📊 Estadísticas:\")\n        for k, v in self.data['_meta']['stats'].items():\n            print(f\"- {k}: {v}\")\n\n    def generate_plotly_html(self, html_output=\"ariel_data_visualization.html\"):\n        \"\"\"Genera un gráfico interactivo con Plotly y lo guarda como HTML\"\"\"\n\n        # Extraer información del JSON\n        instruments = list(self.data[\"index\"][\"instruments\"].keys())\n        signals_count = []\n        calibrations_count = []\n        planets_per_instrument = []\n\n        for inst in instruments:\n            signal_types = self.data[\"index\"][\"signals\"].get(inst, {})\n            total_signals = sum(len(keys) for keys in signal_types.values())\n            signals_count.append(total_signals)\n\n            calib_types = self.data[\"index\"][\"calibrations\"].get(inst, {})\n            total_calibs = sum(len(set_nums) for set_nums in calib_types.values())\n            calibrations_count.append(total_calibs)\n\n            # Planetas asociados al instrumento\n            planets_per_instrument.append(len(self.data[\"index\"][\"instruments\"][inst]))\n\n        # Crear DataFrame para Plotly\n        df = pd.DataFrame({\n            \"Instrument\": instruments,\n            \"Total Signals\": signals_count,\n            \"Total Calibrations\": calibrations_count,\n            \"Associated Planets\": planets_per_instrument\n        })\n\n        # Gráfico 1: Distribución de Señales y Calibraciones por Instrumento\n        fig1 = px.bar(df, x=\"Instrument\", y=[\"Total Signals\", \"Total Calibrations\"],\n                      title=\"Signal Distribution and Calibrations per Instrument\",\n                      labels={\"value\": \"Quantity\", \"variable\": \"Data type\"},\n                      barmode='group')\n\n        # Gráfico 2: Planetas Asociados por Instrumento\n        fig2 = px.bar(df, x=\"Instrument\", y=\"Associated Planets\",\n                      title=\"Number of Associated Planets per Instrument\",\n                      labels={\"value\": \"Planet quantities\", \"x\": \"Instrument\"})\n\n        # Guardar ambos gráficos en un archivo HTML\n        from plotly.subplots import make_subplots\n        import plotly.graph_objects as go\n\n        fig = make_subplots(rows=2, cols=1, subplot_titles=(\"Signal Distribution and Calibrations\", \"Associated Planets\"))\n\n        # Agregar gráficos\n        fig.add_trace(fig1.data[0], row=1, col=1)\n        fig.add_trace(fig1.data[1], row=1, col=1)\n        fig.add_trace(fig2.data[0], row=2, col=1)\n\n        # Actualizar diseño\n        fig.update_layout(height=800, showlegend=True)\n\n        # Guardar el HTML\n        fig.write_html(html_output)\n        print(f\"📊 Gráfico interactivo guardado como {html_output}\")\n\n\n# Ejemplo de uso\nif __name__ == \"__main__\":\n    mapper = ArielUltraLightMapper(\"/kaggle/input/ariel-data-challenge-2025/train\")\n    mapper.build_mapping()\n    mapper.save_json()\n    mapper.generate_plotly_html()","metadata":{"trusted":true,"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2025-07-04T00:42:21.648099Z","iopub.execute_input":"2025-07-04T00:42:21.648476Z","iopub.status.idle":"2025-07-04T00:42:34.809788Z","shell.execute_reply.started":"2025-07-04T00:42:21.648449Z","shell.execute_reply":"2025-07-04T00:42:34.808884Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### ArielContext: query datasets (signals and calibrations)\n**Purpose**: Finds signal/calibration files with filtering.  \n**Time Complexity**:  \n- **O(P × I × T)** (worst case), where:  \n  - `P`: Active planets (truncatable)  \n  - `I`: Instruments queried  \n  - `T`: Signal/calibration types  \n**Key Notes**:  \n- Uses caching via `ArielContext` to avoid repeated filesystem scans.  \n- `truncate_planets()` reduces `P` for faster queries.  ","metadata":{}},{"cell_type":"code","source":"%%time\n\"\"\"Truncar lectura, query disk\"\"\"\n\nimport json\nfrom pathlib import Path\nfrom typing import List, Optional, Union\nfrom tqdm import tqdm\nimport os\nimport glob\n\n\ndef truncate_planets(planets: List[str], truncate: Union[float, int, None]) -> List[str]:\n    if truncate is None:\n        return planets\n    if isinstance(truncate, float) and 0 < truncate <= 1:\n        return planets[:int(len(planets) * truncate)]\n    if isinstance(truncate, int) and truncate > 0:\n        return planets[:truncate]\n    raise ValueError(\"truncate must be None, float (0<x<=1) or positive int\")\n\n\nclass DiskParquetCounter:\n    @staticmethod\n    def count_parquet_files(path: Path) -> int:\n        if path.is_file() and path.suffix == '.parquet':\n            return 1\n        elif path.is_dir():\n            return len(list(path.rglob(\"*.parquet\")))\n        return 0\n\n\nclass ArielContext:\n    def __init__(self, json_path: str, planet_id: Optional[str] = None, truncate_ids: Union[int, float, None] = None):\n        with open(json_path) as f:\n            self.data = json.load(f)\n\n        self.base_path = Path(self.data[\"_meta\"][\"root_path\"])\n        self.signal_pattern = self.data[\"_meta\"][\"file_patterns\"][\"signal\"]\n        self.calib_pattern = self.data[\"_meta\"][\"file_patterns\"][\"calibration\"]\n\n        self.planets = self.data[\"index\"][\"planets\"]\n        self.instruments = self.data[\"index\"][\"instruments\"]\n        self.calibrations = self.data[\"index\"][\"calibrations\"]\n        self.paths_index = self.data.get(\"paths\", {})\n\n        self.planet_id = planet_id\n        self.truncate_ids = truncate_ids\n        self.active_planets = [planet_id] if planet_id else truncate_planets(self.planets, truncate_ids)\n\n    def planet_has_instrument(self, planet: str, instr: str) -> bool:\n        return planet in self.instruments.get(instr, [])\n\n\nclass SignalFinder:\n    def __init__(self, context: ArielContext):\n        self.ctx = context\n\n    def find(self, instrument: Optional[str] = None, signal_type: Optional[Union[str, int]] = None, count_files: bool = False) -> List[str]:\n        results = []\n        file_count = 0\n        instruments = [instrument] if instrument else list(self.ctx.instruments.keys())\n\n        if signal_type == \"2\" or signal_type == 2:\n            types = [\"0\", \"1\"]\n        elif signal_type in [\"0\", \"1\"]:\n            types = [signal_type]\n        else:\n            types = [\"0\", \"1\"]  # default fallback\n\n        for planet in tqdm(self.ctx.active_planets, desc=\"Planets(signal)\"):\n            for instr in instruments:\n                if not self.ctx.planet_has_instrument(planet, instr):\n                    continue\n                for stype in types:\n                    rel = f\"{planet}/{self.ctx.signal_pattern.format(instrument=instr, obs_type=stype)}\"\n                    full = self.ctx.base_path / rel\n                    if full.exists():\n                        results.append(str(full))\n                        if count_files:\n                            file_count += DiskParquetCounter.count_parquet_files(full)\n        print(f\"📁 Total signal paths: {len(results)}\")\n        if count_files:\n            print(f\"📊 Total actual signal .parquet files on disk: {file_count}\")\n        return results\n\n\nclass CalibrationFinder:\n    def __init__(self, context: ArielContext):\n        self.ctx = context\n\n    def find(self, instrument: Optional[str] = None, calib_types: Optional[Union[List[str], int, str]] = None, calib_set: Optional[Union[str, int]] = None, count_files: bool = False) -> List[str]:\n        results = []\n        file_count = 0\n        instruments = [instrument] if instrument else list(self.ctx.calibrations.keys())\n\n        if calib_types == \"2\" or calib_types == 2:\n            if instrument:\n                types = list(self.ctx.calibrations.get(instrument, {}).keys())\n            else:\n                types = list(set().union(*[self.ctx.calibrations[i].keys() for i in self.ctx.calibrations]))\n        elif calib_types:\n            types = calib_types\n        else:\n            if instrument:\n                types = list(self.ctx.calibrations.get(instrument, {}).keys())\n            else:\n                types = list(set().union(*[self.ctx.calibrations[i].keys() for i in self.ctx.calibrations]))\n\n        sets_map = self.ctx.calibrations\n        known_paths = set(self.ctx.paths_index.values())\n\n        for planet in tqdm(self.ctx.active_planets, desc=\"Planets(calibration)\"):\n            for instr in instruments:\n                if not self.ctx.planet_has_instrument(planet, instr):\n                    continue\n                for ctype in types:\n                    if calib_set == \"2\" or calib_set == 2:\n                        available_sets = list(sets_map.get(instr, {}).get(ctype, {}).keys())\n                    elif calib_set:\n                        available_sets = [calib_set]\n                    else:\n                        available_sets = list(sets_map.get(instr, {}).get(ctype, {}).keys())\n\n                    for s in available_sets:\n                        rel_path = f\"{planet}/{self.ctx.calib_pattern.format(instrument=instr, set_num=s, calib_type=ctype)}\"\n                        full = self.ctx.base_path / rel_path\n                        if full.exists() or rel_path in known_paths:\n                            results.append(str(full))\n                            if count_files:\n                                file_count += DiskParquetCounter.count_parquet_files(full)\n        print(f\"📁 Total calibration paths: {len(results)}\")\n        if count_files:\n            print(f\"📊 Total actual calibration .parquet files on disk: {file_count}\")\n        return results\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-04T01:14:41.325631Z","iopub.execute_input":"2025-07-04T01:14:41.325997Z","iopub.status.idle":"2025-07-04T01:14:41.345758Z","shell.execute_reply.started":"2025-07-04T01:14:41.325973Z","shell.execute_reply":"2025-07-04T01:14:41.344113Z"},"_kg_hide-input":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ==================== EJEMPLOS ====================\nif __name__ == \"__main__\":\n    # Crear contexto una vez y compartirlo\n    ctx1 = ArielContext(\"ariel_ultralight.json\", truncate_ids=None)\n    ctx2 = ArielContext(\"ariel_ultralight.json\", planet_id=\"1810380816\")\n    ctx3 = ArielContext(\"ariel_ultralight.json\", truncate_ids=0.5)\n    ctx4 = ArielContext(\"ariel_ultralight.json\", planet_id=\"989956432\")\n\n    # 1) Señales AIRS-CH0, tipo 0, primeros 10 planetas\n    sigs = SignalFinder(ctx1).find(instrument=\"AIRS-CH0\", signal_type=\"2\", )\n    print(f\"Found {len(sigs)} signal paths.\")\n\n    # 2) Todas las señales de un planeta específico\n    sigs2 = SignalFinder(ctx2).find(signal_type=\"1\",)\n    print(f\"Planet 1810380816 signals: {len(sigs2)} files.\")\n\n    # 3) Calibraciones dark y flat de FGS1 en 50% de planetas\n    cals = CalibrationFinder(ctx3).find(instrument=\"FGS1\", calib_types=[\"dark\", \"flat\"])\n    print(f\"Found {len(cals)} calibration files.\")\n\n    # 4) Calibraciones dead y dark de FGS1 para un planeta\n    cals2 = CalibrationFinder(ctx4).find(instrument=\"FGS1\", calib_types=[\"dead\", \"dark\"], calib_set=\"2\")\n    print(f\"Planet 989956432 FGS1 calibrations: {len(cals2)} files.\")","metadata":{"trusted":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2025-07-04T01:10:54.233526Z","iopub.execute_input":"2025-07-04T01:10:54.233888Z","iopub.status.idle":"2025-07-04T01:10:57.173136Z","shell.execute_reply.started":"2025-07-04T01:10:54.233865Z","shell.execute_reply":"2025-07-04T01:10:57.17221Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"####  ChunkedSignalReaderWithQueue\n**Purpose**: Parallel parquet processing with Dask.  \n**Time Complexity**:  \n- **O(N / C)**, where:  \n  - `N`: Total files  \n  - `C`: Chunk size (e.g., 64)  \n**Key Notes**:  \n- Queue-based batching avoids memory overload.  \n- Dask parallelizes each chunk (`O(1)` per chunk).","metadata":{}},{"cell_type":"code","source":"%%time\n\"\"\"Dask sin computar \"\"\"\nimport numpy as np\nimport pandas as pd\nfrom pathlib import Path\nfrom tqdm import tqdm\nfrom dask.distributed import Queue\nimport dask.dataframe as dd\n\n\nclass ChunkedSignalReaderWithQueue:\n    def __init__(self, context, dask_client, queue_name=\"signal-queue\"):\n        self.ctx = context\n        self.client = dask_client\n        self.queue = Queue(name=queue_name, client=self.client)\n\n    def feed_queue(self, instrument, signal_type, chunk_size):\n        paths = SignalFinder(self.ctx).find(instrument=instrument, signal_type=signal_type)\n        total_chunks = int(np.ceil(len(paths) / chunk_size))\n        self.total_chunks = total_chunks  # Store for tqdm in processing\n        with tqdm(total=total_chunks, desc=\"📤 Enqueuing chunks\", unit=\"chunk\") as pbar:\n            for i in range(0, len(paths), chunk_size):\n                batch_paths = paths[i:i + chunk_size]\n                try:\n                    self.queue.put(batch_paths)\n                except Exception as e:\n                    print(f\"❌ Error queueing batch: {e}\")\n                pbar.update(1)\n\n    def process_queue(self):\n        processed = 0\n        with tqdm(total=getattr(self, 'total_chunks', None), desc=\"📥 Processing chunks\", unit=\"chunk\") as pbar:\n            while True:\n                try:\n                    batch_paths = self.queue.get(timeout=5)\n                    ddf = dd.read_parquet(batch_paths, engine=\"pyarrow\")\n                    result = ddf.columns[:3]  # Light dummy access\n                    _ = result  # avoid print\n                    processed += 1\n                    pbar.update(1)\n                except Exception:\n                    print(\"✅ Finished all queued chunks.\")\n                    break\n\n\nif __name__ == \"__main__\":\n    #from DaskClusterManager import DaskClusterManager\n\n    # ✅ Parámetros\n    JSON_PATH = \"ariel_ultralight.json\"\n    TRUNCATE_PLANETS = None\n    INSTRUMENT = \"AIRS-CH0\"\n    SIGNAL_TYPE = \"0\"\n    CHUNK_SIZE = 64 # More size less time\n    QUEUE_NAME = \"signal-queue\"\n\n    ctx = ArielContext(JSON_PATH, truncate_ids=TRUNCATE_PLANETS)\n    client = DaskClusterManager().get_client()\n\n    reader = ChunkedSignalReaderWithQueue(ctx, client, queue_name=QUEUE_NAME)\n    reader.feed_queue(INSTRUMENT, SIGNAL_TYPE, CHUNK_SIZE)\n    reader.process_queue()\n\n#16 ch = 8:53min\n#32 ch = 4:10min\n#64 ch = 2:11min","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-04T00:42:55.179782Z","iopub.execute_input":"2025-07-04T00:42:55.180071Z","iopub.status.idle":"2025-07-04T00:45:06.384049Z","shell.execute_reply.started":"2025-07-04T00:42:55.180043Z","shell.execute_reply":"2025-07-04T00:45:06.383204Z"},"_kg_hide-input":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n\"\"\"Sin dask\"\"\"\n\nimport numpy as np\nimport pandas as pd\nimport queue\nimport threading\nfrom pathlib import Path\nfrom tqdm import tqdm\nimport concurrent.futures\n\n\nclass ChunkedSignalReaderMultiprocess:\n    def __init__(self, context, queue_size=50):\n        self.ctx = context\n        self.paths = []\n        self.queue = queue.Queue(maxsize=queue_size)\n\n    def feed_queue(self, instrument, signal_type, chunk_size):\n        from time import sleep\n        self.paths = SignalFinder(self.ctx).find(instrument=instrument, signal_type=signal_type)\n        total_chunks = int(np.ceil(len(self.paths) / chunk_size))\n\n        def producer():\n            for i in tqdm(range(0, len(self.paths), chunk_size), desc=\"📤 Enqueuing chunks\", unit=\"chunk\"):\n                batch_paths = self.paths[i:i + chunk_size]\n                try:\n                    dfs = [pd.read_parquet(p) for p in batch_paths]\n                    combined = pd.concat(dfs, ignore_index=True)\n                    self.queue.put(combined)  # bloquea si está llena\n                except Exception as e:\n                    print(f\"❌ Error loading batch: {e}\")\n\n            # señal de cierre\n            for _ in range(3):\n                self.queue.put(None)\n\n        threading.Thread(target=producer, daemon=True).start()\n\n    def process_queue(self, num_workers=3):\n        def worker(worker_id):\n            while True:\n                df = self.queue.get()\n                if df is None:\n                    break\n                print(f\"🧪 [Worker {worker_id}] Head:\\n\", df.head(2))\n                self.queue.task_done()\n\n        with tqdm(total=self.queue.qsize(), desc=\"📥 Processing chunks\", unit=\"chunk\") as pbar:\n            with concurrent.futures.ThreadPoolExecutor(max_workers=num_workers) as executor:\n                futures = [executor.submit(worker, wid) for wid in range(num_workers)]\n                self.queue.join()\n\n\nif __name__ == \"__main__\":\n    # ✅ Parámetros\n    JSON_PATH = \"ariel_ultralight.json\"\n    TRUNCATE_PLANETS = None\n    INSTRUMENT = \"AIRS-CH0\"\n    SIGNAL_TYPE = \"0\"\n    CHUNK_SIZE = 2\n\n    ctx = ArielContext(JSON_PATH, truncate_ids=TRUNCATE_PLANETS)\n\n    reader = ChunkedSignalReaderMultiprocess(ctx)\n    reader.feed_queue(INSTRUMENT, SIGNAL_TYPE, CHUNK_SIZE)\n    reader.process_queue(num_workers=3)\n","metadata":{"trusted":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Maintenance Tips  \n1. **Optimize Chunk Sizes**: Larger chunks (e.g., 64) reduce Dask overhead.  \n2. **Pre-filter Data**: Use `truncate_ids`/`planet_id` to limit scans.  \n3. **Cache Contexts**: Reuse `ArielContext` for repeated queries.  \n4. **Consult Dask** documentation if a method is deprecated, consult the Distributed section for further details","metadata":{}},{"cell_type":"code","source":"%%time\n\"\"\"TESTING PURPOSES: This class needs to be optimized in the workers memory to process all the queue\"\"\"\nimport numpy as np\nimport plotly.graph_objects as go\nfrom tqdm import tqdm\nimport pandas as pd\nimport os\nfrom pathlib import Path\n\nclass BoundedSignalVisualizer:\n    def __init__(self, context, output_dir, chunk_size):\n        self.ctx = context\n        self.chunk_size = chunk_size\n        self.output_dir = Path(output_dir)\n        os.makedirs(self.output_dir, exist_ok=True)\n    \n    def _safe_sum(self, signal):\n        \"\"\"Safe summation handling different array dimensions\"\"\"\n        if signal.ndim == 1:\n            return signal\n        return signal.sum(axis=tuple(range(1, signal.ndim)))\n    \n    def process_signals(self, instrument, signal_type):\n        \"\"\"Process signals in chunks and calculate min/max bounds\"\"\"\n        paths = SignalFinder(self.ctx).find(instrument=instrument, signal_type=signal_type)\n        total_chunks = int(np.ceil(len(paths) / self.chunk_size))\n        \n        all_min, all_max, all_time = [], [], []\n        \n        with tqdm(total=total_chunks, desc=f\"Processing {instrument}\", leave=False) as pbar:\n            for i in range(0, len(paths), self.chunk_size):\n                batch = paths[i:i + self.chunk_size]\n                chunk_data = []\n                \n                for path in batch:\n                    try:\n                        signal = pd.read_parquet(path).values.astype(np.float64)\n                        chunk_data.append(self._safe_sum(signal))\n                    except Exception as e:\n                        continue\n                \n                if chunk_data:\n                    chunk_stack = np.array(chunk_data)\n                    time_points = np.arange(chunk_stack.shape[1])\n                    all_min.append(np.min(chunk_stack, axis=0))\n                    all_max.append(np.max(chunk_stack, axis=0))\n                    all_time.append(time_points)\n                pbar.update(1)\n        \n        return all_time, all_min, all_max\n    \n    def save_bounded_plot(self, instrument, signal_type):\n        \"\"\"Generate and save interactive plot as HTML\"\"\"\n        try:\n            times, mins, maxs = self.process_signals(instrument, signal_type)\n            if not times:\n                return\n            \n            fig = go.Figure()\n            \n            for i, (t, min_vals, max_vals) in enumerate(zip(times, mins, maxs)):\n                if len(t) == 0:\n                    continue\n                \n                norm_factor = np.mean(max_vals) or 1\n                min_norm = min_vals / norm_factor\n                max_norm = max_vals / norm_factor\n                \n                fig.add_trace(go.Scatter(\n                    x=np.concatenate([t, t[::-1]]),\n                    y=np.concatenate([max_norm, min_norm[::-1]]),\n                    fill='toself',\n                    fillcolor=f'rgba(100, 150, 200, {0.3 + 0.7*(i/len(times))})',\n                    line=dict(color='rgba(255,255,255,0)'),\n                    name=f'Chunk {i+1}'\n                ))\n            \n            fig.update_layout(\n                title=f\"{instrument} Signal Bounds (Normalized)\",\n                xaxis_title=\"Time (frame index)\",\n                yaxis_title=\"Normalized Flux\",\n                height=700,\n                template=\"plotly_dark\"\n            )\n            \n            output_file = self.output_dir / f\"{instrument}_{signal_type}_bounds.html\"\n            fig.write_html(output_file)\n            \n        except Exception as e:\n            print(f\"Error processing {instrument}: {str(e)}\")\n\nif __name__ == \"__main__\":    \n    CONFIG = {\n        \"json_path\": \"ariel_ultralight.json\",\n        \"output_dir\": \"signal_plots\",\n        \"chunk_size\": 64,\n        \"truncate_ids\": 0.1,  # Truncate percentage, None for full dataset\n        \"instruments\": [\"AIRS-CH0\", \"FGS1\"],\n        \"signal_types\": [\"2\"]\n    }\n    \n    # ===== EXECUTION =====\n    ctx = ArielContext(CONFIG[\"json_path\"], truncate_ids=CONFIG[\"truncate_ids\"])\n    visualizer = BoundedSignalVisualizer(\n        ctx,\n        output_dir=CONFIG[\"output_dir\"],\n        chunk_size=CONFIG[\"chunk_size\"]\n    )\n    \n    for instrument in CONFIG[\"instruments\"]:\n        for signal_type in CONFIG[\"signal_types\"]:\n            print(f\"⚙️ Processing {instrument} signal {signal_type}...\")\n            visualizer.save_bounded_plot(instrument, signal_type)\n    \n    print(f\"✅ All plots saved to {CONFIG['output_dir']} directory\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-04T01:39:40.772499Z","iopub.execute_input":"2025-07-04T01:39:40.77292Z","iopub.status.idle":"2025-07-04T01:49:47.750592Z","shell.execute_reply.started":"2025-07-04T01:39:40.772884Z","shell.execute_reply":"2025-07-04T01:49:47.748806Z"},"_kg_hide-input":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n\"\"\"Testing Purposes\"\"\"\nimport numpy as np\nimport plotly.graph_objects as go\nfrom tqdm import tqdm\nimport pandas as pd\nimport os\nfrom pathlib import Path\n\nclass NonLinearityVisualizer:\n    def __init__(self, context, output_dir, sample_pixels):\n        self.ctx = context\n        self.output_dir = Path(output_dir)\n        self.sample_pixels = sample_pixels  # Número de píxeles a muestrear\n        os.makedirs(self.output_dir, exist_ok=True)\n    \n    def load_calibration_data(self, planet_id, instrument):\n        \"\"\"Carga los coeficientes de no linealidad para un planeta e instrumento\"\"\"\n        calib_files = CalibrationFinder(self.ctx).find(\n            instrument=instrument,\n            calib_types=[\"linear_corr\"],\n            calib_set=\"0\"\n        )\n        \n        # Filtrar por planeta específico\n        planet_files = [f for f in calib_files if f\"/{planet_id}/\" in f]\n        \n        if not planet_files:\n            return None\n            \n        try:\n            df = pd.read_parquet(planet_files[0])\n            return df.values.astype(np.float64)\n        except Exception as e:\n            print(f\"Error loading {planet_files[0]}: {str(e)}\")\n            return None\n    \n    def generate_response_curves(self, coefficients, pixel_indices=None):\n        \"\"\"Genera curvas de respuesta usando los coeficientes polinomiales\"\"\"\n        if pixel_indices is None:\n            # Selecciona píxeles aleatorios si no se especifican\n            rng = np.random.default_rng()\n            pixel_indices = rng.choice(\n                coefficients.shape[1], \n                size=min(self.sample_pixels, coefficients.shape[1]), \n                replace=False\n            )\n        \n        curves = {}\n        x = np.linspace(0, 1, 100)  # Rango normalizado de electrones\n        \n        for idx in pixel_indices:\n            # Los coeficientes están en orden de mayor a menor grado\n            poly = np.poly1d(coefficients[:, idx])\n            y = poly(x)\n            curves[f\"Pixel_{idx}\"] = (x, y)\n            \n        return curves\n    \n    def save_nonlinearity_plot(self, planet_id, instrument):\n        \"\"\"Genera y guarda el gráfico de no linealidad\"\"\"\n        coefficients = self.load_calibration_data(planet_id, instrument)\n        if coefficients is None:\n            return\n            \n        # Tomamos los primeros 5 píxeles para el ejemplo\n        pixel_indices = range(min(5, coefficients.shape[1]))\n        curves = self.generate_response_curves(coefficients, pixel_indices)\n        \n        fig = go.Figure()\n        \n        for name, (x, y) in curves.items():\n            fig.add_trace(go.Scatter(\n                x=x,\n                y=y,\n                name=name,\n                mode='lines',\n                line=dict(width=2)\n            ))\n        \n        fig.update_layout(\n            title=f\"Curva de Respuesta No Lineal<br>{instrument} - Planeta {planet_id}\",\n            xaxis_title=\"Electrones (normalizado)\",\n            yaxis_title=\"Señal del pixel\",\n            legend_title=\"Píxeles\",\n            template=\"plotly_white\",\n            height=600\n        )\n        \n        output_file = self.output_dir / f\"nonlinearity_{instrument}_{planet_id}.html\"\n        fig.write_html(output_file)\n\nif __name__ == \"__main__\":\n    # ===== CONFIGURACIÓN =====\n    CONFIG = {\n        \"json_path\": \"ariel_ultralight.json\",\n        \"output_dir\": \"nonlinearity_plots\",\n        \"sample_planets\": 3,  # Número de planetas a muestrear\n        \"sample_pixels\": 5,   # Píxeles por gráfico\n        \"instruments\": [\"AIRS-CH0\", \"FGS1\"]\n    }\n    \n    # ===== EJECUCIÓN =====\n    ctx = ArielContext(CONFIG[\"json_path\"])\n    visualizer = NonLinearityVisualizer(\n        ctx,\n        output_dir=CONFIG[\"output_dir\"],\n        sample_pixels=CONFIG[\"sample_pixels\"]\n    )\n    \n    # Seleccionar subconjunto de planetas\n    planet_ids = ctx.planets[:CONFIG[\"sample_planets\"]]\n    \n    for planet_id in planet_ids:\n        for instrument in CONFIG[\"instruments\"]:\n            print(f\"⚙️ Procesando {instrument} para planeta {planet_id}...\")\n            visualizer.save_nonlinearity_plot(planet_id, instrument)\n    \n    print(f\"✅ Todos los gráficos guardados en {CONFIG['output_dir']}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-04T01:53:58.825975Z","iopub.execute_input":"2025-07-04T01:53:58.826445Z","iopub.status.idle":"2025-07-04T01:54:02.816985Z","shell.execute_reply.started":"2025-07-04T01:53:58.826415Z","shell.execute_reply":"2025-07-04T01:54:02.815771Z"},"_kg_hide-input":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## IF you found this notebook usefull, please upvote 🥳","metadata":{}}]}