{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":101849,"databundleVersionId":12846694,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":12342681,"sourceType":"datasetVersion","datasetId":7780951}],"dockerImageVersionId":31041,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# # GDL-PINN for Ariel Data Challenge 2025\n# \n# This notebook implements a Geometric Deep Learning approach with Physics-Informed Neural Networks\n# for exoplanet spectroscopy analysis.","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-01T12:56:56.296502Z","iopub.execute_input":"2025-07-01T12:56:56.296764Z","iopub.status.idle":"2025-07-01T12:56:56.30093Z","shell.execute_reply.started":"2025-07-01T12:56:56.296741Z","shell.execute_reply":"2025-07-01T12:56:56.300105Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Environment Setup and Imports\nimport torch.nn.functional as F\n\nimport os\nimport sys\nimport gc\nimport warnings\nwarnings.filterwarnings('ignore')\n\nimport numpy as np\nimport pandas as pd\nfrom pathlib import Path\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom tqdm.auto import tqdm\nimport pyarrow.parquet as pq\nfrom typing import Dict, List, Tuple, Optional\n\n# Check GPU availability\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nprint(f\"Using device: {device}\")\nif torch.cuda.is_available():\n    print(f\"GPU: {torch.cuda.get_device_name(0)}\")\n    print(f\"Memory: {torch.cuda.get_device_properties(0).total_memory / 1e9:.2f} GB\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-01T17:20:57.658024Z","iopub.execute_input":"2025-07-01T17:20:57.658298Z","iopub.status.idle":"2025-07-01T17:20:57.664847Z","shell.execute_reply.started":"2025-07-01T17:20:57.65828Z","shell.execute_reply":"2025-07-01T17:20:57.66408Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!python3 --version\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-01T17:21:01.875326Z","iopub.execute_input":"2025-07-01T17:21:01.875678Z","iopub.status.idle":"2025-07-01T17:21:02.041406Z","shell.execute_reply.started":"2025-07-01T17:21:01.875644Z","shell.execute_reply":"2025-07-01T17:21:02.040123Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nprint(f\"PyTorch version: {torch.__version__}\")\nprint(f\"CUDA available: {torch.cuda.is_available()}\")\nprint(f\"CUDA version: {torch.version.cuda}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-01T17:21:04.401362Z","iopub.execute_input":"2025-07-01T17:21:04.402038Z","iopub.status.idle":"2025-07-01T17:21:04.406375Z","shell.execute_reply.started":"2025-07-01T17:21:04.402012Z","shell.execute_reply":"2025-07-01T17:21:04.405742Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 1: Downgrade PyTorch to 2.3.0 + CUDA 11.8\n!pip install -q torch==2.3.0+cu118 torchvision torchaudio --index-url https://download.pytorch.org/whl/cu118\n\n# Step 2: Install PyTorch Geometric dependencies\n!pip install -q torch-scatter torch-sparse -f https://data.pyg.org/whl/torch-2.3.0+cu118.html\n\n# Step 3: Install torch-geometric\n!pip install -q torch-geometric\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-01T17:21:21.073964Z","iopub.execute_input":"2025-07-01T17:21:21.074294Z","iopub.status.idle":"2025-07-01T17:21:50.592426Z","shell.execute_reply.started":"2025-07-01T17:21:21.074278Z","shell.execute_reply":"2025-07-01T17:21:50.591605Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch_geometric\nimport torch_scatter\nimport torch_sparse\n\nfrom torch_geometric.data import Data\nfrom torch_geometric.nn import MessagePassing, global_mean_pool\n\nprint(\"✅ PyTorch version:\", torch.__version__)\nprint(\"✅ torch_geometric:\", torch_geometric.__version__)\nprint(\"✅ torch_scatter:\", torch_scatter.__version__)\nprint(\"✅ torch_sparse:\", torch_sparse.__version__)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-01T17:21:13.947504Z","iopub.execute_input":"2025-07-01T17:21:13.948208Z","iopub.status.idle":"2025-07-01T17:21:13.965526Z","shell.execute_reply.started":"2025-07-01T17:21:13.948188Z","shell.execute_reply":"2025-07-01T17:21:13.964591Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sys\nimport torch\n\nprint(\"🐍 Python version:\", sys.version)\nprint(\"🔥 PyTorch version:\", torch.__version__)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-01T17:21:50.631075Z","iopub.execute_input":"2025-07-01T17:21:50.631286Z","iopub.status.idle":"2025-07-01T17:21:50.644765Z","shell.execute_reply.started":"2025-07-01T17:21:50.631272Z","shell.execute_reply":"2025-07-01T17:21:50.644016Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# List wheels (just to check)\n!ls /kaggle/input/pyg-cu118-torch260-zip\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-01T15:56:40.913837Z","iopub.execute_input":"2025-07-01T15:56:40.914135Z","iopub.status.idle":"2025-07-01T15:56:41.047148Z","shell.execute_reply.started":"2025-07-01T15:56:40.914111Z","shell.execute_reply":"2025-07-01T15:56:41.046195Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install /kaggle/input/pyg-cu118-torch260-zip/torch_scatter-*.whl\n!pip install /kaggle/input/pyg-cu118-torch260-zip/torch_sparse-*.whl\n!pip install /kaggle/input/pyg-cu118-torch260-zip/torch_geometric-*.whl\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-01T15:57:10.999206Z","iopub.execute_input":"2025-07-01T15:57:10.999893Z","iopub.status.idle":"2025-07-01T15:57:21.978671Z","shell.execute_reply.started":"2025-07-01T15:57:10.999867Z","shell.execute_reply":"2025-07-01T15:57:21.977554Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch_geometric\nprint(\"✅ PyTorch Geometric:\", torch_geometric.__version__)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-01T15:59:32.78699Z","iopub.execute_input":"2025-07-01T15:59:32.787733Z","iopub.status.idle":"2025-07-01T15:59:32.791535Z","shell.execute_reply.started":"2025-07-01T15:59:32.78771Z","shell.execute_reply":"2025-07-01T15:59:32.790637Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nfrom torch_geometric.data import Data\n\n# A mini graph with 3 nodes and 2 edges: 0→1, 1→2\nedge_index = torch.tensor([[0, 1], [1, 2]], dtype=torch.long)\nedge_index = edge_index.t().contiguous()\n\n# Each node has 2 features\nx = torch.tensor([[1, 2], [3, 4], [5, 6]], dtype=torch.float)\n\ndata = Data(x=x, edge_index=edge_index)\nprint(data)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-01T16:00:24.836678Z","iopub.execute_input":"2025-07-01T16:00:24.837391Z","iopub.status.idle":"2025-07-01T16:00:24.847783Z","shell.execute_reply.started":"2025-07-01T16:00:24.837371Z","shell.execute_reply":"2025-07-01T16:00:24.846804Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch_geometric\nfrom torch_geometric.nn import MessagePassing, global_mean_pool\nfrom torch_geometric.data import Data, DataLoader\n\nprint(f\"PyTorch Geometric version: {torch_geometric.__version__}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-01T16:04:12.926507Z","iopub.execute_input":"2025-07-01T16:04:12.927301Z","iopub.status.idle":"2025-07-01T16:04:12.931462Z","shell.execute_reply.started":"2025-07-01T16:04:12.927277Z","shell.execute_reply":"2025-07-01T16:04:12.930657Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# PART 1: MODEL ARCHITECTURE COMPONENTS\n# ============================================\nimport numpy as np\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nfrom torch_geometric.data import Data\nfrom torch_geometric.nn import MessagePassing, global_mean_pool\nfrom typing import Dict, Any\nfrom typing import Dict, Tuple, List, Any\nfrom pathlib import Path\nimport torch.nn.functional as F\n\n\n\n\nclass SpectralGraphConstructor:\n    \"\"\"Constructs graph representations from spectral data\"\"\"\n    \n    def __init__(self, k_neighbors: int = 10, wavelength_threshold: float = 0.1):\n        self.k_neighbors = k_neighbors\n        self.wavelength_threshold = wavelength_threshold\n    \n    def construct_spectral_graph(self, \n                                wavelengths: np.ndarray,\n                                spectral_data: np.ndarray,\n                                instrument_info: Dict) -> Data:\n        n_channels = len(wavelengths)\n        \n        # Node features\n        node_features = []\n        for i, wl in enumerate(wavelengths):\n            features = [\n                wl,\n                np.log10(wl) if wl > 0 else 0,  # Handle zero wavelengths\n                spectral_data[i] if len(spectral_data) > i else 0,\n                1.0 if wl < 0.8 else 0.0,  # FGS1\n                1.0 if 1.95 <= wl <= 3.9 else 0.0,  # AIRS\n            ]\n            node_features.append(features)\n        \n        x = torch.tensor(node_features, dtype=torch.float32)\n        \n        # Edge construction\n        edge_index = []\n        edge_attr = []\n        \n        # Ensure we create edges even for small graphs\n        for i in range(n_channels):\n            # Connect to next few channels (increase range if needed)\n            for j in range(i+1, min(i+10, n_channels)):  # Increased from 5 to 10\n                wl_diff = abs(wavelengths[i] - wavelengths[j])\n                # Increased threshold to ensure more edges\n                if wl_diff < self.wavelength_threshold * 2:  # Doubled threshold\n                    edge_index.extend([[i, j], [j, i]])\n                    edge_features = [\n                        wl_diff,\n                        1.0 / (1.0 + wl_diff),\n                        float(wavelengths[i] < 0.8 and wavelengths[j] < 0.8),\n                    ]\n                    edge_attr.extend([edge_features, edge_features])\n        \n        # If no edges were created, create a minimal connected graph\n        if len(edge_index) == 0:\n            # Connect each node to its neighbors\n            for i in range(n_channels - 1):\n                edge_index.extend([[i, i+1], [i+1, i]])\n                edge_features = [0.01, 0.99, 1.0]  # Default edge features\n                edge_attr.extend([edge_features, edge_features])\n        \n        # Convert to tensors with proper shape\n        if len(edge_index) > 0:\n            edge_index = torch.tensor(edge_index, dtype=torch.long).t().contiguous()\n            edge_attr = torch.tensor(edge_attr, dtype=torch.float32)\n        else:\n            # Fallback: create empty but properly shaped tensors\n            edge_index = torch.zeros((2, 0), dtype=torch.long)\n            edge_attr = torch.zeros((0, 3), dtype=torch.float32)\n        \n        return Data(x=x, edge_index=edge_index, edge_attr=edge_attr)\n\n\n# Also update the model forward method to handle edge cases better\ndef fixed_forward(self, batch_data: Dict) -> Tuple[torch.Tensor, torch.Tensor]:\n    \"\"\"Fixed forward method with better error handling\"\"\"\n    \n    try:\n        # Process FGS1\n        fgs1_features = self.fgs1_encoder(batch_data['fgs1_features'])\n        \n        # Process AIRS with graph\n        wavelengths = batch_data['wavelengths']\n        airs_features = batch_data['airs_features']\n        \n        # Construct graph\n        graph = self.graph_constructor.construct_spectral_graph(\n            wavelengths.cpu().numpy(),\n            airs_features.cpu().numpy(),\n            {'instrument': 'AIRS-CH0'}\n        )\n        \n        # Move to device\n        x = graph.x.to(self.device if hasattr(self, 'device') else device)\n        edge_index = graph.edge_index.to(self.device if hasattr(self, 'device') else device)\n        edge_attr = graph.edge_attr.to(self.device if hasattr(self, 'device') else device)\n        \n        # Check if we have edges\n        if edge_index.shape[1] > 0:\n            # Apply graph convolutions\n            for conv in self.spectral_convs:\n                x = conv(x, edge_index, edge_attr)\n                x = F.relu(x)\n                x = F.dropout(x, p=0.1, training=self.training)\n        else:\n            # If no edges, just apply linear transformations\n            for conv in self.spectral_convs:\n                x = conv.lin(x)\n                x = F.relu(x)\n                x = F.dropout(x, p=0.1, training=self.training)\n        \n        # Global pooling\n        airs_pooled = x.mean(dim=0, keepdim=True)  # Simple mean if no batch\n        \n        # Combine features\n        combined = torch.cat([airs_pooled, fgs1_features.unsqueeze(0)], dim=-1)\n        decoded = self.physics_decoder(combined)\n        \n        # Generate predictions\n        mean_pred = self.mean_head(decoded)\n        log_uncertainty = self.uncertainty_head(decoded)\n        uncertainty = torch.exp(log_uncertainty).clamp(min=1e-6, max=100)\n        \n        return mean_pred.squeeze(0), uncertainty.squeeze(0)\n    \n    except Exception as e:\n        print(f\"Error in forward pass: {e}\")\n        # Return default predictions\n        return torch.ones(284), torch.ones(284) * 10.0\n\nclass PhysicsInformedSpectralConv(MessagePassing):\n    \"\"\"Physics-informed spectral convolution layer\"\"\"\n    \n    def __init__(self, in_channels: int, out_channels: int, physics_dim: int = 16):\n        super().__init__(aggr='add')\n        \n        self.lin = nn.Linear(in_channels, out_channels)\n        self.lin_edge = nn.Linear(3, physics_dim)\n        self.lin_physics = nn.Linear(physics_dim + out_channels, out_channels)\n        \n        self.opacity_kernel = nn.Parameter(torch.randn(physics_dim))\n        self.absorption_kernel = nn.Parameter(torch.randn(physics_dim))\n        \n    def forward(self, x, edge_index, edge_attr):\n        x = self.lin(x)\n        return self.propagate(edge_index, x=x, edge_attr=edge_attr)\n    \n    def message(self, x_j, edge_attr):\n        edge_embedding = self.lin_edge(edge_attr)\n        opacity_weight = torch.sigmoid(torch.matmul(edge_embedding, self.opacity_kernel))\n        absorption_weight = torch.tanh(torch.matmul(edge_embedding, self.absorption_kernel))\n        message = x_j * opacity_weight.unsqueeze(-1) * (1 - absorption_weight.unsqueeze(-1))\n        return message\n    \n    def update(self, aggr_out, x):\n        combined = torch.cat([x, aggr_out], dim=-1)\n        return self.lin_physics(combined)\n\n     \nclass GeometricPINNExoplanet(nn.Module):\n    \"\"\"Main GDL-PINN model architecture with fixed dimensions\"\"\"\n    \n    def __init__(self, \n                 n_wavelengths: int = 283,\n                 hidden_dim: int = 256,\n                 n_graph_layers: int = 4,\n                 dropout: float = 0.1):\n        super().__init__()\n        \n        self.n_wavelengths = n_wavelengths\n        self.hidden_dim = hidden_dim\n        \n        self.graph_constructor = SpectralGraphConstructor()\n        \n        # Spectral encoder with graph convolutions\n        self.spectral_convs = nn.ModuleList([\n            PhysicsInformedSpectralConv(\n                5 if i == 0 else hidden_dim,\n                hidden_dim,\n                physics_dim=32\n            ) for i in range(n_graph_layers)\n        ])\n        \n        # FGS1 encoder - output hidden_dim//2 = 128\n        self.fgs1_encoder = nn.Sequential(\n            nn.Linear(1024, hidden_dim),\n            nn.LayerNorm(hidden_dim),\n            nn.ReLU(),\n            nn.Dropout(dropout),\n            nn.Linear(hidden_dim, hidden_dim // 2)  # Outputs 128\n        )\n        \n        # Physics decoder - expects hidden_dim (256) + hidden_dim//2 (128) = 384\n        self.physics_decoder = nn.Sequential(\n            nn.Linear(hidden_dim + hidden_dim // 2, hidden_dim),  # 384 -> 256\n            nn.LayerNorm(hidden_dim),\n            nn.ReLU(),\n            nn.Dropout(dropout),\n            nn.Linear(hidden_dim, hidden_dim),  # 256 -> 256\n        )\n        \n        # Output heads for mean and uncertainty\n        self.mean_head = nn.Linear(hidden_dim, n_wavelengths + 1)  # 256 -> 284\n        self.uncertainty_head = nn.Linear(hidden_dim, n_wavelengths + 1)  # 256 -> 284\n    \n    def forward(self, batch_data: Dict) -> Tuple[torch.Tensor, torch.Tensor]:\n        try:\n            # Process FGS1\n            fgs1_features = self.fgs1_encoder(batch_data['fgs1_features'])  # Shape: [128]\n            \n            # Process AIRS with graph\n            wavelengths = batch_data['wavelengths']\n            airs_features = batch_data['airs_features']\n            \n            # Construct graph\n            graph = self.graph_constructor.construct_spectral_graph(\n                wavelengths.cpu().numpy(),\n                airs_features.cpu().numpy(),\n                {'instrument': 'AIRS-CH0'}\n            )\n            \n            # Move to device\n            x = graph.x.to(device)\n            edge_index = graph.edge_index.to(device)\n            edge_attr = graph.edge_attr.to(device)\n            \n            # Apply graph convolutions\n            if edge_index.shape[1] > 0:\n                for conv in self.spectral_convs:\n                    x = conv(x, edge_index, edge_attr)\n                    x = F.relu(x)\n                    x = F.dropout(x, p=0.1, training=self.training)\n            else:\n                # Fallback for no edges\n                for conv in self.spectral_convs:\n                    x = conv.lin(x)\n                    x = F.relu(x)\n                    x = F.dropout(x, p=0.1, training=self.training)\n            \n            # Global pooling - ensure we get hidden_dim features\n            # x shape: [n_nodes, hidden_dim]\n            airs_pooled = x.mean(dim=0)  # Shape: [hidden_dim=256]\n            \n            # Ensure correct dimensions\n            if airs_pooled.dim() == 1:\n                airs_pooled = airs_pooled.unsqueeze(0)  # Shape: [1, 256]\n            \n            if fgs1_features.dim() == 1:\n                fgs1_features = fgs1_features.unsqueeze(0)  # Shape: [1, 128]\n            \n            # Combine features\n            # airs_pooled: [1, 256], fgs1_features: [1, 128]\n            combined = torch.cat([airs_pooled, fgs1_features], dim=-1)  # Shape: [1, 384]\n            \n            # Decode\n            decoded = self.physics_decoder(combined)  # Shape: [1, 256]\n            \n            # Generate predictions\n            mean_pred = self.mean_head(decoded)  # Shape: [1, 284]\n            log_uncertainty = self.uncertainty_head(decoded)  # Shape: [1, 284]\n            uncertainty = torch.exp(log_uncertainty).clamp(min=1e-6, max=100)\n            \n            # Remove batch dimension\n            return mean_pred.squeeze(0), uncertainty.squeeze(0)\n            \n        except Exception as e:\n            print(f\"Error in forward pass: {e}\")\n            import traceback\n            traceback.print_exc()\n            # Return default predictions\n            return torch.ones(284).to(device), torch.ones(284).to(device) * 10.0\n\n\n# Alternative simpler fix - just add this projection layer after physics_decoder\nclass SimplifiedGeometricPINNExoplanet(nn.Module):\n    \"\"\"Simplified version with dimension fix\"\"\"\n    \n    def __init__(self, \n                 n_wavelengths: int = 283,\n                 hidden_dim: int = 256,\n                 n_graph_layers: int = 4,\n                 dropout: float = 0.1):\n        super().__init__()\n        \n        self.n_wavelengths = n_wavelengths\n        self.hidden_dim = hidden_dim\n        \n        self.graph_constructor = SpectralGraphConstructor()\n        \n        # Spectral encoder\n        self.spectral_encoder = nn.Sequential(\n            nn.Linear(5, hidden_dim),\n            nn.ReLU(),\n            nn.Dropout(dropout),\n            nn.Linear(hidden_dim, hidden_dim)\n        )\n        \n        # FGS1 encoder\n        self.fgs1_encoder = nn.Sequential(\n            nn.Linear(1024, hidden_dim),\n            nn.ReLU(),\n            nn.Dropout(dropout),\n            nn.Linear(hidden_dim, hidden_dim)\n        )\n        \n        # Simple fusion\n        self.fusion = nn.Sequential(\n            nn.Linear(hidden_dim * 2, hidden_dim),\n            nn.LayerNorm(hidden_dim),\n            nn.ReLU(),\n            nn.Dropout(dropout)\n        )\n        \n        # Output heads\n        self.mean_head = nn.Linear(hidden_dim, n_wavelengths + 1)\n        self.uncertainty_head = nn.Linear(hidden_dim, n_wavelengths + 1)\n    \n    def forward(self, batch_data: Dict) -> Tuple[torch.Tensor, torch.Tensor]:\n        # Simple encoding without graph convolutions\n        fgs1_feat = self.fgs1_encoder(batch_data['fgs1_features'])\n        \n        # Simple AIRS encoding\n        airs_feat = self.spectral_encoder(\n            torch.cat([\n                batch_data['wavelengths'].unsqueeze(-1),\n                torch.log10(batch_data['wavelengths'] + 1e-6).unsqueeze(-1),\n                batch_data['airs_features'].unsqueeze(-1),\n                (batch_data['wavelengths'] < 0.8).float().unsqueeze(-1),\n                ((batch_data['wavelengths'] >= 1.95) & (batch_data['wavelengths'] <= 3.9)).float().unsqueeze(-1)\n            ], dim=-1).mean(dim=0)  # Average over wavelengths\n        )\n        \n        # Ensure correct dimensions\n        if fgs1_feat.dim() == 1:\n            fgs1_feat = fgs1_feat.unsqueeze(0)\n        if airs_feat.dim() == 1:\n            airs_feat = airs_feat.unsqueeze(0)\n        \n        # Combine\n        combined = torch.cat([fgs1_feat, airs_feat], dim=-1)\n        fused = self.fusion(combined)\n        \n        # Predictions\n        mean_pred = self.mean_head(fused)\n        uncertainty = torch.exp(self.uncertainty_head(fused)).clamp(min=1e-6, max=100)\n        \n        return mean_pred.squeeze(0), uncertainty.squeeze(0)\n# ============================================\n# PART 2: DATA PROCESSING FUNCTIONS\n# ============================================\n\ndef load_competition_data(data_dir: Path, planet_id: str) -> Dict:\n    \"\"\"Load all data for a single planet\"\"\"\n    \n    planet_dir = data_dir / planet_id\n    \n    # Load ADC info\n    adc_info = pd.read_csv('/kaggle/input/ariel-data-challenge-2025/adc_info.csv')\n    \n    # Load FGS1 data\n    fgs1_files = sorted(planet_dir.glob('FGS1_signal_*.parquet'))\n    fgs1_data = []\n    \n    fgs1_gain = float(adc_info['FGS1_adc_gain'].iloc[0])\n    fgs1_offset = float(adc_info['FGS1_adc_offset'].iloc[0])\n    \n    for file in fgs1_files:\n        df = pd.read_parquet(file)\n        signal = df.values * fgs1_gain + fgs1_offset\n        fgs1_data.append(signal)\n\n    \n    # Load AIRS data\n    \n    airs_files = sorted(planet_dir.glob('AIRS-CH0_signal_*.parquet'))\n    airs_data = []\n    \n    airs_gain = float(adc_info['AIRS-CH0_adc_gain'].iloc[0])\n    airs_offset = float(adc_info['AIRS-CH0_adc_offset'].iloc[0])\n    \n    for file in airs_files:\n        df = pd.read_parquet(file)\n        signal = df.values * airs_gain + airs_offset\n        airs_data.append(signal)\n\n\n    \n    # Load calibration\n    calibration = {}\n    for instrument in ['FGS1', 'AIRS-CH0']:\n        cal_dir = planet_dir / f'{instrument}_calibration'\n        cal_data = {}\n        for cal_type in ['dark', 'flat']:\n            file_path = cal_dir / f'{cal_type}.parquet'\n            if file_path.exists():\n                cal_array = pd.read_parquet(file_path).values\n                cal_data[cal_type] = cal_array\n        calibration[instrument] = cal_data\n    \n    return {\n        'fgs1_raw': fgs1_data,\n        'airs_raw': airs_data,\n        'calibration': calibration\n    }\n\ndef apply_basic_calibration(raw_data: np.ndarray, calibration: Dict) -> np.ndarray:\n    \"\"\"Apply basic calibration to raw data\"\"\"\n    calibrated = raw_data.copy()\n\n    # Dark subtraction\n    if 'dark' in calibration and calibration['dark'] is not None:\n        calibrated = calibrated - calibration['dark']\n\n    # Flat fielding\n    if 'flat' in calibration and calibration['flat'] is not None:\n        flat_array = calibration['flat']\n        if isinstance(flat_array, np.ndarray):\n            flat_norm = flat_array / np.median(flat_array)\n            calibrated = calibrated / flat_norm\n        else:\n            print(\"⚠️ Warning: Flat field calibration is not an array\")\n\n    return calibrated\n\n\ndef extract_features(planet_data: Dict) -> Dict:\n    \"\"\"Extract features from planet data\"\"\"\n    \n    # Process FGS1\n    fgs1_features = []\n    for raw in planet_data['fgs1_raw']:\n        cal = apply_basic_calibration(raw, planet_data['calibration']['FGS1'])\n        # Reshape from flattened format\n        cal_reshaped = cal.reshape(-1, 32, 32)\n        # Simple feature: average over spatial dimensions\n        light_curve = cal_reshaped.mean(axis=(1, 2))\n        # Normalize\n        light_curve = light_curve / np.median(light_curve)\n        # Take statistics\n        features = [\n            light_curve.mean(),\n            light_curve.std(),\n            light_curve.min(),\n            light_curve.max(),\n            1.0 - light_curve.min(),  # Transit depth\n        ]\n        fgs1_features.extend(features)\n    \n    # Pad or truncate to fixed size\n    fgs1_features = np.array(fgs1_features)\n    if len(fgs1_features) < 1024:\n        fgs1_features = np.pad(fgs1_features, (0, 1024 - len(fgs1_features)))\n    else:\n        fgs1_features = fgs1_features[:1024]\n    \n    # Process AIRS\n    airs_features = []\n    for raw in planet_data['airs_raw']:\n        cal = apply_basic_calibration(raw, planet_data['calibration']['AIRS-CH0'])\n        # Reshape from flattened format\n        cal_reshaped = cal.reshape(-1, 32, 356)\n        # Average spectrum\n        spectrum = cal_reshaped.mean(axis=(0, 1))[:283]\n        airs_features.append(spectrum)\n    \n    # Average over observations\n    if len(airs_features) > 0:\n        airs_features = np.mean(airs_features, axis=0)\n    else:\n        airs_features = np.ones(283)  # Default\n    \n    return {\n        'fgs1_features': torch.tensor(fgs1_features, dtype=torch.float32),\n        'airs_features': torch.tensor(airs_features, dtype=torch.float32)\n    }\n\n# ============================================\n# PART 3: INFERENCE AND SUBMISSION\n# ============================================\n\ndef process_single_planet(planet_id: str, model: nn.Module, data_dir: Path) -> Dict:\n    \"\"\"Process a single planet and return predictions\"\"\"\n    \n    try:\n        # Load planet data\n        planet_data = load_competition_data(data_dir, planet_id)\n        \n        # Extract features\n        features = extract_features(planet_data)\n        \n        # Load wavelengths\n        wavelengths = pd.read_csv('/kaggle/input/ariel-data-challenge-2025/wavelengths.csv')\n        \n        # Prepare batch\n        batch_data = {\n            'fgs1_features': features['fgs1_features'].to(device),\n            'airs_features': features['airs_features'].to(device),\n            'wavelengths': torch.tensor(wavelengths.values[:283, 0], dtype=torch.float32).to(device),\n        }\n        \n        # Get predictions\n        with torch.no_grad():\n            pred_mean, pred_uncertainty = model(batch_data)\n        \n        return {\n            'success': True,\n            'mean': pred_mean.cpu().numpy(),\n            'uncertainty': pred_uncertainty.cpu().numpy()\n        }\n        \n    except Exception as e:\n        print(f\"Error processing {planet_id}: {e}\")\n        # Return default predictions\n        return {\n            'success': False,\n            'mean': np.ones(284),\n            'uncertainty': np.ones(284) * 10.0\n        }\n# Fixed version of the submission function\n\ndef create_submission(model: nn.Module, test_dir: Path, output_path: str = 'submission.csv'):\n    \"\"\"Create competition submission file\"\"\"\n    \n    # Load test data info\n    test_star_info = pd.read_csv('/kaggle/input/ariel-data-challenge-2025/test_star_info.csv')\n    \n    # Initialize results\n    results = []\n    \n    print(f\"Processing {len(test_star_info)} test planets...\")\n    \n    # Process each planet\n    for idx, row in tqdm(test_star_info.iterrows(), total=len(test_star_info)):\n        # FIX: Convert planet_id to string properly\n        planet_id_raw = row['planet_id']\n        if isinstance(planet_id_raw, float):\n            planet_id = str(int(planet_id_raw))\n        else:\n            planet_id = str(planet_id_raw)\n        \n        # Get predictions\n        pred_result = process_single_planet(planet_id, model, test_dir)\n        \n        # Format results - planet_id is already a string now\n        result = {'planet_id': planet_id}\n        \n        # Add spectral predictions\n        pred_mean = pred_result['mean']\n        pred_unc = pred_result['uncertainty']\n        \n        # AIRS channels (283)\n        for i in range(283):\n            result[f'wl_{i+1}'] = float(pred_mean[i])\n        \n        # FGS1 channel\n        result['wl_284'] = float(pred_mean[283])\n        \n        # Uncertainties - using correct column names\n        for i in range(283):\n            result[f'sigma_{i+1}'] = float(pred_unc[i])\n        \n        # Note: FGS1 uncertainty might not be needed - check sample_submission.csv\n        \n        results.append(result)\n        \n        # Memory cleanup every 100 planets\n        if idx % 100 == 0:\n            gc.collect()\n            torch.cuda.empty_cache()\n    \n    # Create DataFrame\n    submission_df = pd.DataFrame(results)\n    \n    # Verify column order matches sample submission\n    sample_sub = pd.read_csv('/kaggle/input/ariel-data-challenge-2025/sample_submission.csv')\n    \n    # Make sure we have exactly the columns in sample_sub\n    submission_df = submission_df[sample_sub.columns]\n    \n    # Save submission\n    submission_df.to_csv(output_path, index=False)\n    \n    return submission_df\n\n# ============================================\n# PART 4: OPTIMIZATION UTILITIES\n# ============================================\n\ndef verify_submission(submission_df: pd.DataFrame) -> bool:\n    \"\"\"Verify submission format\"\"\"\n    sample = pd.read_csv('/kaggle/input/ariel-data-challenge-2025/sample_submission.csv')\n    \n    checks = {\n        'shape_match': submission_df.shape == sample.shape,\n        'columns_match': list(submission_df.columns) == list(sample.columns),\n        'no_nulls': submission_df.isnull().sum().sum() == 0,\n        'planet_ids_are_strings': submission_df['planet_id'].dtype == 'object',\n        'values_are_numeric': all(submission_df.iloc[:, 1:].dtypes == 'float64')\n    }\n    \n    print(\"Submission verification:\")\n    for check, passed in checks.items():\n        print(f\"  {check}: {'✓' if passed else '✗'}\")\n    \n    return all(checks.values())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-01T17:22:40.617228Z","iopub.execute_input":"2025-07-01T17:22:40.617495Z","iopub.status.idle":"2025-07-01T17:22:40.666847Z","shell.execute_reply.started":"2025-07-01T17:22:40.617476Z","shell.execute_reply":"2025-07-01T17:22:40.666169Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\nadc_info = pd.read_csv('/kaggle/input/ariel-data-challenge-2025/adc_info.csv')\nprint(adc_info.dtypes)\nprint(adc_info.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-01T17:18:28.191938Z","iopub.execute_input":"2025-07-01T17:18:28.192396Z","iopub.status.idle":"2025-07-01T17:18:28.201805Z","shell.execute_reply.started":"2025-07-01T17:18:28.192375Z","shell.execute_reply":"2025-07-01T17:18:28.201077Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_df = create_submission(model, test_dir, output_path='submission.csv')\nverify_submission(submission_df)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-01T17:09:11.173833Z","iopub.execute_input":"2025-07-01T17:09:11.174098Z","iopub.status.idle":"2025-07-01T17:09:22.94739Z","shell.execute_reply.started":"2025-07-01T17:09:11.17408Z","shell.execute_reply":"2025-07-01T17:09:22.946872Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\n\n# Automatically use GPU if available, else fallback to CPU\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-01T17:22:46.970617Z","iopub.execute_input":"2025-07-01T17:22:46.970887Z","iopub.status.idle":"2025-07-01T17:22:46.974893Z","shell.execute_reply.started":"2025-07-01T17:22:46.97087Z","shell.execute_reply":"2025-07-01T17:22:46.974161Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Initialize model\nprint(\"Initializing GDL-PINN model...\")\nmodel = GeometricPINNExoplanet(\n    n_wavelengths=283,\n    hidden_dim=256,\n    n_graph_layers=4,\n    dropout=0.0  # No dropout during inference\n)\nmodel.to(device)\nmodel.eval()\n\n# Note: In a real submission, you would load pre-trained weights here\n# For now, we'll use random initialization (this won't give good results)\nprint(f\"Model parameters: {sum(p.numel() for p in model.parameters()):,}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-01T17:22:57.975132Z","iopub.execute_input":"2025-07-01T17:22:57.975902Z","iopub.status.idle":"2025-07-01T17:22:58.011857Z","shell.execute_reply.started":"2025-07-01T17:22:57.97587Z","shell.execute_reply":"2025-07-01T17:22:58.011101Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\nfrom tqdm.notebook import tqdm\nimport gc\nimport torch\n\n\n\n# Path to test dataset\ntest_dir = Path('/kaggle/input/ariel-data-challenge-2025/test')\n\n# Run model inference on test set and generate submission\nsubmission_df = create_submission(model, test_dir, output_path='submission.csv')\n\nprint(f\"Submission created with {len(submission_df)} predictions\")\nprint(\"\\nFirst 5 predictions:\")\nprint(submission_df.head())\n\n# Check if submission is valid before uploading to Kaggle\nif verify_submission(submission_df):\n    print(\"\\n✓ Submission format is valid!\")\nelse:\n    print(\"\\n✗ Warning: Submission format issues detected!\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-01T17:23:01.006281Z","iopub.execute_input":"2025-07-01T17:23:01.006827Z","iopub.status.idle":"2025-07-01T17:23:07.662765Z","shell.execute_reply.started":"2025-07-01T17:23:01.006804Z","shell.execute_reply":"2025-07-01T17:23:07.662034Z"}},"outputs":[],"execution_count":null}]}