{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":101849,"databundleVersionId":12846694,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-06-28T10:44:35.240622Z","iopub.execute_input":"2025-06-28T10:44:35.240916Z","iopub.status.idle":"2025-06-28T10:44:58.245703Z","shell.execute_reply.started":"2025-06-28T10:44:35.240893Z","shell.execute_reply":"2025-06-28T10:44:58.244663Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 1: Setup and Correctly Loading the Metadata","metadata":{}},{"cell_type":"code","source":"# Import essential libraries\nimport numpy as np\nimport pandas as pd\nimport os # Essential for handling file paths\nimport matplotlib.pyplot as plt\nfrom tqdm.notebook import tqdm # For progress bars\n\n# --- Define Paths (Kaggle Environment) ---\n# This is the standard path in Kaggle competitions\nBASE_DIR = '/kaggle/input/ariel-data-challenge-2025/'\n\n# Check if the directory exists to provide a helpful error\nif not os.path.exists(BASE_DIR):\n    # If not in Kaggle, update this path to where you saved the data\n    print(\"Kaggle directory not found. Trying local 'ariel-data-challenge-2025/'...\")\n    BASE_DIR = 'ariel-data-challenge-2025/'\n    if not os.path.exists(BASE_DIR):\n        print(\"Error: Data directory not found. Please download the data and check the BASE_DIR path.\")\n        # Stop execution if data is not found\n        # In a real script, you might raise an exception here.\n        exit()\n\nTRAIN_DIR = os.path.join(BASE_DIR, 'train')\nTEST_DIR = os.path.join(BASE_DIR, 'test')\n\n# --- Load the Metadata CSVs ---\ntry:\n    # This file contains the planet IDs and their ground truth target fluxes\n    train_labels_df = pd.read_csv(os.path.join(BASE_DIR, 'train.csv'))\n    \n    # This file contains information about the wavelengths for the target fluxes\n    wavelengths_df = pd.read_csv(os.path.join(BASE_DIR, 'wavelengths.csv'))\n    \n    # The sample submission tells us the required format for our predictions\n    sample_submission_df = pd.read_csv(os.path.join(BASE_DIR, 'sample_submission.csv'))\n\n    print(\"Metadata loaded successfully!\")\n    print(\"\\nTraining labels shape:\", train_labels_df.shape)\n    print(train_labels_df.head())\n    \nexcept FileNotFoundError as e:\n    print(f\"Error loading metadata: {e}\")\n    print(\"Please ensure your data directory is structured correctly.\")\n    exit()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T11:01:13.495397Z","iopub.execute_input":"2025-06-28T11:01:13.495796Z","iopub.status.idle":"2025-06-28T11:01:13.848649Z","shell.execute_reply.started":"2025-06-28T11:01:13.495771Z","shell.execute_reply":"2025-06-28T11:01:13.846804Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 2: Loading the Real Signal Data (Parquet Files)","metadata":{}},{"cell_type":"code","source":"def load_and_process_planet_data(planet_id, data_dir):\n    \"\"\"\n    Loads the time-series signal data for a single planet and engineers basic features.\n    \"\"\"\n    try:\n        # Construct the path to the planet's signal file.\n        # NOTE: The file might have a name like 'FGS1_signal_0.parquet' or similar.\n        # We use a wildcard (*) to find the parquet file.\n        # This example assumes we're using the FGS1 instrument data.\n        # You should explore using AIRS data as well.\n        \n        # A more robust way is to find the file explicitly\n        planet_folder = os.path.join(data_dir, str(planet_id))\n        signal_file_path = None\n        for file in os.listdir(planet_folder):\n            if \"FGS1_signal\" in file and file.endswith('.parquet'):\n                signal_file_path = os.path.join(planet_folder, file)\n                break\n        \n        if signal_file_path is None:\n            # print(f\"Warning: No FGS1 signal file found for planet {planet_id}\")\n            return None\n\n        # Load the parquet file\n        signal_df = pd.read_parquet(signal_file_path)\n        \n        # --- Baseline Feature Engineering ---\n        # The signal_df contains time-series data. Each column is a detector pixel (or similar).\n        # For a simple baseline, let's compute the mean and std dev for each column.\n        # This flattens the time-series into a single feature vector.\n        features_mean = signal_df.mean().values\n        features_std = signal_df.std().values\n        \n        # Combine the features into a single array\n        features = np.concatenate([features_mean, features_std])\n        \n        return features\n        \n    except Exception as e:\n        # print(f\"Error processing planet {planet_id}: {e}\")\n        return None\n\n# --- Process all planets in the training set ---\n# Use tqdm for a nice progress bar\ntqdm.pandas(desc=\"Processing Planets\")\n\n# Get a list of planet IDs from our labels file\nplanet_ids = train_labels_df['planet_id'].unique()\n\n# Create features for all planets\nall_features = []\ncorresponding_planet_ids = []\n\nfor planet_id in tqdm(planet_ids, desc=\"Creating Training Features\"):\n    features = load_and_process_planet_data(planet_id, TRAIN_DIR)\n    if features is not None:\n        all_features.append(features)\n        corresponding_planet_ids.append(planet_id)\n\n# Create our final feature matrix (X)\nX_train_processed = pd.DataFrame(all_features)\nX_train_processed['planet_id'] = corresponding_planet_ids\n\n# Merge with labels to ensure correct alignment\n# Set index to planet_id for easy merging\ntrain_labels_indexed = train_labels_df.set_index('planet_id')\nX_train_final = X_train_processed.set_index('planet_id').join(train_labels_indexed, how='inner')\n\n# Separate features (X) and targets (y)\ny_train_final = X_train_final[train_labels_df.columns.drop('planet_id')]\nX_train_final = X_train_final.drop(columns=y_train_final.columns)\n\nprint(\"\\nFeature matrix created.\")\nprint(\"Shape of X_train:\", X_train_final.shape)\nprint(\"Shape of y_train:\", y_train_final.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T11:01:46.770338Z","iopub.execute_input":"2025-06-28T11:01:46.771281Z","iopub.status.idle":"2025-06-28T11:41:44.805483Z","shell.execute_reply.started":"2025-06-28T11:01:46.77125Z","shell.execute_reply":"2025-06-28T11:41:44.80378Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 3: Preprocessing, Modeling, and Training ","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nimport tensorflow as tf\n\n# --- Data Preprocessing ---\nscaler = StandardScaler()\nX_scaled = scaler.fit_transform(X_train_final)\n\n# Split for validation\nX_train, X_val, y_train, y_val = train_test_split(\n    X_scaled, y_train_final.values, test_size=0.2, random_state=42\n)\n\n# --- Build the Model (Multi-output) ---\ndef build_multi_output_model(input_shape, output_shape):\n    \"\"\"Builds a neural network for multi-output regression.\"\"\"\n    model = tf.keras.Sequential([\n        tf.keras.layers.Dense(256, activation='relu', input_shape=[input_shape]),\n        tf.keras.layers.Dropout(0.3),\n        tf.keras.layers.Dense(128, activation='relu'),\n        tf.keras.layers.Dropout(0.3),\n        # The output layer must have one neuron for each target wavelength\n        tf.keras.layers.Dense(output_shape)\n    ])\n\n    optimizer = tf.keras.optimizers.Adam(learning_rate=0.001)\n    model.compile(loss='mean_squared_error', optimizer=optimizer)\n    return model\n\n# Create the model\ninput_shape = X_train.shape[1]\noutput_shape = y_train.shape[1] # Number of target wavelengths\nmodel = build_multi_output_model(input_shape, output_shape)\nmodel.summary()\n\n\n# --- Train the Model ---\nearly_stopping = tf.keras.callbacks.EarlyStopping(monitor='val_loss', patience=10, restore_best_weights=True)\n\nhistory = model.fit(\n    X_train, y_train,\n    epochs=100,\n    validation_data=(X_val, y_val),\n    verbose=1,\n    callbacks=[early_stopping]\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T11:41:51.74032Z","iopub.execute_input":"2025-06-28T11:41:51.741699Z","iopub.status.idle":"2025-06-28T11:42:11.800547Z","shell.execute_reply.started":"2025-06-28T11:41:51.741639Z","shell.execute_reply":"2025-06-28T11:42:11.799379Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 4: Prediction and Submission ","metadata":{}},{"cell_type":"code","source":"# --- Process the Test Set ---\n# Get a list of test planet IDs from the sample submission file\ntest_planet_ids = sample_submission_df['planet_id'].unique()\n\nall_test_features = []\ncorresponding_test_ids = []\n\nfor planet_id in tqdm(test_planet_ids, desc=\"Creating Test Features\"):\n    features = load_and_process_planet_data(planet_id, TEST_DIR)\n    if features is not None:\n        all_test_features.append(features)\n        corresponding_test_ids.append(planet_id)\n\n# Create the final test feature matrix\nX_test_processed = pd.DataFrame(all_test_features)\n\n# Scale the test features using the *same scaler* fitted on the training data\nX_test_scaled = scaler.transform(X_test_processed)\n\n# --- Generate Predictions ---\nprint(\"Generating test predictions...\")\npredictions = model.predict(X_test_scaled)\n\n# --- Create Submission File ---\n# Create a DataFrame with the predictions\npred_df = pd.DataFrame(predictions, columns=y_train_final.columns)\npred_df['planet_id'] = corresponding_test_ids\n\n# Reorder columns to match submission format if necessary\npred_df = pred_df[['planet_id'] + list(y_train_final.columns)]\n\n# Save to csv\npred_df.to_csv('submission.csv', index=False)\n\nprint(\"\\nSubmission file created successfully!\")\nprint(pred_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T11:42:31.713656Z","iopub.execute_input":"2025-06-28T11:42:31.714027Z","iopub.status.idle":"2025-06-28T11:42:34.888408Z","shell.execute_reply.started":"2025-06-28T11:42:31.714003Z","shell.execute_reply":"2025-06-28T11:42:34.88701Z"}},"outputs":[],"execution_count":null}]}