{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":56537,"databundleVersionId":8877088,"sourceType":"competition"},{"sourceId":8409068,"sourceType":"datasetVersion","datasetId":5004471},{"sourceId":184657777,"sourceType":"kernelVersion"}],"dockerImageVersionId":30732,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import tensorflow as tf\nimport os\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom tensorflow.keras import layers, regularizers\nfrom tensorflow.keras.callbacks import EarlyStopping\nimport keras_tuner as kt\nfrom sklearn.metrics import r2_score\nimport pandas as pd\n\n# Load the saved statistics from the added dataset\nmean_std_file_path = '/kaggle/input/data-preparation-atmospheric-physics/mean_std_stats.npz'\nsaved_stats = np.load(mean_std_file_path)\nmean_x = tf.constant(saved_stats['mean_x'])\nstd_x = tf.constant(saved_stats['std_x'])\nmean_targets = tf.constant(saved_stats['mean_targets'])\nstd_targets = tf.constant(saved_stats['std_targets'])\n\n# Define the normalization function using the loaded statistics\ndef normalize_data(x, targets):\n    x = (x - mean_x) / std_x\n    targets = (targets - mean_targets) / std_targets\n    return x, targets\n\n# Function to create dataset\ndef _parse_function(example_proto):\n    feature_description = {\n        'x': tf.io.FixedLenFeature([556], tf.float32),\n        'targets': tf.io.FixedLenFeature([368], tf.float32),\n    }\n    proto = tf.io.parse_single_example(example_proto, feature_description)\n    return proto['x'], proto['targets']\n\ndef create_normalized_dataset(file_list, batch_size=2048, shuffle_buffer_size=100000):\n    dataset = tf.data.TFRecordDataset(file_list)\n    dataset = dataset.map(_parse_function, num_parallel_calls=tf.data.experimental.AUTOTUNE)\n    dataset = dataset.map(lambda x, targets: normalize_data(x, targets), num_parallel_calls=tf.data.experimental.AUTOTUNE)\n    dataset = dataset.shuffle(buffer_size=shuffle_buffer_size, seed=42)\n    dataset = dataset.batch(batch_size)\n    dataset = dataset.prefetch(buffer_size=tf.data.experimental.AUTOTUNE)\n    return dataset\n\n# Paths for TFRecord files\nTFRec_path = '/kaggle/input/leap-train-tfrecords'\ntrain_files = [os.path.join(TFRec_path, f\"train_{i:03d}.tfrec\") for i in range(50)]\nvalid_files = [os.path.join(TFRec_path, \"train_100.tfrec\")]\n\n# Create normalized training and validation datasets\ntrain_dataset = create_normalized_dataset(train_files, shuffle_buffer_size=100000)\nvalid_dataset = create_normalized_dataset(valid_files, shuffle_buffer_size=1)\n\n# Data pipeline optimization\ndef optimize_dataset(dataset):\n    dataset = dataset.cache()  # Cache the dataset in memory\n    dataset = dataset.shuffle(buffer_size=1000)  # Shuffle the dataset\n    dataset = dataset.prefetch(buffer_size=tf.data.experimental.AUTOTUNE)  # Prefetch for better performance\n    return dataset\n\n# Optimize datasets\ntrain_dataset = optimize_dataset(train_dataset)\nvalid_dataset = optimize_dataset(valid_dataset)\n\n### Cell for visualization to check normalization process ###\n# Load and Normalize the Data\ndef create_dataset(file_list):\n    dataset = tf.data.TFRecordDataset(file_list)\n    dataset = dataset.map(_parse_function, num_parallel_calls=tf.data.experimental.AUTOTUNE)\n    return dataset\n\n# Create the full dataset without batching or shuffling for visualization\nfull_train_dataset = create_dataset(train_files)\n\n# Normalize the dataset\nnormalized_dataset = full_train_dataset.map(lambda x, targets: normalize_data(x, targets))\n\n# Visualize the normalization process using a subset of the data\nx_values = []\ntargets_values = []\n\nfor x, targets in normalized_dataset.take(1000):  # Use a subset of 1000 examples for visualization\n    x_values.append(x.numpy())\n    targets_values.append(targets.numpy())\n\nx_values = np.array(x_values)\ntargets_values = np.array(targets_values)\n\n# Plot the mean and standard deviation of normalized data\nmean_x_normalized = np.mean(x_values, axis=0)\nstd_x_normalized = np.std(x_values, axis=0)\n\nmean_targets_normalized = np.mean(targets_values, axis=0)\nstd_targets_normalized = np.std(targets_values, axis=0)\n\n# Plot the mean and standard deviation of the normalized features\nplt.figure(figsize=(12, 6))\n\nplt.subplot(1, 2, 1)\nplt.plot(mean_x_normalized, label='Mean of normalized x')\nplt.axhline(y=0, color='r', linestyle='--')\nplt.xlabel('Feature Index')\nplt.ylabel('Mean')\nplt.title('Mean of Normalized Features')\nplt.legend()\n\nplt.subplot(1, 2, 2)\nplt.plot(std_x_normalized, label='Std of normalized x')\nplt.axhline(y=1, color='r', linestyle='--')\nplt.xlabel('Feature Index')\nplt.ylabel('Standard Deviation')\nplt.title('Standard Deviation of Normalized Features')\nplt.legend()\n\nplt.tight_layout()\nplt.show()\n\n# Plot the mean and standard deviation of the normalized targets\nplt.figure(figsize=(12, 6))\n\nplt.subplot(1, 2, 1)\nplt.plot(mean_targets_normalized, label='Mean of normalized targets')\nplt.axhline(y=0, color='r', linestyle='--')\nplt.xlabel('Target Index')\nplt.ylabel('Mean')\nplt.title('Mean of Normalized Targets')\nplt.legend()\n\nplt.subplot(1, 2, 2)\nplt.plot(std_targets_normalized, label='Std of normalized targets')\nplt.axhline(y=1, color='r', linestyle='--')\nplt.xlabel('Target Index')\nplt.ylabel('Standard Deviation')\nplt.title('Standard Deviation of Normalized Targets')\nplt.legend()\n\nplt.tight_layout()\nplt.show()\n\n# Define custom callback to compute R² score\nclass R2ScoreCallback(tf.keras.callbacks.Callback):\n    def __init__(self, validation_data):\n        super().__init__()\n        self.validation_data = validation_data\n\n    def on_epoch_end(self, epoch, logs=None):\n        val_pred = self.model.predict(self.validation_data[0])\n        val_true = self.validation_data[1]\n        r2 = r2_score(val_true, val_pred)\n        print(f\" - val_r2: {r2:.4f}\")\n        logs['val_r2'] = r2\n\n# Define the model-building function for Keras Tuner\ndef build_model(hp):\n    inputs = tf.keras.Input(shape=(556,))\n    \n    # Hyperparameter tuning for the number of units in the first Dense layer\n    x = layers.Dense(\n        units=hp.Int('units_1', min_value=256, max_value=512, step=64), \n        activation='relu', \n        kernel_regularizer=regularizers.l2(hp.Float('l2_1', min_value=1e-5, max_value=1e-3, sampling='LOG'))\n    )(inputs)\n    x = layers.BatchNormalization()(x)\n    x = layers.Dropout(hp.Float('dropout_1', min_value=0.3, max_value=0.7, step=0.1))(x)\n    \n    # Hyperparameter tuning for the number of units in the second Dense layer\n    x = layers.Dense(\n        units=hp.Int('units_2', min_value=128, max_value=256, step=64), \n        activation='relu', \n        kernel_regularizer=regularizers.l2(hp.Float('l2_2', min_value=1e-5, max_value=1e-3, sampling='LOG'))\n    )(x)\n    x = layers.BatchNormalization()(x)\n    x = layers.Dropout(hp.Float('dropout_2', min_value=0.3, max_value=0.7, step=0.1))(x)\n    \n    outputs = layers.Dense(368, activation='linear', dtype='float32')(x)\n    \n    model = tf.keras.Model(inputs=inputs, outputs=outputs)\n    \n    # Hyperparameter tuning for the learning rate\n    model.compile(\n        optimizer=tf.keras.optimizers.Adam(hp.Float('learning_rate', min_value=1e-4, max_value=1e-2, sampling='LOG')),\n        loss='mse',\n        metrics=['mse']\n    )\n    return model\n\n# Define the tuner\ntuner = kt.Hyperband(\n    build_model,\n    objective='val_loss',\n    max_epochs=50,\n    factor=3,\n    directory='/kaggle/working/tuner_results',\n    project_name='tuning'\n)\n\n# Early stopping callback to stop training when val_loss doesn't improve\nearly_stopping = EarlyStopping(\n    monitor='val_loss', \n    patience=10,\n    restore_best_weights=True\n)\n\n# Prepare validation data for R² score callback\nval_data = next(iter(valid_dataset))\nval_x, val_y = val_data\n\n# Define R² score callback\nr2_callback = R2ScoreCallback(validation_data=(val_x, val_y))\n\n# Search for the best hyperparameters\ntuner.search(\n    train_dataset,\n    validation_data=valid_dataset,\n    epochs=50,\n    callbacks=[early_stopping]\n)\n\n# Get the optimal hyperparameters\nbest_hps = tuner.get_best_hyperparameters(num_trials=1)[0]\n\n# Build the model with the best hyperparameters\nbest_model = build_model(best_hps)\n\n# Train the model\nhistory = best_model.fit(\n    train_dataset,\n    validation_data=valid_dataset,\n    epochs=100,\n    callbacks=[early_stopping, r2_callback],\n    verbose=1\n)\n\n# Save the trained model\nmodel_save_path = '/kaggle/working/best_model_tuned.h5'\nbest_model.save(model_save_path)\nprint(f\"Model saved to {model_save_path}\")\n\n# Plotting learning curves\nhistory_df = pd.DataFrame(history.history)\nhistory_df.loc[:, ['loss', 'val_loss']].plot()\nplt.show()\n\n# Plotting R² score\nif 'val_r2' in history_df.columns:\n    history_df.loc[:, ['val_r2']].plot()\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-06-22T02:59:07.021036Z","iopub.execute_input":"2024-06-22T02:59:07.021409Z","iopub.status.idle":"2024-06-22T05:15:39.763377Z","shell.execute_reply.started":"2024-06-22T02:59:07.021382Z","shell.execute_reply":"2024-06-22T05:15:39.762303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}