{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":56537,"databundleVersionId":8877088,"sourceType":"competition"}],"dockerImageVersionId":30733,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd \nimport os\nimport polars as pl\nimport tensorflow as tf\nfrom sklearn.preprocessing import StandardScaler\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Dropout\nimport matplotlib.pyplot as plt\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#INPUT_DATA = '/kaggle/input/leap-atmospheric-physics-ai-climsim'\n\n#train_df = pl.scan_csv(f'{INPUT_DATA}/train.csv')\n\n#test_df = pl.scan_csv(f'{INPUT_DATA}/test.csv')\n\n#sample_submission_df = pl.scan_csv(f'{INPUT_DATA}/sample_submission.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def csv_to_tfrecord(csv_file_path, tfrecord_file_path, chunk_size=10000):\n    writer = tf.io.TFRecordWriter(tfrecord_file_path)\n\n    for chunk in pd.read_csv(csv_file_path, chunksize=chunk_size):\n        for index, row in chunk.iterrows():\n            features = {\n                'features': tf.train.Feature(float_list=tf.train.FloatList(value=row.drop('sample_id').values)),\n                'label': tf.train.Feature(float_list=tf.train.FloatList(value=[row['sample_id']])),\n            }\n            example = tf.train.Example(features=tf.train.Features(feature=features))\n            writer.write(example.SerializeToString())\n\n    writer.close()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert train.csv to TFRecord\ncsv_to_tfrecord('/kaggle/input/leap-atmospheric-physics-ai-climsim/train.csv', 'train.tfrecord')\n\n# Convert test.csv to TFRecord\ncsv_to_tfrecord('/kaggle/input/leap-atmospheric-physics-ai-climsim/test.csv', 'test.tfrecord')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def _parse_function(proto):\n    feature_description = {\n        'features': tf.io.FixedLenFeature([924], tf.float32),\n        'label': tf.io.FixedLenFeature([1], tf.float32),\n    }\n    parsed_features = tf.io.parse_single_example(proto, feature_description)\n    return parsed_features['features'], parsed_features['label']\n\ndef create_dataset(tfrecord_file_path, batch_size=10000):\n    dataset = tf.data.TFRecordDataset(tfrecord_file_path)\n    dataset = dataset.map(_parse_function, num_parallel_calls=tf.data.experimental.AUTOTUNE)\n    dataset = dataset.batch(batch_size)\n    dataset = dataset.prefetch(tf.data.experimental.AUTOTUNE)\n    return dataset\n\ntrain_dataset = create_dataset('train.tfrecord', batch_size=10000)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a validation dataset from a small portion of the training data\nval_df = pl.read_csv('train.csv', n_rows=20000)\nX_val, y_val = preprocess_batch(val_df, scaler=scaler, fit_scaler=True)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.callbacks import ModelCheckpoint\n\n# Create a distributed strategy\nstrategy = tf.distribute.MirroredStrategy()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the number of epochs per increment\nepochs_per_increment = 5\ntotal_epochs = 50","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a directory to save the model checkpoints\ncheckpoint_dir = './checkpoints'\nos.makedirs(checkpoint_dir, exist_ok=True)\ncheckpoint_callback = ModelCheckpoint(\n    filepath=os.path.join(checkpoint_dir, 'model_epoch_{epoch:02d}.h5'),\n    save_freq='epoch',  # Save at the end of every epoch\n    save_weights_only=True,\n    verbose=1\n)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Enable mixed precision training\npolicy = tf.keras.mixed_precision.Policy('mixed_float16')\ntf.keras.mixed_precision.set_global_policy(policy)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with strategy.scope():\n    # Build the model\n    model = Sequential([\n        Dense(128, activation='relu', input_shape=(X_val.shape[1],)),\n        Dropout(0.2),\n        Dense(64, activation='relu'),\n        Dropout(0.2),\n        Dense(32, activation='relu'),\n        Dense(1)\n    ])\n\n    # Compile the model\n    model.compile(optimizer='adam', loss='mean_squared_error')\n\n    for epoch in range(0, total_epochs, epochs_per_increment):\n        print(f'Training epoch range {epoch} to {epoch + epochs_per_increment - 1}')\n        model.fit(\n            train_dataset,\n            initial_epoch=epoch,\n            epochs=epoch + epochs_per_increment,\n            validation_data=(X_val, y_val),\n            steps_per_epoch=steps_per_epoch,\n            callbacks=[checkpoint_callback]\n        )","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Generate predictions for the validation set\ny_pred = model.predict(X_val)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the weighting file\nsample_submission_df = pl.read_csv(submission_file_path)\nweights = sample_submission_df.drop(['sample_id']).to_pandas().values","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Apply weights to the predictions\ny_pred_weighted = y_pred * weights","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate the custom R-squared metric\nSS_res = np.sum((y_val - y_pred_weighted) ** 2)\nSS_tot = np.sum((y_val - np.mean(y_val)) ** 2)\nR2_score = 1 - (SS_res / SS_tot)\n\nprint(f'Custom Weighted R2 Score: {R2_score}')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot training & validation loss values\nimport matplotlib.pyplot as plt\nplt.figure(figsize=(12, 6))\nplt.plot(history.history['loss'], label='Train Loss')\nplt.plot(history.history['val_loss'], label='Validation Loss')\nplt.title('Model Loss')\nplt.ylabel('Loss')\nplt.xlabel('Epoch')\nplt.legend(loc='upper right')\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Scatter plot of actual vs predicted values\nplt.figure(figsize=(12, 6))\nplt.scatter(y_val, y_pred_weighted, alpha=0.5)\nplt.xlabel('Actual Values')\nplt.ylabel('Predicted Values')\nplt.title('Actual vs Predicted Values')\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Process the test data in chunks and generate predictions\ntest_dataset = create_dataset('test.tfrecord', batch_size=10000)\ntest_predictions = model.predict(test_dataset)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Apply weights to the predictions\nweights_test = sample_submission_df.drop(columns=['sample_id']).to_pandas().values\ntest_predictions_weighted = test_predictions * weights_test\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a DataFrame for the submission\nsubmission_df = pl.DataFrame({\n    'sample_id': pl.read_csv(test_file_path)['sample_id'],\n    'prediction': test_predictions_weighted.flatten()\n})","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save the submission DataFrame to a CSV file\nsubmission_df.write_csv('submission.csv')\n\nprint(\"Submission file created successfully!\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Function to preprocess a batch of data\ndef preprocess_batch(df, scaler=None, fit_scaler=False):\n    X = df.drop(['sample_id']).to_pandas()\n    y = df['sample_id'].to_pandas().values\n\n    if fit_scaler and scaler:\n        X_scaled = scaler.fit_transform(X)\n    elif scaler:\n        X_scaled = scaler.transform(X)\n    else:\n        X_scaled = X.values\n\n    return X_scaled, y","metadata":{"execution":{"iopub.status.busy":"2024-06-20T12:54:10.319787Z","iopub.execute_input":"2024-06-20T12:54:10.320122Z","iopub.status.idle":"2024-06-20T12:54:10.351672Z","shell.execute_reply.started":"2024-06-20T12:54:10.320092Z","shell.execute_reply":"2024-06-20T12:54:10.350886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize the scaler\nscaler = StandardScaler()","metadata":{"execution":{"iopub.status.busy":"2024-06-20T12:54:13.134016Z","iopub.execute_input":"2024-06-20T12:54:13.134837Z","iopub.status.idle":"2024-06-20T12:54:13.139465Z","shell.execute_reply.started":"2024-06-20T12:54:13.134797Z","shell.execute_reply":"2024-06-20T12:54:13.138127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define file paths\ntrain_file_path = '/kaggle/input/leap-atmospheric-physics-ai-climsim/train.csv'\ntest_file_path = '/kaggle/input/leap-atmospheric-physics-ai-climsim/test.csv'\nsubmission_file_path = '/kaggle/input/leap-atmospheric-physics-ai-climsim/sample_submission.csv'","metadata":{"execution":{"iopub.status.busy":"2024-06-20T12:54:15.586554Z","iopub.execute_input":"2024-06-20T12:54:15.586905Z","iopub.status.idle":"2024-06-20T12:54:15.592422Z","shell.execute_reply.started":"2024-06-20T12:54:15.586877Z","shell.execute_reply":"2024-06-20T12:54:15.591544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the batch size\nbatch_size = 10000","metadata":{"execution":{"iopub.status.busy":"2024-06-20T12:54:17.270356Z","iopub.execute_input":"2024-06-20T12:54:17.270702Z","iopub.status.idle":"2024-06-20T12:54:17.274674Z","shell.execute_reply.started":"2024-06-20T12:54:17.270674Z","shell.execute_reply":"2024-06-20T12:54:17.273786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a generator to yield batches of data\ndef data_generator(file_path, batch_size=10000, scaler=None, fit_scaler=False):\n    for chunk in pl.read_csv(file_path, batch_size=batch_size):\n        X_batch, y_batch = preprocess_batch(chunk, scaler=scaler, fit_scaler=fit_scaler)\n        for i in range(0, len(X_batch), batch_size):\n            yield X_batch[i:i+batch_size], y_batch[i:i+batch_size]","metadata":{"execution":{"iopub.status.busy":"2024-06-20T12:54:19.912756Z","iopub.execute_input":"2024-06-20T12:54:19.913758Z","iopub.status.idle":"2024-06-20T12:54:19.920082Z","shell.execute_reply.started":"2024-06-20T12:54:19.913716Z","shell.execute_reply":"2024-06-20T12:54:19.919164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Training data generator\ntrain_generator = data_generator(train_file_path, batch_size=batch_size, scaler=scaler, fit_scaler=True)","metadata":{"execution":{"iopub.status.busy":"2024-06-20T12:54:21.200089Z","iopub.execute_input":"2024-06-20T12:54:21.200571Z","iopub.status.idle":"2024-06-20T12:54:21.20501Z","shell.execute_reply.started":"2024-06-20T12:54:21.200539Z","shell.execute_reply":"2024-06-20T12:54:21.20398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert the generator to a tf.data.Dataset\ndef generator_to_dataset(generator, output_signature):\n    return tf.data.Dataset.from_generator(\n        lambda: generator,\n        output_signature=output_signature\n    )\n\noutput_signature = (\n    tf.TensorSpec(shape=(None, 924), dtype=tf.float32),\n    tf.TensorSpec(shape=(None,), dtype=tf.float32)\n)\n\ntrain_dataset = generator_to_dataset(train_generator, output_signature)\ntrain_dataset = train_dataset.prefetch(tf.data.experimental.AUTOTUNE).cache()","metadata":{"execution":{"iopub.status.busy":"2024-06-20T12:54:23.447979Z","iopub.execute_input":"2024-06-20T12:54:23.448585Z","iopub.status.idle":"2024-06-20T12:54:24.083013Z","shell.execute_reply.started":"2024-06-20T12:54:23.448551Z","shell.execute_reply":"2024-06-20T12:54:24.082251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a validation dataset from a small portion of the training data\nval_df = pl.read_csv(train_file_path, n_rows=20000)\nX_val, y_val = preprocess_batch(val_df, scaler=scaler, fit_scaler=True)","metadata":{"execution":{"iopub.status.busy":"2024-06-20T12:54:26.610645Z","iopub.execute_input":"2024-06-20T12:54:26.611473Z","iopub.status.idle":"2024-06-20T12:54:28.878547Z","shell.execute_reply.started":"2024-06-20T12:54:26.611443Z","shell.execute_reply":"2024-06-20T12:54:28.877506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a distributed strategy\nstrategy = tf.distribute.MirroredStrategy()","metadata":{"execution":{"iopub.status.busy":"2024-06-20T12:54:29.022184Z","iopub.execute_input":"2024-06-20T12:54:29.022992Z","iopub.status.idle":"2024-06-20T12:54:29.030779Z","shell.execute_reply.started":"2024-06-20T12:54:29.022962Z","shell.execute_reply":"2024-06-20T12:54:29.029854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the batch size and number of epochs\nbatch_size = 1000  # Adjust batch size as needed\nnum_epochs = 50\nsave_period = 5 ","metadata":{"execution":{"iopub.status.busy":"2024-06-20T12:54:32.003302Z","iopub.execute_input":"2024-06-20T12:54:32.004117Z","iopub.status.idle":"2024-06-20T12:54:32.008198Z","shell.execute_reply.started":"2024-06-20T12:54:32.004085Z","shell.execute_reply":"2024-06-20T12:54:32.007182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate the number of steps per epoch\ntotal_samples = 10091520  # Replace with the actual number of samples in your training data\nsteps_per_epoch = int(np.ceil(total_samples / (batch_size * strategy.num_replicas_in_sync)))","metadata":{"execution":{"iopub.status.busy":"2024-06-20T12:54:33.725674Z","iopub.execute_input":"2024-06-20T12:54:33.726531Z","iopub.status.idle":"2024-06-20T12:54:33.731052Z","shell.execute_reply.started":"2024-06-20T12:54:33.726497Z","shell.execute_reply":"2024-06-20T12:54:33.730045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a directory to save the model checkpoints\ncheckpoint_dir = './checkpoints'\nos.makedirs(checkpoint_dir, exist_ok=True)\ncheckpoint_callback = tf.keras.callbacks.ModelCheckpoint(\n    filepath=os.path.join(checkpoint_dir, 'model_epoch_{epoch:02d}.weights.h5'),\n    save_freq='epoch',  # Save at the end of every epoch\n    save_weights_only=True,\n    verbose=1\n)","metadata":{"execution":{"iopub.status.busy":"2024-06-20T12:54:36.11749Z","iopub.execute_input":"2024-06-20T12:54:36.118371Z","iopub.status.idle":"2024-06-20T12:54:36.123205Z","shell.execute_reply.started":"2024-06-20T12:54:36.118337Z","shell.execute_reply":"2024-06-20T12:54:36.122494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Enable mixed precision training\npolicy = tf.keras.mixed_precision.Policy('mixed_float16')\ntf.keras.mixed_precision.set_global_policy(policy)","metadata":{"execution":{"iopub.status.busy":"2024-06-20T12:54:37.988462Z","iopub.execute_input":"2024-06-20T12:54:37.988811Z","iopub.status.idle":"2024-06-20T12:54:37.993612Z","shell.execute_reply.started":"2024-06-20T12:54:37.988785Z","shell.execute_reply":"2024-06-20T12:54:37.992554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train the model within the distributed strategy scope\nwith strategy.scope():\n    # Build the model\n    model = Sequential([\n        Dense(128, activation='relu', input_shape=(X_val.shape[1],)),\n        Dropout(0.2),\n        Dense(64, activation='relu'),\n        Dropout(0.2),\n        Dense(32, activation='relu'),\n        Dense(1)\n    ])\n\n    # Compile the model\n    model.compile(optimizer='adam', loss='mean_squared_error')\n\n    # Train the model\n    history = model.fit(\n        train_dataset,\n        epochs=num_epochs,\n        validation_data=(X_val, y_val),\n        steps_per_epoch=steps_per_epoch,\n        callbacks=[checkpoint_callback]\n    )","metadata":{"execution":{"iopub.status.busy":"2024-06-20T12:54:40.837347Z","iopub.execute_input":"2024-06-20T12:54:40.8377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Generate predictions for the validation set\ny_pred = model.predict(X_val)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the weighting file\nsample_submission_df = pl.read_csv('sample_submission.csv')\nweights = sample_submission_df.drop('sample_id').to_pandas().values","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Apply weights to the predictions\ny_pred_weighted = y_pred * weights","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate the custom R-squared metric\nSS_res = np.sum((y_val - y_pred_weighted) ** 2)\nSS_tot = np.sum((y_val - np.mean(y_val)) ** 2)\nR2_score = 1 - (SS_res / SS_tot)\n\nprint(f'Custom Weighted R2 Score: {R2_score}')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot training & validation loss values\nplt.figure(figsize=(12, 6))\nplt.plot(history.history['loss'], label='Train Loss')\nplt.plot(history.history['val_loss'], label='Validation Loss')\nplt.title('Model Loss')\nplt.ylabel('Loss')\nplt.xlabel('Epoch')\nplt.legend(loc='upper right')\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Scatter plot of actual vs predicted values\nplt.figure(figsize=(12, 6))\nplt.scatter(y_val, y_pred_weighted, alpha=0.5)\nplt.xlabel('Actual Values')\nplt.ylabel('Predicted Values')\nplt.title('Actual vs Predicted Values')\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Process the test data in chunks and generate predictions\ntest_predictions_list = []\nfor chunk in pl.read_csv(test_file_path, batch_size=10000):\n    X_test_chunk, _ = preprocess_batch(chunk, scaler=scaler, fit_scaler=False)\n    test_predictions_chunk = model.predict(X_test_chunk)\n    test_predictions_list.append(test_predictions_chunk)\n\ntest_predictions = np.concatenate(test_predictions_list, axis=0)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Apply weights to the predictions\nweights_test = sample_submission_df.drop(columns=['sample_id']).to_pandas().values\ntest_predictions_weighted = test_predictions * weights_test","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a DataFrame for the submission\nsubmission_df = pl.DataFrame({\n    'sample_id': pl.read_csv(test_file_path)['sample_id'],\n    'prediction': test_predictions_weighted.flatten()\n})","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save the submission DataFrame to a CSV file\nsubmission_df.write_csv('submission.csv')\n\nprint(\"Submission file created successfully!\")","metadata":{},"execution_count":null,"outputs":[]}]}