{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":56537,"databundleVersionId":8015876,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# packages\n\n# standard\nimport numpy as np\nimport pandas as pd\nimport time\n\n# plots\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# faster alternative to pandas\nimport polars as pl","metadata":{"execution":{"iopub.status.busy":"2024-04-28T02:54:26.803503Z","iopub.execute_input":"2024-04-28T02:54:26.803886Z","iopub.status.idle":"2024-04-28T02:54:26.809512Z","shell.execute_reply.started":"2024-04-28T02:54:26.803859Z","shell.execute_reply":"2024-04-28T02:54:26.808269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# configs\npd.set_option('display.max_columns', None) # we want to display all columns in this notebook\n\n# aesthetics\ndefault_color_1 = 'darkblue'\ndefault_color_2 = 'darkgreen'\ndefault_color_3 = 'darkred'\n\n# random seed\nmy_random_seed = 111","metadata":{"execution":{"iopub.status.busy":"2024-04-28T02:54:26.816844Z","iopub.execute_input":"2024-04-28T02:54:26.817177Z","iopub.status.idle":"2024-04-28T02:54:26.823986Z","shell.execute_reply.started":"2024-04-28T02:54:26.817151Z","shell.execute_reply":"2024-04-28T02:54:26.823238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport polars as pl\nfrom sklearn.decomposition import PCA\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.preprocessing import StandardScaler\nimport warnings\nimport os\n\nwarnings.filterwarnings('ignore', category=FutureWarning)\n\n# Paths\ntrain_path = \"/kaggle/input/leap-atmospheric-physics-ai-climsim/train.csv\"\ntest_path = \"/kaggle/input/leap-atmospheric-physics-ai-climsim/test.csv\"\nsubmission_path = \"/kaggle/input/leap-atmospheric-physics-ai-climsim/sample_submission.csv\"\n\n# Check if the dataset files exist\nif not os.path.exists(train_path):\n    raise FileNotFoundError(f\"The specified training file does not exist: {train_path}\")\nif not os.path.exists(test_path):\n    raise FileNotFoundError(f\"The specified test file does not exist: {test_path}\")\nif not os.path.exists(submission_path):\n    raise FileNotFoundError(f\"The specified submission file does not exist: {submission_path}\")\n\n# Data loading using Polars\ndf_train = pl.read_csv(train_path).to_pandas()\nx_train = df_train.iloc[:, 1:557].values\ny_train = df_train.iloc[:, 557:].values\n\ndf_test = pl.read_csv(test_path).to_pandas()\nx_test = df_test.iloc[:, 1:557].values\n\n# Data normalization\nscaler = StandardScaler()\nx_train_scaled = scaler.fit_transform(x_train)\nx_test_scaled = scaler.transform(x_test)\n\n# PCA transformation\npca = PCA(n_components=50)\nx_train_pca = pca.fit_transform(x_train_scaled)\nx_test_pca = pca.transform(x_test_scaled)\n\n# Random Forest model\nrf = RandomForestRegressor(n_estimators=100, random_state=42)\nrf.fit(x_train_pca, y_train)\n\n# Prediction\ny_pred = rf.predict(x_test_pca)\n\n# Prepare submission using Polars\nss = pl.read_csv(submission_path).to_pandas()\nss.iloc[:, 1:] = y_pred\nss.to_csv(\"submission.csv\", index=False)\n\nprint(\"Submission file has been created.\")\n\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check column counts and names\nprint(df_train.columns)\nprint(df_test.columns)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-28T02:59:16.585376Z","iopub.status.idle":"2024-04-28T02:59:16.58573Z","shell.execute_reply.started":"2024-04-28T02:59:16.585555Z","shell.execute_reply":"2024-04-28T02:59:16.585569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import polars as pl\nimport tensorflow as tf\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nimport matplotlib.pyplot as plt\n\n# Paths\ntrain_path = \"/kaggle/input/leap-atmospheric-physics-ai-climsim/train.csv\"\ntest_path = \"/kaggle/input/leap-atmospheric-physics-ai-climsim/test.csv\"\nsubmission_path = \"/kaggle/input/leap-atmospheric-physics-ai-climsim/sample_submission.csv\"\n\n# Data Loading using Polars\ndf_train = pl.read_csv(train_path).to_pandas()\nX = df_train.iloc[:, 1:557].values\ny = df_train.iloc[:, 557:].values\n\n# Data normalization\nscaler_X = StandardScaler()\nX_scaled = scaler_X.fit_transform(X)\nscaler_y = StandardScaler()\ny_scaled = scaler_y.fit_transform(y)\n\n# Data splitting\nX_train, X_val, y_train, y_val = train_test_split(X_scaled, y_scaled, test_size=0.2, random_state=42)\n\n# Model definition\nmodel = tf.keras.Sequential([\n    tf.keras.layers.Dense(512, activation='relu', input_shape=(X_train.shape[1],)),\n    tf.keras.layers.Dropout(0.3),\n    tf.keras.layers.Dense(256, activation='relu'),\n    tf.keras.layers.Dropout(0.3),\n    tf.keras.layers.Dense(128, activation='relu'),\n    tf.keras.layers.Dense(1, activation='linear')\n])\n\n# Model compilation\nmodel.compile(optimizer='adam', loss='mse')\n\n# Model training\nhistory = model.fit(X_train, y_train, validation_data=(X_val, y_val), epochs=100, batch_size=32, verbose=1)\n\n# Evaluation\nval_loss = model.evaluate(X_val, y_val)\n\n# Plot training and validation loss\nplt.plot(history.history['loss'], label='Training loss')\nplt.plot(history.history['val_loss'], label='Validation loss')\nplt.title('Training and Validation Loss')\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.legend()\nplt.show()\n\n# Prediction\ny_pred = model.predict(X_val)  # Adjust this line if you need predictions on the test set\n\n# Prepare Submission\nss = pl.read_csv(submission_path).to_pandas()\nss.iloc[:, 1:] = y_pred\nss.to_csv(\"submission.csv\", index=False)\n\nprint(\"Submission file has been created.\")\n","metadata":{"execution":{"iopub.status.busy":"2024-04-28T03:03:43.608692Z","iopub.execute_input":"2024-04-28T03:03:43.609106Z"},"trusted":true},"execution_count":null,"outputs":[]}]}