{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":56537,"databundleVersionId":8015876,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import polars as pl\nimport numpy as np\nimport pandas as pd\nimport lightgbm as lgb\nfrom sklearn.model_selection import train_test_split, StratifiedGroupKFold\nfrom sklearn.metrics import roc_auc_score \nfrom pathlib import Path\nfrom glob import glob\nimport gc\nimport pandas as pd\nfrom sklearn.preprocessing import LabelEncoder\nimport polars as pl\nimport polars.selectors as cs\nfrom tqdm import tqdm\n\ndataPath = \"/kaggle/input/leap-atmospheric-physics-ai-climsim/\"\nROOT = Path(\"/kaggle/input/leap-atmospheric-physics-ai-climsim\")","metadata":{"execution":{"iopub.status.busy":"2024-04-24T04:21:41.115908Z","iopub.status.idle":"2024-04-24T04:21:41.116489Z","shell.execute_reply.started":"2024-04-24T04:21:41.116227Z","shell.execute_reply":"2024-04-24T04:21:41.116253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pl.scan_csv(\n    ROOT / \"train.csv\"\n).head(100000).drop(\"sample_id\")\ndf_train","metadata":{"execution":{"iopub.status.busy":"2024-04-24T05:23:06.7701Z","iopub.execute_input":"2024-04-24T05:23:06.770738Z","iopub.status.idle":"2024-04-24T05:23:06.841376Z","shell.execute_reply.started":"2024-04-24T05:23:06.770688Z","shell.execute_reply":"2024-04-24T05:23:06.840497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(df_train.head().collect())","metadata":{"execution":{"iopub.status.busy":"2024-04-24T05:23:09.721579Z","iopub.execute_input":"2024-04-24T05:23:09.72289Z","iopub.status.idle":"2024-04-24T05:23:09.828501Z","shell.execute_reply.started":"2024-04-24T05:23:09.72284Z","shell.execute_reply":"2024-04-24T05:23:09.82683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = pl.scan_csv(\n    ROOT / \"train.csv\"\n).head(100000).drop(\"sample_id\")\ndf_test","metadata":{"execution":{"iopub.status.busy":"2024-04-24T05:23:28.733279Z","iopub.execute_input":"2024-04-24T05:23:28.733808Z","iopub.status.idle":"2024-04-24T05:23:28.78538Z","shell.execute_reply.started":"2024-04-24T05:23:28.733767Z","shell.execute_reply":"2024-04-24T05:23:28.784118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(df_train.head().collect())","metadata":{"execution":{"iopub.status.busy":"2024-04-24T05:23:31.630532Z","iopub.execute_input":"2024-04-24T05:23:31.630978Z","iopub.status.idle":"2024-04-24T05:23:31.698378Z","shell.execute_reply.started":"2024-04-24T05:23:31.630945Z","shell.execute_reply":"2024-04-24T05:23:31.697129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission = pd.read_csv(ROOT / \"sample_submission.csv\", nrows=100000)\ndf_submission","metadata":{"execution":{"iopub.status.busy":"2024-04-24T05:11:35.789575Z","iopub.execute_input":"2024-04-24T05:11:35.790167Z","iopub.status.idle":"2024-04-24T05:11:43.797738Z","shell.execute_reply.started":"2024-04-24T05:11:35.790126Z","shell.execute_reply":"2024-04-24T05:11:43.796393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TARGET_SELECTORS = [\n    cs.starts_with('ptend_'),\n    cs.starts_with('cam_out_'),\n]","metadata":{"execution":{"iopub.status.busy":"2024-04-24T05:23:44.480449Z","iopub.execute_input":"2024-04-24T05:23:44.480911Z","iopub.status.idle":"2024-04-24T05:23:44.487276Z","shell.execute_reply.started":"2024-04-24T05:23:44.48088Z","shell.execute_reply":"2024-04-24T05:23:44.485876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = df_train.drop(*TARGET_SELECTORS).collect().to_pandas()\ny_train = df_train.select(*TARGET_SELECTORS).collect().to_pandas()","metadata":{"execution":{"iopub.status.busy":"2024-04-24T05:23:47.826167Z","iopub.execute_input":"2024-04-24T05:23:47.826665Z","iopub.status.idle":"2024-04-24T05:23:53.373808Z","shell.execute_reply.started":"2024-04-24T05:23:47.826628Z","shell.execute_reply":"2024-04-24T05:23:53.372489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train","metadata":{"execution":{"iopub.status.busy":"2024-04-24T05:24:08.735979Z","iopub.execute_input":"2024-04-24T05:24:08.736525Z","iopub.status.idle":"2024-04-24T05:24:08.786381Z","shell.execute_reply.started":"2024-04-24T05:24:08.736486Z","shell.execute_reply":"2024-04-24T05:24:08.78495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = df_test.drop(*TARGET_SELECTORS).collect().to_pandas()\ny_test = df_test.select(*TARGET_SELECTORS).collect().to_pandas()","metadata":{"execution":{"iopub.status.busy":"2024-04-24T05:24:29.350572Z","iopub.execute_input":"2024-04-24T05:24:29.351494Z","iopub.status.idle":"2024-04-24T05:24:35.475638Z","shell.execute_reply.started":"2024-04-24T05:24:29.351447Z","shell.execute_reply":"2024-04-24T05:24:35.474636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_x, valid_x, train_y, valid_y = train_test_split(X_train, y_train, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T05:24:38.063172Z","iopub.execute_input":"2024-04-24T05:24:38.063815Z","iopub.status.idle":"2024-04-24T05:24:38.930812Z","shell.execute_reply.started":"2024-04-24T05:24:38.063763Z","shell.execute_reply":"2024-04-24T05:24:38.929497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_x = train_x.apply(pd.to_numeric, errors='coerce')\ntrain_x.fillna(0.0, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T05:26:00.201593Z","iopub.execute_input":"2024-04-24T05:26:00.202323Z","iopub.status.idle":"2024-04-24T05:26:00.934365Z","shell.execute_reply.started":"2024-04-24T05:26:00.202269Z","shell.execute_reply":"2024-04-24T05:26:00.933042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_x = X_test.apply(pd.to_numeric, errors='coerce')\ntest_x.fillna(0.0, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T05:26:03.004103Z","iopub.execute_input":"2024-04-24T05:26:03.004682Z","iopub.status.idle":"2024-04-24T05:26:03.512242Z","shell.execute_reply.started":"2024-04-24T05:26:03.00464Z","shell.execute_reply":"2024-04-24T05:26:03.511057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.datasets import make_regression\nfrom sklearn.multioutput import MultiOutputRegressor\nfrom sklearn.linear_model import Ridge\nfrom sklearn.metrics import mean_squared_error\n\n# Initialize the base regressor (e.g., Ridge regression)\nbase_regressor = Ridge()\n\n# Initialize the multi-output regressor with the base regressor\nmulti_output_regressor = MultiOutputRegressor(base_regressor)\n\n# Fit the multi-output regressor to the training data with tqdm progress bar\nwith tqdm(total=len(train_x)) as pbar:\n    multi_output_regressor.fit(train_x, train_y)\n    pbar.update(len(train_x))","metadata":{"execution":{"iopub.status.busy":"2024-04-24T04:21:41.140678Z","iopub.status.idle":"2024-04-24T04:21:41.141056Z","shell.execute_reply.started":"2024-04-24T04:21:41.14088Z","shell.execute_reply":"2024-04-24T04:21:41.140895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_valid = multi_output_regressor.predict(valid_x)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T04:21:41.143704Z","iopub.status.idle":"2024-04-24T04:21:41.144097Z","shell.execute_reply.started":"2024-04-24T04:21:41.143916Z","shell.execute_reply":"2024-04-24T04:21:41.143931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mse = mean_squared_error(valid_y, y_valid)\nmse","metadata":{"execution":{"iopub.status.busy":"2024-04-24T04:22:13.746657Z","iopub.execute_input":"2024-04-24T04:22:13.747176Z","iopub.status.idle":"2024-04-24T04:22:13.812028Z","shell.execute_reply.started":"2024-04-24T04:22:13.747135Z","shell.execute_reply":"2024-04-24T04:22:13.810763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"titles = valid_y.columns.tolist()","metadata":{"execution":{"iopub.status.busy":"2024-04-24T05:22:48.21788Z","iopub.execute_input":"2024-04-24T05:22:48.218379Z","iopub.status.idle":"2024-04-24T05:22:48.22657Z","shell.execute_reply.started":"2024-04-24T05:22:48.218345Z","shell.execute_reply":"2024-04-24T05:22:48.224733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\ncolor_list = ['#cd5c5c','#dc143c','#ff1493','#ff8c00','#ffff00','#228b22','#4682b4','#ff00ff','#a52a2a']\nfig, axes = plt.subplots(2, 4, figsize=(16, 8))\nplt.subplots_adjust(hspace=0.5) \nfor i in range(2):\n    for j in range(4):\n        idx = i * 60 + j * 100\n        ax = axes[i, j]\n        ax.scatter(np.arange(len(y_valid[0])), valid_y.values[idx], color='blue', label='True Values')\n        ax.scatter(np.arange(len(y_valid[0])), y_valid[idx], color='red', label='Predicted Values', alpha=0.5)\n        ax.set_xlabel('Index')\n        ax.set_ylabel(titles[idx])\n        ax.set_title(f'{titles[idx]}')\n        ax.legend()\n        ax.grid(True)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T05:26:25.003218Z","iopub.execute_input":"2024-04-24T05:26:25.003703Z","iopub.status.idle":"2024-04-24T05:26:27.895493Z","shell.execute_reply.started":"2024-04-24T05:26:25.003667Z","shell.execute_reply":"2024-04-24T05:26:27.893848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = multi_output_regressor.predict(test_x)\nprint(y_pred.shape)\nprint(y_pred)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T05:01:02.65265Z","iopub.execute_input":"2024-04-24T05:01:02.653145Z","iopub.status.idle":"2024-04-24T05:01:40.292681Z","shell.execute_reply.started":"2024-04-24T05:01:02.653107Z","shell.execute_reply":"2024-04-24T05:01:40.291433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"part1 = df_submission.iloc[:,:1]\npart2 = df_submission.iloc[:,1:]\npart2[:] = y_pred\ndf_subm = pd.concat([part1, part2], axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T05:17:21.484365Z","iopub.execute_input":"2024-04-24T05:17:21.484971Z","iopub.status.idle":"2024-04-24T05:17:22.146061Z","shell.execute_reply.started":"2024-04-24T05:17:21.484927Z","shell.execute_reply":"2024-04-24T05:17:22.144934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_subm","metadata":{"execution":{"iopub.status.busy":"2024-04-24T05:17:25.619822Z","iopub.execute_input":"2024-04-24T05:17:25.620303Z","iopub.status.idle":"2024-04-24T05:17:25.684294Z","shell.execute_reply.started":"2024-04-24T05:17:25.620268Z","shell.execute_reply":"2024-04-24T05:17:25.682904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_subm.to_csv(\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-04-24T05:17:39.771553Z","iopub.execute_input":"2024-04-24T05:17:39.772004Z","iopub.status.idle":"2024-04-24T05:18:45.951024Z","shell.execute_reply.started":"2024-04-24T05:17:39.771971Z","shell.execute_reply":"2024-04-24T05:18:45.949497Z"},"trusted":true},"execution_count":null,"outputs":[]}]}