{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":56537,"databundleVersionId":8015876,"sourceType":"competition"}],"dockerImageVersionId":30699,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import torch \nimport torch.nn as nn \nimport warnings\n\n# Suppress FutureWarning messages\nwarnings.simplefilter(action='ignore', category=FutureWarning)\n\nimport os\nimport polars as pl\nimport pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2024-05-04T13:18:11.097979Z","iopub.execute_input":"2024-05-04T13:18:11.098294Z","iopub.status.idle":"2024-05-04T13:18:16.207811Z","shell.execute_reply.started":"2024-05-04T13:18:11.098267Z","shell.execute_reply":"2024-05-04T13:18:16.206685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain_df = pl.read_csv_batched('/kaggle/input/leap-atmospheric-physics-ai-climsim/train.csv', batch_size=1000000)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T13:18:16.209698Z","iopub.execute_input":"2024-05-04T13:18:16.21015Z","iopub.status.idle":"2024-05-04T13:18:16.28233Z","shell.execute_reply.started":"2024-05-04T13:18:16.210124Z","shell.execute_reply":"2024-05-04T13:18:16.281431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loading a single Batch of training samples into data variable\ndata = train_df.next_batches(1)[0]    ","metadata":{"execution":{"iopub.status.busy":"2024-05-04T13:18:41.600301Z","iopub.execute_input":"2024-05-04T13:18:41.600936Z","iopub.status.idle":"2024-05-04T13:23:21.595079Z","shell.execute_reply.started":"2024-05-04T13:18:41.600903Z","shell.execute_reply":"2024-05-04T13:23:21.594063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Variable Names for Feature and Prediction Columns\nFEAT_COLS = data.columns[1:557]\nTARGET_COLS = data.columns[557:]","metadata":{"execution":{"iopub.status.busy":"2024-05-04T13:23:21.597046Z","iopub.execute_input":"2024-05-04T13:23:21.597669Z","iopub.status.idle":"2024-05-04T13:23:21.605203Z","shell.execute_reply.started":"2024-05-04T13:23:21.597635Z","shell.execute_reply":"2024-05-04T13:23:21.604363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = data.select(FEAT_COLS)\ny = data.select(TARGET_COLS)\n\n# Converting Data from F64 to F32 Format\nfor col in FEAT_COLS:\n    X = X.with_columns(pl.col(col).cast(pl.Float32))\nfor col in TARGET_COLS:\n    y = y.with_columns(pl.col(col).cast(pl.Float32))\n\n# Figuring out the fixed columns with 0 std in our dataset\nfixed_targets = [TARGET_COLS[int(i)] for i in list(np.where(np.std(y.to_numpy(), axis=0) == 0.0)[0])]\ny_fixed = y.select(fixed_targets)\ny = y.drop(fixed_targets)\n\n# To be used during Submission\ny_columns = y.columns\ny_fixed_columns = y_fixed.columns\n\ny_fixed = np.mean(y_fixed.to_numpy(), axis=0)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T13:23:21.606247Z","iopub.execute_input":"2024-05-04T13:23:21.607101Z","iopub.status.idle":"2024-05-04T13:23:33.386579Z","shell.execute_reply.started":"2024-05-04T13:23:21.607069Z","shell.execute_reply":"2024-05-04T13:23:33.385661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(y_fixed.shape)\nprint(y.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T13:23:33.388649Z","iopub.execute_input":"2024-05-04T13:23:33.388945Z","iopub.status.idle":"2024-05-04T13:23:33.394119Z","shell.execute_reply.started":"2024-05-04T13:23:33.38892Z","shell.execute_reply":"2024-05-04T13:23:33.393189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Converting Data into numpy format and splitting into train and test data\nX, y = X.to_numpy(), y.to_numpy()\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.01, shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T13:23:33.39517Z","iopub.execute_input":"2024-05-04T13:23:33.395464Z","iopub.status.idle":"2024-05-04T13:23:57.833038Z","shell.execute_reply.started":"2024-05-04T13:23:33.395426Z","shell.execute_reply":"2024-05-04T13:23:57.832227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(X_train.shape)\nprint(y_train.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T13:23:57.834155Z","iopub.execute_input":"2024-05-04T13:23:57.834537Z","iopub.status.idle":"2024-05-04T13:23:57.839556Z","shell.execute_reply.started":"2024-05-04T13:23:57.834506Z","shell.execute_reply":"2024-05-04T13:23:57.838783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(X_test.shape)\nprint(y_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T13:23:57.840666Z","iopub.execute_input":"2024-05-04T13:23:57.840916Z","iopub.status.idle":"2024-05-04T13:23:57.852297Z","shell.execute_reply.started":"2024-05-04T13:23:57.840895Z","shell.execute_reply":"2024-05-04T13:23:57.851398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pre-Processing - Deleting Feature Columns with 0 or almost 0 standard deviation \ndeleted_columns = np.where(np.std(X_train, axis=0) < 1e-6)[0]\nX_train_cleaned = np.delete(X_train, deleted_columns, 1)\nprint(X_train_cleaned.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T13:23:57.853292Z","iopub.execute_input":"2024-05-04T13:23:57.853568Z","iopub.status.idle":"2024-05-04T13:24:05.443979Z","shell.execute_reply.started":"2024-05-04T13:23:57.853547Z","shell.execute_reply":"2024-05-04T13:24:05.443067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pre-Processing - Standard Scaling our Cleaned Features\nfrom sklearn.preprocessing import StandardScaler\nscaler = StandardScaler()\nscaler.fit(X_train_cleaned)\nX_train_preprocessed = scaler.transform(X_train_cleaned)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T13:24:05.445132Z","iopub.execute_input":"2024-05-04T13:24:05.445448Z","iopub.status.idle":"2024-05-04T13:24:14.297968Z","shell.execute_reply.started":"2024-05-04T13:24:05.445422Z","shell.execute_reply":"2024-05-04T13:24:14.296853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Defining our XGBRegressor model to be computed using GPU\nfrom catboost import CatBoostRegressor\n# from sklearn.metrics import mean_squared_error\n# from sklearn.model_selection import GridSearchCV\n# from sklearn.multioutput import MultiOutputRegressor\n\nmulti_model = CatBoostRegressor(task_type=\"GPU\", devices='0', objective='MultiRMSE')\n# multi_model = MultiOutputRegressor(model)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T13:24:14.30073Z","iopub.execute_input":"2024-05-04T13:24:14.301Z","iopub.status.idle":"2024-05-04T13:24:14.866862Z","shell.execute_reply.started":"2024-05-04T13:24:14.300977Z","shell.execute_reply":"2024-05-04T13:24:14.865836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nmulti_model.fit(X_train_preprocessed, y_train)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T13:24:14.870072Z","iopub.execute_input":"2024-05-04T13:24:14.870391Z","iopub.status.idle":"2024-05-04T14:11:41.949696Z","shell.execute_reply.started":"2024-05-04T13:24:14.870354Z","shell.execute_reply":"2024-05-04T14:11:41.948614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del X\ndel y\ndel X_train\ndel y_train","metadata":{"execution":{"iopub.status.busy":"2024-05-04T14:11:46.340858Z","iopub.execute_input":"2024-05-04T14:11:46.341221Z","iopub.status.idle":"2024-05-04T14:11:46.575054Z","shell.execute_reply.started":"2024-05-04T14:11:46.341192Z","shell.execute_reply":"2024-05-04T14:11:46.573926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pre-Processing our X_test data and calculating predictions using trained multi_model\nX_test_cleaned = np.delete(X_test, deleted_columns, 1)\nX_test_preprocessed = scaler.transform(X_test_cleaned)\npredictions = multi_model.predict(X_test_preprocessed)\npredictions","metadata":{"execution":{"iopub.status.busy":"2024-05-04T14:11:50.16519Z","iopub.execute_input":"2024-05-04T14:11:50.165569Z","iopub.status.idle":"2024-05-04T14:11:50.822874Z","shell.execute_reply.started":"2024-05-04T14:11:50.165539Z","shell.execute_reply":"2024-05-04T14:11:50.821947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Getting our score for Test Dataset\nmulti_model.score(X_test_preprocessed,y_test)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T14:12:10.285782Z","iopub.execute_input":"2024-05-04T14:12:10.286099Z","iopub.status.idle":"2024-05-04T14:12:10.917974Z","shell.execute_reply.started":"2024-05-04T14:12:10.286076Z","shell.execute_reply":"2024-05-04T14:12:10.917022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pickle\n\nwith open(\"catboost_1M.pkl\", \"wb\") as f:\n    pickle.dump(multi_model, f)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T14:12:17.163931Z","iopub.execute_input":"2024-05-04T14:12:17.164736Z","iopub.status.idle":"2024-05-04T14:12:17.741766Z","shell.execute_reply.started":"2024-05-04T14:12:17.164702Z","shell.execute_reply":"2024-05-04T14:12:17.7409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open(\"1M_deleted_columns.pkl\", \"wb\") as f:\n    pickle.dump(deleted_columns, f)\n\nwith open(\"1M_y_fixed_columns.pkl\", \"wb\") as f:\n    pickle.dump(y_fixed_columns, f)\n\nwith open(\"1M_y_columns.pkl\", \"wb\") as f:\n    pickle.dump(y_columns, f)\n    \nwith open(\"1M_y_fixed_means.pkl\", \"wb\") as f:\n    pickle.dump(y_fixed, f)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T14:12:45.077305Z","iopub.execute_input":"2024-05-04T14:12:45.077689Z","iopub.status.idle":"2024-05-04T14:12:45.086537Z","shell.execute_reply.started":"2024-05-04T14:12:45.07766Z","shell.execute_reply":"2024-05-04T14:12:45.085513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load Test Sample Features\ntest_df = pl.read_csv('/kaggle/input/leap-atmospheric-physics-ai-climsim/test.csv')","metadata":{"execution":{"iopub.status.busy":"2024-05-04T14:13:47.253761Z","iopub.execute_input":"2024-05-04T14:13:47.25413Z","iopub.status.idle":"2024-05-04T14:14:15.691479Z","shell.execute_reply.started":"2024-05-04T14:13:47.254102Z","shell.execute_reply":"2024-05-04T14:14:15.69066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in FEAT_COLS:\n    test_dataframe = test_df.select(FEAT_COLS).with_columns(pl.col(col).cast(pl.Float32))\n\ntest_dataframe = test_dataframe.to_numpy()\n\nprint(test_dataframe.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T14:14:25.881324Z","iopub.execute_input":"2024-05-04T14:14:25.881952Z","iopub.status.idle":"2024-05-04T14:14:30.925711Z","shell.execute_reply.started":"2024-05-04T14:14:25.881922Z","shell.execute_reply":"2024-05-04T14:14:30.92481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del test_df","metadata":{"execution":{"iopub.status.busy":"2024-05-04T14:14:40.800144Z","iopub.execute_input":"2024-05-04T14:14:40.80052Z","iopub.status.idle":"2024-05-04T14:14:41.017445Z","shell.execute_reply.started":"2024-05-04T14:14:40.800493Z","shell.execute_reply":"2024-05-04T14:14:41.016344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pre-Processing our Test Samples\ntest_dataframe_cleaned = np.delete(test_dataframe, deleted_columns, 1)\ntest_dataframe_preprocessed = scaler.transform(test_dataframe_cleaned)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T14:14:48.02058Z","iopub.execute_input":"2024-05-04T14:14:48.021167Z","iopub.status.idle":"2024-05-04T14:14:49.83732Z","shell.execute_reply.started":"2024-05-04T14:14:48.021123Z","shell.execute_reply":"2024-05-04T14:14:49.836498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Getting the output from our trained model for the test samples\npredictions = multi_model.predict(test_dataframe_preprocessed)\npredictions","metadata":{"execution":{"iopub.status.busy":"2024-05-04T14:14:52.24875Z","iopub.execute_input":"2024-05-04T14:14:52.249462Z","iopub.status.idle":"2024-05-04T14:15:35.751996Z","shell.execute_reply.started":"2024-05-04T14:14:52.249429Z","shell.execute_reply":"2024-05-04T14:15:35.751046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del test_dataframe\ndel test_dataframe_cleaned\ndel test_dataframe_preprocessed","metadata":{"execution":{"iopub.status.busy":"2024-05-04T14:16:10.767776Z","iopub.execute_input":"2024-05-04T14:16:10.768128Z","iopub.status.idle":"2024-05-04T14:16:10.971225Z","shell.execute_reply.started":"2024-05-04T14:16:10.768101Z","shell.execute_reply":"2024-05-04T14:16:10.970083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loading submission File\nimport pandas as pd\nsub = pd.read_csv(\"/kaggle/input/leap-atmospheric-physics-ai-climsim/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-05-04T14:16:13.192575Z","iopub.execute_input":"2024-05-04T14:16:13.193242Z","iopub.status.idle":"2024-05-04T14:17:42.104077Z","shell.execute_reply.started":"2024-05-04T14:16:13.1932Z","shell.execute_reply":"2024-05-04T14:17:42.103097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-04T14:18:56.275482Z","iopub.execute_input":"2024-05-04T14:18:56.276502Z","iopub.status.idle":"2024-05-04T14:18:56.319054Z","shell.execute_reply.started":"2024-05-04T14:18:56.276466Z","shell.execute_reply":"2024-05-04T14:18:56.317807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Using the submission sample target values as weights \nsub.loc[:,y_columns] *= predictions","metadata":{"execution":{"iopub.status.busy":"2024-05-04T14:18:59.800576Z","iopub.execute_input":"2024-05-04T14:18:59.800933Z","iopub.status.idle":"2024-05-04T14:19:01.299093Z","shell.execute_reply.started":"2024-05-04T14:18:59.800905Z","shell.execute_reply":"2024-05-04T14:19:01.298313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.loc[:, y_fixed_columns] *= y_fixed","metadata":{"execution":{"iopub.status.busy":"2024-05-04T14:19:12.84085Z","iopub.execute_input":"2024-05-04T14:19:12.84157Z","iopub.status.idle":"2024-05-04T14:19:13.258353Z","shell.execute_reply.started":"2024-05-04T14:19:12.841537Z","shell.execute_reply":"2024-05-04T14:19:13.257556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-04T14:19:14.864125Z","iopub.execute_input":"2024-05-04T14:19:14.864802Z","iopub.status.idle":"2024-05-04T14:19:14.888624Z","shell.execute_reply.started":"2024-05-04T14:19:14.864772Z","shell.execute_reply":"2024-05-04T14:19:14.887603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(sub.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T14:19:17.056471Z","iopub.execute_input":"2024-05-04T14:19:17.057473Z","iopub.status.idle":"2024-05-04T14:19:17.062193Z","shell.execute_reply.started":"2024-05-04T14:19:17.057429Z","shell.execute_reply":"2024-05-04T14:19:17.061154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Outputting our submission file\ntest_polars = pl.from_pandas(sub[[\"sample_id\"]+TARGET_COLS])\ntest_polars.write_csv(\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-05-04T14:19:20.440345Z","iopub.execute_input":"2024-05-04T14:19:20.441188Z","iopub.status.idle":"2024-05-04T14:19:38.246044Z","shell.execute_reply.started":"2024-05-04T14:19:20.441156Z","shell.execute_reply":"2024-05-04T14:19:38.244948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}