{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":56537,"databundleVersionId":8015876,"sourceType":"competition"},{"sourceId":8210320,"sourceType":"datasetVersion","datasetId":4865495}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"do_cudf = False","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-11T09:40:52.892714Z","iopub.execute_input":"2024-05-11T09:40:52.893641Z","iopub.status.idle":"2024-05-11T09:40:52.904536Z","shell.execute_reply.started":"2024-05-11T09:40:52.893576Z","shell.execute_reply":"2024-05-11T09:40:52.903531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if do_cudf:\n    !pip install \\\n        --extra-index-url=https://pypi.nvidia.com \\\n        cudf-cu12==24.4.* dask-cudf-cu12==24.4.* cuml-cu12==24.4.* \\\n        cugraph-cu12==24.4.* cuspatial-cu12==24.4.* cuproj-cu12==24.4.* \\\n        cuxfilter-cu12==24.4.* cucim-cu12==24.4.* pylibraft-cu12==24.4.* \\\n        raft-dask-cu12==24.4.* cuvs-cu12==24.4.*","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-11T09:40:52.906294Z","iopub.execute_input":"2024-05-11T09:40:52.906652Z","iopub.status.idle":"2024-05-11T09:40:52.913887Z","shell.execute_reply.started":"2024-05-11T09:40:52.90662Z","shell.execute_reply":"2024-05-11T09:40:52.913122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if do_cudf:\n    %load_ext cudf.pandas\n    \nfrom fastai.imports import *\nfrom pathlib import Path\n#from sklearn.metrics import mean_squared_error\nimport lightgbm as lgb\nimport optuna \n\nimport gc\nimport cudf\nimport pickle\nimport numpy as np\nfrom glob import glob\nfrom tqdm.auto import tqdm\nimport polars as pl\nfrom sklearn.metrics import r2_score\nimport warnings\n\nwarnings.simplefilter(action='ignore', category=FutureWarning)\npd.options.display.max_columns = 100","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-11T09:40:52.917956Z","iopub.execute_input":"2024-05-11T09:40:52.918193Z","iopub.status.idle":"2024-05-11T09:40:59.425878Z","shell.execute_reply.started":"2024-05-11T09:40:52.918172Z","shell.execute_reply":"2024-05-11T09:40:59.424899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subsample_size = 1_000_000","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-11T09:40:59.427675Z","iopub.execute_input":"2024-05-11T09:40:59.428677Z","iopub.status.idle":"2024-05-11T09:40:59.433148Z","shell.execute_reply.started":"2024-05-11T09:40:59.428641Z","shell.execute_reply":"2024-05-11T09:40:59.432194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nn_rows = 1_000_000 \ndata = pl.read_csv(\"/kaggle/input/leap-atmospheric-physics-ai-climsim/train.csv\", n_rows=n_rows*2).to_pandas()\n\ntrain = data[:n_rows]\nvalid = data[n_rows:]\n\ntrain.shape, valid.shape","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:40:59.434334Z","iopub.execute_input":"2024-05-11T09:40:59.434905Z","iopub.status.idle":"2024-05-11T09:41:19.126142Z","shell.execute_reply.started":"2024-05-11T09:40:59.434873Z","shell.execute_reply":"2024-05-11T09:41:19.125304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:41:19.128298Z","iopub.execute_input":"2024-05-11T09:41:19.128586Z","iopub.status.idle":"2024-05-11T09:41:19.217138Z","shell.execute_reply.started":"2024-05-11T09:41:19.128561Z","shell.execute_reply":"2024-05-11T09:41:19.216318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:41:19.218385Z","iopub.execute_input":"2024-05-11T09:41:19.218744Z","iopub.status.idle":"2024-05-11T09:41:19.297805Z","shell.execute_reply.started":"2024-05-11T09:41:19.218713Z","shell.execute_reply":"2024-05-11T09:41:19.296965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_columns = [col for col in train.columns if col.startswith(\"cam_out\") or col.startswith(\"ptend\")]\nlen(target_columns)","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:41:19.29902Z","iopub.execute_input":"2024-05-11T09:41:19.299394Z","iopub.status.idle":"2024-05-11T09:41:19.306577Z","shell.execute_reply.started":"2024-05-11T09:41:19.299351Z","shell.execute_reply":"2024-05-11T09:41:19.305653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_columns = [col for col in train.columns if not col.startswith(\"cam_out\") and not col.startswith(\"ptend\") and col != \"sample_id\"]\nlen(feature_columns)","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:41:19.30765Z","iopub.execute_input":"2024-05-11T09:41:19.307939Z","iopub.status.idle":"2024-05-11T09:41:19.316798Z","shell.execute_reply.started":"2024-05-11T09:41:19.307915Z","shell.execute_reply":"2024-05-11T09:41:19.315911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# duplicates and nan","metadata":{}},{"cell_type":"code","source":"train.duplicated().sum(), valid.duplicated().sum()","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:41:19.317859Z","iopub.execute_input":"2024-05-11T09:41:19.318189Z","iopub.status.idle":"2024-05-11T09:41:28.91152Z","shell.execute_reply.started":"2024-05-11T09:41:19.31816Z","shell.execute_reply":"2024-05-11T09:41:28.910501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.isna().sum().sum(), valid.isna().sum().sum()","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:41:28.912965Z","iopub.execute_input":"2024-05-11T09:41:28.913704Z","iopub.status.idle":"2024-05-11T09:41:29.2116Z","shell.execute_reply.started":"2024-05-11T09:41:28.913667Z","shell.execute_reply":"2024-05-11T09:41:29.210678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# feature engineering","metadata":{}},{"cell_type":"code","source":"# group features:\n# state_t\n# state_q0001\n# state_q0002\n# state_q0003\n# state_u\n# state_v\n# pbuf_ozone\n# pbuf_CH4\n# pbuf_N2O\n\npd.options.mode.chained_assignment = None\n\nstate_t_cols = [c for c in train.columns if c.startswith(\"state_t\")]\nstate_q0001_cols = [c for c in train.columns if c.startswith(\"state_q0001\")]\nstate_q0002_cols = [c for c in train.columns if c.startswith(\"state_q0002\")]\nstate_q0003_cols = [c for c in train.columns if c.startswith(\"state_q0003\")]\nstate_u_cols = [c for c in train.columns if c.startswith(\"state_u\")]\nstate_v_cols = [c for c in train.columns if c.startswith(\"state_v\")]\npbuf_ozone_cols = [c for c in train.columns if c.startswith(\"pbuf_ozone\")]\npbuf_CH4_cols = [c for c in train.columns if c.startswith(\"pbuf_CH4\")]\npbuf_N2O_cols = [c for c in train.columns if c.startswith(\"pbuf_N2O\")]\n\ngroup_cols = dict(state_t_cols=state_t_cols,state_q0001_cols=state_q0001_cols,state_q0002_cols=state_q0002_cols,\n                 state_q0003_cols=state_q0003_cols,state_u_cols=state_u_cols,state_v_cols=state_v_cols,\n                 pbuf_ozone_cols=pbuf_ozone_cols,pbuf_CH4_cols=pbuf_CH4_cols,pbuf_N2O_cols=pbuf_N2O_cols)","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:41:29.216595Z","iopub.execute_input":"2024-05-11T09:41:29.21695Z","iopub.status.idle":"2024-05-11T09:41:29.227562Z","shell.execute_reply.started":"2024-05-11T09:41:29.216926Z","shell.execute_reply":"2024-05-11T09:41:29.226799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for key in group_cols:\n    train.loc[:, f\"{key}_mean\"] = train[group_cols[key]].mean(axis=1)\n    train.loc[:, f\"{key}_std\"] = train[group_cols[key]].std(axis=1)\n    ","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:41:29.228526Z","iopub.execute_input":"2024-05-11T09:41:29.228762Z","iopub.status.idle":"2024-05-11T09:41:30.74427Z","shell.execute_reply.started":"2024-05-11T09:41:29.228742Z","shell.execute_reply":"2024-05-11T09:41:30.74319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def add_group_col_stats(dataframe):\n    for key in group_cols:\n        dataframe.loc[:, f\"{key}_mean\"] = dataframe[group_cols[key]].mean(axis=1)\n        dataframe.loc[:, f\"{key}_std\"] = dataframe[group_cols[key]].std(axis=1)\n        feature_columns.append(f\"{key}_mean\")\n        feature_columns.append(f\"{key}_std\")\n    return dataframe\n\nvalid = add_group_col_stats(valid)","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:41:30.745547Z","iopub.execute_input":"2024-05-11T09:41:30.745895Z","iopub.status.idle":"2024-05-11T09:41:32.239341Z","shell.execute_reply.started":"2024-05-11T09:41:30.745864Z","shell.execute_reply":"2024-05-11T09:41:32.238534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# camin_longwave = cam_in_ALDIF + cam_in_ALDIR\n# camin_shortwave = cam_in_ASDIF + cam_in_ASDIR\n# wind_speed_magnitude = sqrt(state_u^2 + state_v^2)\n\ntrain.loc[:,'camin_longwave'] = train['cam_in_ALDIF'] + train['cam_in_ALDIR']\ntrain.loc[:,'camin_shortwave'] = train['cam_in_ASDIF'] + train['cam_in_ASDIR']\ntrain.loc[:,'wind_speed_magnitude'] = np.sqrt(train['state_u_cols_mean']**2 + train['state_v_cols_mean']**2)","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:41:32.240439Z","iopub.execute_input":"2024-05-11T09:41:32.240734Z","iopub.status.idle":"2024-05-11T09:41:32.252031Z","shell.execute_reply.started":"2024-05-11T09:41:32.240708Z","shell.execute_reply":"2024-05-11T09:41:32.251301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def add_camin(dataframe):\n    dataframe.loc[:,'camin_longwave'] = dataframe['cam_in_ALDIF'] + dataframe['cam_in_ALDIR']\n    dataframe.loc[:,'camin_shortwave'] = dataframe['cam_in_ASDIF'] + dataframe['cam_in_ASDIR']\n    feature_columns.append(f\"camin_longwave\")\n    feature_columns.append(f\"camin_shortwave\")\n    return dataframe\n\ndef add_wind_speed(dataframe):\n    dataframe.loc[:,'wind_speed_magnitude'] = np.sqrt(dataframe['state_u_cols_mean']**2 + dataframe['state_v_cols_mean']**2)\n    feature_columns.append(f\"wind_speed_magnitude\")\n    return dataframe\n\nvalid = add_camin(valid)\nvalid = add_wind_speed(valid)","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:41:32.253367Z","iopub.execute_input":"2024-05-11T09:41:32.253724Z","iopub.status.idle":"2024-05-11T09:41:32.265825Z","shell.execute_reply.started":"2024-05-11T09:41:32.253692Z","shell.execute_reply":"2024-05-11T09:41:32.265068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create zones\n# Assign a categorical label to each grid cell based on the dominant surface type identified:\n# If land fraction > 50%, assign 'Land'. cam_in_ICEFRAC\n# If ocean fraction > 50%, assign 'Ocean'. cam_in_LANDFRAC\n# If sea-ice fraction > 50%, assign 'Sea-Ice'. cam_in_OCNFRAC\n# If none of the above conditions are met, it could indicate a mixed or transitional zone, and you may assign a label such as 'Mixed' or 'Transitional'.\n\ntrain.loc[:,'dominant_surface'] = train[['cam_in_ICEFRAC','cam_in_LANDFRAC','cam_in_OCNFRAC']].idxmax(axis=1).astype(\"category\").cat.codes\n\ndef add_dominant_surface(dataframe):\n    dataframe.loc[:,'dominant_surface'] = dataframe[['cam_in_ICEFRAC','cam_in_LANDFRAC','cam_in_OCNFRAC']].idxmax(axis=1).astype(\"category\").cat.codes\n    \n    feature_columns.append(f\"dominant_surface\")\n    return dataframe\n\nvalid = add_dominant_surface(valid)","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:41:32.266916Z","iopub.execute_input":"2024-05-11T09:41:32.267229Z","iopub.status.idle":"2024-05-11T09:41:32.338595Z","shell.execute_reply.started":"2024-05-11T09:41:32.267205Z","shell.execute_reply":"2024-05-11T09:41:32.33776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.dtypes","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:41:32.339606Z","iopub.execute_input":"2024-05-11T09:41:32.339855Z","iopub.status.idle":"2024-05-11T09:41:32.347032Z","shell.execute_reply.started":"2024-05-11T09:41:32.339827Z","shell.execute_reply":"2024-05-11T09:41:32.346124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Daytime/Nighttime Indicator Creation:\n# Use solar insolation or solar zenith angle to determine whether it's daytime or nighttime at each time step and location.\n# For solar insolation, set a threshold (e.g., if solar insolation > 0, it's daytime; otherwise, it's nighttime).\n# For the solar zenith angle, set a threshold (e.g., if solar zenith angle < 90°, it's daytime; otherwise, it's nighttime).\n# Create a binary indicator (0 for nighttime, 1 for daytime) based on the chosen criterion.\n","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:41:32.34816Z","iopub.execute_input":"2024-05-11T09:41:32.348444Z","iopub.status.idle":"2024-05-11T09:41:32.353288Z","shell.execute_reply.started":"2024-05-11T09:41:32.348421Z","shell.execute_reply":"2024-05-11T09:41:32.352406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:41:32.354511Z","iopub.execute_input":"2024-05-11T09:41:32.355104Z","iopub.status.idle":"2024-05-11T09:41:32.364669Z","shell.execute_reply.started":"2024-05-11T09:41:32.355073Z","shell.execute_reply":"2024-05-11T09:41:32.363816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# find zero variance columns","metadata":{}},{"cell_type":"code","source":"nuniques = pd.DataFrame(train[feature_columns].nunique(axis=0), columns=[\"n_unique\"])\nnuniques.shape","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:41:32.365839Z","iopub.execute_input":"2024-05-11T09:41:32.366139Z","iopub.status.idle":"2024-05-11T09:41:34.550876Z","shell.execute_reply.started":"2024-05-11T09:41:32.366116Z","shell.execute_reply":"2024-05-11T09:41:34.549972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nuniques_le_one = nuniques.query('n_unique == 1')\nzero_var_colums = array(nuniques_le_one.index)\nzero_var_colums.shape","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:41:34.552127Z","iopub.execute_input":"2024-05-11T09:41:34.552807Z","iopub.status.idle":"2024-05-11T09:41:34.561978Z","shell.execute_reply.started":"2024-05-11T09:41:34.552771Z","shell.execute_reply":"2024-05-11T09:41:34.56098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"non_zero_features = array(nuniques[nuniques.gt(1).n_unique.values].index)\nlen(non_zero_features)","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:41:34.563397Z","iopub.execute_input":"2024-05-11T09:41:34.564057Z","iopub.status.idle":"2024-05-11T09:41:34.571664Z","shell.execute_reply.started":"2024-05-11T09:41:34.564026Z","shell.execute_reply":"2024-05-11T09:41:34.570667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# unimportant columns according to lgb\nunimportant_features = array(['state_q0002_15', 'state_q0002_17', 'pbuf_ozone_1', 'pbuf_ozone_2',\n       'pbuf_ozone_3', 'pbuf_ozone_4', 'pbuf_ozone_5', 'pbuf_ozone_6',\n       'pbuf_CH4_2', 'pbuf_CH4_4', 'pbuf_CH4_5', 'pbuf_CH4_6',\n       'pbuf_CH4_7', 'pbuf_CH4_8', 'pbuf_CH4_9', 'pbuf_CH4_10',\n       'pbuf_CH4_12', 'pbuf_CH4_13', 'pbuf_CH4_14', 'pbuf_CH4_15',\n       'pbuf_CH4_16', 'pbuf_CH4_17', 'pbuf_N2O_1', 'pbuf_N2O_2',\n       'pbuf_N2O_3', 'pbuf_N2O_4', 'pbuf_N2O_5', 'pbuf_N2O_6',\n       'pbuf_N2O_7', 'pbuf_N2O_8', 'pbuf_N2O_9', 'pbuf_N2O_10',\n       'pbuf_N2O_11', 'pbuf_N2O_12', 'pbuf_N2O_13', 'pbuf_N2O_14',\n       'pbuf_N2O_15', 'pbuf_N2O_16', 'pbuf_N2O_18', 'pbuf_N2O_19',\n       'pbuf_N2O_20', 'pbuf_N2O_21', 'pbuf_N2O_23', 'pbuf_N2O_24',\n       'pbuf_N2O_26'])\n\nlen(unimportant_features)","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:41:34.572895Z","iopub.execute_input":"2024-05-11T09:41:34.573219Z","iopub.status.idle":"2024-05-11T09:41:34.582819Z","shell.execute_reply.started":"2024-05-11T09:41:34.573194Z","shell.execute_reply":"2024-05-11T09:41:34.581944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# substract unimportant features from array\nnon_zero_features = array([f for f in non_zero_features if f not in unimportant_features])","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:41:34.584025Z","iopub.execute_input":"2024-05-11T09:41:34.584322Z","iopub.status.idle":"2024-05-11T09:41:34.593237Z","shell.execute_reply.started":"2024-05-11T09:41:34.5843Z","shell.execute_reply":"2024-05-11T09:41:34.592406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(non_zero_features)","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:41:34.594448Z","iopub.execute_input":"2024-05-11T09:41:34.594767Z","iopub.status.idle":"2024-05-11T09:41:34.603275Z","shell.execute_reply.started":"2024-05-11T09:41:34.594744Z","shell.execute_reply":"2024-05-11T09:41:34.602293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# data split and subsampling","metadata":{}},{"cell_type":"code","source":"assert n_rows >= subsample_size","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:41:34.604431Z","iopub.execute_input":"2024-05-11T09:41:34.604718Z","iopub.status.idle":"2024-05-11T09:41:34.611138Z","shell.execute_reply.started":"2024-05-11T09:41:34.604696Z","shell.execute_reply":"2024-05-11T09:41:34.610355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_subsampled = train.sample(subsample_size)\n\nX_train = train_subsampled[non_zero_features]\ny_train = train_subsampled[target_columns]\n\nX_valid = valid[non_zero_features]\ny_valid = valid[target_columns]\n\nX_train.shape, y_train.shape, X_valid.shape, y_valid.shape","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:41:34.612195Z","iopub.execute_input":"2024-05-11T09:41:34.612511Z","iopub.status.idle":"2024-05-11T09:41:34.965301Z","shell.execute_reply.started":"2024-05-11T09:41:34.612488Z","shell.execute_reply":"2024-05-11T09:41:34.964445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX_valid_1, X_valid_2, y_valid_1, y_valid_2 = train_test_split(X_valid, y_valid, train_size=0.1)\nX_valid_1.shape, X_valid_2.shape, y_valid_1.shape, y_valid_2.shape","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:41:34.966596Z","iopub.execute_input":"2024-05-11T09:41:34.966912Z","iopub.status.idle":"2024-05-11T09:41:35.517606Z","shell.execute_reply.started":"2024-05-11T09:41:34.966887Z","shell.execute_reply":"2024-05-11T09:41:35.516634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# targets","metadata":{}},{"cell_type":"code","source":"class TargetMeans:\n    def __init__(self, ys: pd.DataFrame):\n        self.y_means = self._get_means(ys)\n        \n    def _get_means(self, ys):\n        return ys.mean(axis=0).to_dict()","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:41:35.525598Z","iopub.execute_input":"2024-05-11T09:41:35.525954Z","iopub.status.idle":"2024-05-11T09:41:35.530894Z","shell.execute_reply.started":"2024-05-11T09:41:35.52593Z","shell.execute_reply":"2024-05-11T09:41:35.529963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_means = TargetMeans(y_train)\nassert len(target_means.y_means) == 368","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:41:35.532044Z","iopub.execute_input":"2024-05-11T09:41:35.532365Z","iopub.status.idle":"2024-05-11T09:41:35.552748Z","shell.execute_reply.started":"2024-05-11T09:41:35.53232Z","shell.execute_reply":"2024-05-11T09:41:35.551878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train.shape","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:41:35.55373Z","iopub.execute_input":"2024-05-11T09:41:35.55401Z","iopub.status.idle":"2024-05-11T09:41:35.559486Z","shell.execute_reply.started":"2024-05-11T09:41:35.553988Z","shell.execute_reply":"2024-05-11T09:41:35.558607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.feature_selection import VarianceThreshold\n\nclass VarThreshold(VarianceThreshold):\n    def __init__(self, **kwargs):\n        super().__init__(**kwargs)\n        \n    def transform(self, X):\n        return X.loc[:,self.get_support()]\n    \ntarget_selector = VarThreshold(threshold=0)\ntarget_selector.fit(y_train)","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:41:35.56062Z","iopub.execute_input":"2024-05-11T09:41:35.560893Z","iopub.status.idle":"2024-05-11T09:41:35.754396Z","shell.execute_reply.started":"2024-05-11T09:41:35.560862Z","shell.execute_reply":"2024-05-11T09:41:35.753476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = target_selector.transform(y_train)\ny_train.shape","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:41:35.755741Z","iopub.execute_input":"2024-05-11T09:41:35.756099Z","iopub.status.idle":"2024-05-11T09:41:35.7703Z","shell.execute_reply.started":"2024-05-11T09:41:35.756067Z","shell.execute_reply":"2024-05-11T09:41:35.769475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# y_valid_1 = target_selector.transform(y_valid_1)\n# y_valid_2 = target_selector.transform(y_valid_2)","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:41:35.771327Z","iopub.execute_input":"2024-05-11T09:41:35.771602Z","iopub.status.idle":"2024-05-11T09:41:35.775558Z","shell.execute_reply.started":"2024-05-11T09:41:35.771579Z","shell.execute_reply":"2024-05-11T09:41:35.774572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_valid_1.shape, y_valid_2.shape","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:41:35.776709Z","iopub.execute_input":"2024-05-11T09:41:35.776997Z","iopub.status.idle":"2024-05-11T09:41:35.787671Z","shell.execute_reply.started":"2024-05-11T09:41:35.776974Z","shell.execute_reply":"2024-05-11T09:41:35.786857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# feature scaling","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\n\nfeature_scaler = StandardScaler().fit(X_train)\n\nX_train.values[:] = feature_scaler.transform(X_train)\nX_valid_1.values[:] = feature_scaler.transform(X_valid_1)\nX_valid_2.values[:] = feature_scaler.transform(X_valid_2)\n\nX_train.shape, X_valid_1.shape, X_valid_2.shape","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:41:35.788588Z","iopub.execute_input":"2024-05-11T09:41:35.788865Z","iopub.status.idle":"2024-05-11T09:41:36.555207Z","shell.execute_reply.started":"2024-05-11T09:41:35.788842Z","shell.execute_reply":"2024-05-11T09:41:36.554107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:41:36.556458Z","iopub.execute_input":"2024-05-11T09:41:36.556775Z","iopub.status.idle":"2024-05-11T09:41:36.664097Z","shell.execute_reply.started":"2024-05-11T09:41:36.55675Z","shell.execute_reply":"2024-05-11T09:41:36.66308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train.shape, y_valid_1.shape, y_valid_2.shape","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:41:36.665488Z","iopub.execute_input":"2024-05-11T09:41:36.665805Z","iopub.status.idle":"2024-05-11T09:41:36.672598Z","shell.execute_reply.started":"2024-05-11T09:41:36.665777Z","shell.execute_reply":"2024-05-11T09:41:36.671394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_scaler = StandardScaler().fit(y_train)\n\ny_train.values[:] = target_scaler.transform(y_train)\n#y_valid_1.values[:] = target_scaler.transform(y_valid_1)\n#y_valid_2.values[:] = target_scaler.transform(y_valid_2)","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:41:36.673831Z","iopub.execute_input":"2024-05-11T09:41:36.674138Z","iopub.status.idle":"2024-05-11T09:41:36.748098Z","shell.execute_reply.started":"2024-05-11T09:41:36.674112Z","shell.execute_reply":"2024-05-11T09:41:36.747273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# feature engineering ideas","metadata":{}},{"cell_type":"code","source":"# group features:\n# state_t\n# state_q0001\n# state_q0002\n# state_q0003\n# state_u\n# state_v\n# pbuf_ozone\n# pbuf_CH4\n# pbuf_N2O\n","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:41:36.749196Z","iopub.execute_input":"2024-05-11T09:41:36.749484Z","iopub.status.idle":"2024-05-11T09:41:36.753742Z","shell.execute_reply.started":"2024-05-11T09:41:36.74946Z","shell.execute_reply":"2024-05-11T09:41:36.752718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# camin_longwave = cam_in_ALDIF + cam_in_ALDIR\n# camin_shortwave = cam_in_ASDIF + cam_in_ASDIR\n# wind_speed_magnitude = sqrt(state_u^2 + state_v^2)\n# \n# create zones\n# Assign a categorical label to each grid cell based on the dominant surface type identified:\n# If land fraction > 50%, assign 'Land'.\n# If ocean fraction > 50%, assign 'Ocean'.\n# If sea-ice fraction > 50%, assign 'Sea-Ice'.\n# If none of the above conditions are met, it could indicate a mixed or transitional zone, and you may assign a label such as 'Mixed' or 'Transitional'.\n# \n# Daytime/Nighttime Indicator Creation:\n# Use solar insolation or solar zenith angle to determine whether it's daytime or nighttime at each time step and location.\n# For solar insolation, set a threshold (e.g., if solar insolation > 0, it's daytime; otherwise, it's nighttime).\n# For the solar zenith angle, set a threshold (e.g., if solar zenith angle < 90°, it's daytime; otherwise, it's nighttime).\n# Create a binary indicator (0 for nighttime, 1 for daytime) based on the chosen criterion.\n\n","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:41:36.755038Z","iopub.execute_input":"2024-05-11T09:41:36.75534Z","iopub.status.idle":"2024-05-11T09:41:36.767821Z","shell.execute_reply.started":"2024-05-11T09:41:36.755315Z","shell.execute_reply":"2024-05-11T09:41:36.766909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# modelling","metadata":{}},{"cell_type":"code","source":"# specify your configurations as a dict\n# try more leaves\n# try more bins\n\nparams = {\n    \"boosting_type\": \"gbdt\",\n    \"objective\": \"regression_l1\",\n    \"metric\": [\"l2\", \"l1\"],\n    \"num_leaves\": 62, # was 62\n    \"learning_rate\": 0.1, # was 0.1\n    #\"feature_fraction\": 0.5,\n    #\"bagging_fraction\": 0.4,\n    #\"bagging_freq\": 5,\n    \"verbose\": -1, # -1 for low verbosity\n    \"max_bins\": 15, # or 63, adapted from https://lightgbm.readthedocs.io/en/latest/GPU-Performance.html\n    \"device\": \"gpu\",\n    \"gpu_platform_id\": 0,\n    \"gpu_device_id\": 0,\n    \"early_stopping_round\": 100,\n    \"num_boost_round\": 200, # was 100\n    \"min_gain_to_split\": 1e-12,\n    #\"num_leaves\": \n    #'num_boost_round':20,\n}","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:41:36.768852Z","iopub.execute_input":"2024-05-11T09:41:36.769145Z","iopub.status.idle":"2024-05-11T09:41:36.778097Z","shell.execute_reply.started":"2024-05-11T09:41:36.769121Z","shell.execute_reply":"2024-05-11T09:41:36.777299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nwarnings.simplefilter(\"ignore\", UserWarning)\nmodels = {}\nevals = {}\n\nfor col in tqdm(y_train.columns):    \n    lgb_train = lgb.Dataset(X_train, y_train[col],params={'verbose': -1},)\n    lgb_valid = lgb.Dataset(X_valid_1, y_valid_1[col], reference=lgb_train,params={'verbose': -1},)\n\n    lgb_model = lgb.train(params, lgb_train, valid_sets=[lgb_valid],callbacks = [lgb.record_evaluation(evals)])\n    models[col] = lgb_model","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:41:36.779612Z","iopub.execute_input":"2024-05-11T09:41:36.779981Z","iopub.status.idle":"2024-05-11T09:50:28.288626Z","shell.execute_reply.started":"2024-05-11T09:41:36.779947Z","shell.execute_reply":"2024-05-11T09:50:28.287796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgb.plot_metric(evals, 'l2')","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:52:07.117294Z","iopub.execute_input":"2024-05-11T09:52:07.117665Z","iopub.status.idle":"2024-05-11T09:52:07.478202Z","shell.execute_reply.started":"2024-05-11T09:52:07.117635Z","shell.execute_reply":"2024-05-11T09:52:07.477278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(models)","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:52:11.077234Z","iopub.execute_input":"2024-05-11T09:52:11.077616Z","iopub.status.idle":"2024-05-11T09:52:11.083545Z","shell.execute_reply.started":"2024-05-11T09:52:11.077586Z","shell.execute_reply":"2024-05-11T09:52:11.082638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_valid_1.shape, y_valid_2.shape","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:52:12.208186Z","iopub.execute_input":"2024-05-11T09:52:12.2087Z","iopub.status.idle":"2024-05-11T09:52:12.215239Z","shell.execute_reply.started":"2024-05-11T09:52:12.208664Z","shell.execute_reply":"2024-05-11T09:52:12.2143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# predict\ny_preds = {}\n\nfor target in tqdm(models):\n    y_pred = models[target].predict(X_valid_2,)\n    y_preds[target] = y_pred\n    \ny_preds = pd.DataFrame(y_preds)","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:52:13.24464Z","iopub.execute_input":"2024-05-11T09:52:13.245479Z","iopub.status.idle":"2024-05-11T09:54:10.879991Z","shell.execute_reply.started":"2024-05-11T09:52:13.245445Z","shell.execute_reply":"2024-05-11T09:54:10.879179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_preds.values[:] = target_scaler.inverse_transform(y_preds)","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:54:21.368387Z","iopub.execute_input":"2024-05-11T09:54:21.368744Z","iopub.status.idle":"2024-05-11T09:54:21.57651Z","shell.execute_reply.started":"2024-05-11T09:54:21.368719Z","shell.execute_reply":"2024-05-11T09:54:21.575659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_preds.shape, y_valid_2.shape","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:54:22.7996Z","iopub.execute_input":"2024-05-11T09:54:22.8005Z","iopub.status.idle":"2024-05-11T09:54:22.807307Z","shell.execute_reply.started":"2024-05-11T09:54:22.800454Z","shell.execute_reply":"2024-05-11T09:54:22.806302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# use predictions from models if they are available else use means\ny_preds_all = {\n    c: y_preds.get(c) if y_preds.get(c) is not None else target_means.y_means[c] for c in y_valid_2.columns\n}","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:54:24.712441Z","iopub.execute_input":"2024-05-11T09:54:24.712795Z","iopub.status.idle":"2024-05-11T09:54:24.730838Z","shell.execute_reply.started":"2024-05-11T09:54:24.712768Z","shell.execute_reply":"2024-05-11T09:54:24.729916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_preds_all = pd.DataFrame(y_preds_all)\ny_preds_all.shape","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:54:26.177525Z","iopub.execute_input":"2024-05-11T09:54:26.178134Z","iopub.status.idle":"2024-05-11T09:54:26.275116Z","shell.execute_reply.started":"2024-05-11T09:54:26.178101Z","shell.execute_reply":"2024-05-11T09:54:26.2742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Experimentation log:\n# score dropped from 0.3 to 0.27 when decreasing training samples from 100.000 to 5.000\n# plus excluding features with no importance to the LGBM model\n# train time dropped from 30 min to 5 min\n\n# standardscaling of target features had +0.05 impact on R^2\n# no impact on training time\n\n# adding additional estimators (100->1000) \n\n# previous results are not valid since evaluation was done only on selected targets\n# change lr from 0.01 to 0.1 improved results from 0.13 to 0.19\n# change num of leaves from 31 to 62 is about the same at 0.19146\n# change max_bins from 15 to 65: from 10 min training time to about 20 min, r2 from 0.19146 -> 0.19198\n# num_of_boost_rounds from 500 to 250\n\n# Future\n# change double num of samples and change bagging freq to 0.5 r2 drop to -1486\n\nr2 = r2_score(y_valid_2, y_preds_all[y_valid_2.columns])\nprint(f\"The overall R^2 of prediction is: {r2}\")","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:54:28.999435Z","iopub.execute_input":"2024-05-11T09:54:29.000209Z","iopub.status.idle":"2024-05-11T09:54:29.500369Z","shell.execute_reply.started":"2024-05-11T09:54:29.000169Z","shell.execute_reply":"2024-05-11T09:54:29.499298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"r2_per_target = {}\n\nfor col in y_preds.columns:\n    r2_per_target[col] = r2_score(y_valid_2[col], y_preds[col])","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:54:34.406291Z","iopub.execute_input":"2024-05-11T09:54:34.406925Z","iopub.status.idle":"2024-05-11T09:54:34.785613Z","shell.execute_reply.started":"2024-05-11T09:54:34.406893Z","shell.execute_reply":"2024-05-11T09:54:34.784826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.Series(r2_per_target).sort_values(ascending=False).head(10)","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:54:37.637982Z","iopub.execute_input":"2024-05-11T09:54:37.638344Z","iopub.status.idle":"2024-05-11T09:54:37.648769Z","shell.execute_reply.started":"2024-05-11T09:54:37.638315Z","shell.execute_reply":"2024-05-11T09:54:37.647758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.Series(r2_per_target).sort_values().head(10)","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:54:40.277656Z","iopub.execute_input":"2024-05-11T09:54:40.278314Z","iopub.status.idle":"2024-05-11T09:54:40.287238Z","shell.execute_reply.started":"2024-05-11T09:54:40.278275Z","shell.execute_reply":"2024-05-11T09:54:40.286093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.Series(r2_per_target).hist(bins=30);","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:54:44.551764Z","iopub.execute_input":"2024-05-11T09:54:44.552485Z","iopub.status.idle":"2024-05-11T09:54:44.847966Z","shell.execute_reply.started":"2024-05-11T09:54:44.552446Z","shell.execute_reply":"2024-05-11T09:54:44.847172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"array([r2_per_target[t] for t in r2_per_target]).mean()","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:54:45.842091Z","iopub.execute_input":"2024-05-11T09:54:45.842893Z","iopub.status.idle":"2024-05-11T09:54:45.849295Z","shell.execute_reply.started":"2024-05-11T09:54:45.842865Z","shell.execute_reply":"2024-05-11T09:54:45.848402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# investigate model","metadata":{}},{"cell_type":"code","source":"def get_feature_importances(model):\n    try:\n        return dict(feature=model.feature_names_in_, importance=model.feature_importances_)\n    except AttributeError: \n        return dict(feature=model.feature_name_, importance=model.feature_importances_)\n\ndef get_important_features(model):\n    return pd.DataFrame(get_feature_importances(model)).query('importance != 0.').feature.values","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:54:48.498512Z","iopub.execute_input":"2024-05-11T09:54:48.499422Z","iopub.status.idle":"2024-05-11T09:54:48.504944Z","shell.execute_reply.started":"2024-05-11T09:54:48.499388Z","shell.execute_reply":"2024-05-11T09:54:48.503828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_importances = pd.DataFrame(index=X_train.columns, data={\n    t: models[t].feature_importance(\"gain\") for t in y_train.columns\n})\n\nmean_feature_importances = feature_importances.sum(axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:54:49.157379Z","iopub.execute_input":"2024-05-11T09:54:49.157743Z","iopub.status.idle":"2024-05-11T09:54:49.176125Z","shell.execute_reply.started":"2024-05-11T09:54:49.157715Z","shell.execute_reply":"2024-05-11T09:54:49.175405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scaled_mean_feature_importances = mean_feature_importances #/ mean_feature_importances.max()","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:54:49.872815Z","iopub.execute_input":"2024-05-11T09:54:49.873458Z","iopub.status.idle":"2024-05-11T09:54:49.877868Z","shell.execute_reply.started":"2024-05-11T09:54:49.873426Z","shell.execute_reply":"2024-05-11T09:54:49.876788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scaled_mean_feature_importances.sort_values(ascending=False)[:10].to_dict()","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:55:03.708942Z","iopub.execute_input":"2024-05-11T09:55:03.70936Z","iopub.status.idle":"2024-05-11T09:55:03.718774Z","shell.execute_reply.started":"2024-05-11T09:55:03.709327Z","shell.execute_reply":"2024-05-11T09:55:03.717592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unimportant_features = array(scaled_mean_feature_importances[scaled_mean_feature_importances.eq(0)].index)","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:55:12.064781Z","iopub.execute_input":"2024-05-11T09:55:12.065143Z","iopub.status.idle":"2024-05-11T09:55:12.070024Z","shell.execute_reply.started":"2024-05-11T09:55:12.065115Z","shell.execute_reply":"2024-05-11T09:55:12.069044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(unimportant_features)","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:55:12.890646Z","iopub.execute_input":"2024-05-11T09:55:12.891314Z","iopub.status.idle":"2024-05-11T09:55:12.89707Z","shell.execute_reply.started":"2024-05-11T09:55:12.89128Z","shell.execute_reply":"2024-05-11T09:55:12.896121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unimportant_features","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:55:14.321223Z","iopub.execute_input":"2024-05-11T09:55:14.322134Z","iopub.status.idle":"2024-05-11T09:55:14.327857Z","shell.execute_reply.started":"2024-05-11T09:55:14.3221Z","shell.execute_reply":"2024-05-11T09:55:14.326821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* plots only included for the first target","metadata":{}},{"cell_type":"code","source":"lgb.plot_split_value_histogram(models['ptend_t_0'], feature_columns[0]);","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:55:26.878602Z","iopub.execute_input":"2024-05-11T09:55:26.879449Z","iopub.status.idle":"2024-05-11T09:55:27.156199Z","shell.execute_reply.started":"2024-05-11T09:55:26.87941Z","shell.execute_reply":"2024-05-11T09:55:27.155285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgb.plot_tree(models['ptend_t_5'],figsize=(36,12),dpi=360);","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:55:28.590531Z","iopub.execute_input":"2024-05-11T09:55:28.591389Z","iopub.status.idle":"2024-05-11T09:55:35.598868Z","shell.execute_reply.started":"2024-05-11T09:55:28.591354Z","shell.execute_reply":"2024-05-11T09:55:35.59796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgb.plot_importance(models['ptend_t_0'], importance_type=\"gain\", max_num_features=20, dpi=120,\n                    figsize=(7,6), title=\"LightGBM Feature Importance (Gain)\");","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:55:35.60074Z","iopub.execute_input":"2024-05-11T09:55:35.601075Z","iopub.status.idle":"2024-05-11T09:55:36.140896Z","shell.execute_reply.started":"2024-05-11T09:55:35.601046Z","shell.execute_reply":"2024-05-11T09:55:36.139982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgb.plot_importance(models['ptend_t_0'], importance_type=\"split\", max_num_features=20, dpi=120,\n                    figsize=(7, 6), title=\"LightGBM Feature Importance (Split)\");","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:55:36.142086Z","iopub.execute_input":"2024-05-11T09:55:36.142399Z","iopub.status.idle":"2024-05-11T09:55:36.664968Z","shell.execute_reply.started":"2024-05-11T09:55:36.142372Z","shell.execute_reply":"2024-05-11T09:55:36.664076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# submit predictions","metadata":{}},{"cell_type":"code","source":"%%time\n\ntest = pl.read_csv(\"/kaggle/input/leap-atmospheric-physics-ai-climsim/test.csv\")\ntest.shape","metadata":{"execution":{"iopub.status.busy":"2024-05-11T09:55:36.666926Z","iopub.execute_input":"2024-05-11T09:55:36.667295Z","iopub.status.idle":"2024-05-11T09:56:10.647526Z","shell.execute_reply.started":"2024-05-11T09:55:36.667239Z","shell.execute_reply":"2024-05-11T09:56:10.646515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = test.to_pandas()","metadata":{"execution":{"iopub.status.busy":"2024-05-11T10:00:02.903614Z","iopub.execute_input":"2024-05-11T10:00:02.904539Z","iopub.status.idle":"2024-05-11T10:00:05.122588Z","shell.execute_reply.started":"2024-05-11T10:00:02.904504Z","shell.execute_reply":"2024-05-11T10:00:05.121825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# add features\ntest = add_group_col_stats(test)\ntest = add_camin(test)\ntest = add_wind_speed(test)\ntest = add_dominant_surface(test)","metadata":{"execution":{"iopub.status.busy":"2024-05-11T10:00:06.824861Z","iopub.execute_input":"2024-05-11T10:00:06.825618Z","iopub.status.idle":"2024-05-11T10:00:16.354652Z","shell.execute_reply.started":"2024-05-11T10:00:06.825589Z","shell.execute_reply":"2024-05-11T10:00:16.353591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# transform features\nX_test = test[non_zero_features]\nX_test.values[:] = feature_scaler.transform(X_test)","metadata":{"execution":{"iopub.status.busy":"2024-05-11T10:01:58.192438Z","iopub.execute_input":"2024-05-11T10:01:58.193433Z","iopub.status.idle":"2024-05-11T10:02:07.395848Z","shell.execute_reply.started":"2024-05-11T10:01:58.193388Z","shell.execute_reply":"2024-05-11T10:02:07.394921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# predict\ny_preds_test = {}\n\nfor target in tqdm(models):\n    y_pred = models[target].predict(X_test,)\n    y_preds_test[target] = y_pred\n    \ny_preds_test = pd.DataFrame(y_preds_test)","metadata":{"execution":{"iopub.status.busy":"2024-05-11T10:02:09.160162Z","iopub.execute_input":"2024-05-11T10:02:09.16088Z","iopub.status.idle":"2024-05-11T10:15:03.802623Z","shell.execute_reply.started":"2024-05-11T10:02:09.160849Z","shell.execute_reply":"2024-05-11T10:15:03.801516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# inverse trasform targets\ny_preds_test.values[:] = target_scaler.inverse_transform(y_preds_test)","metadata":{"execution":{"iopub.status.busy":"2024-05-11T10:18:38.225414Z","iopub.execute_input":"2024-05-11T10:18:38.225807Z","iopub.status.idle":"2024-05-11T10:18:39.575626Z","shell.execute_reply.started":"2024-05-11T10:18:38.225774Z","shell.execute_reply":"2024-05-11T10:18:39.574703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# use predictions from models if they are available else use means\ny_preds_test_all = {\n    c: y_preds_test.get(c) if y_preds_test.get(c) is not None else target_means.y_means[c] for c in y_valid_2.columns\n}","metadata":{"execution":{"iopub.status.busy":"2024-05-11T10:18:41.099101Z","iopub.execute_input":"2024-05-11T10:18:41.099898Z","iopub.status.idle":"2024-05-11T10:18:41.117264Z","shell.execute_reply.started":"2024-05-11T10:18:41.099866Z","shell.execute_reply":"2024-05-11T10:18:41.116395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create dataframe\nsubmission = pd.DataFrame(y_preds_test_all)\nsubmission.shape","metadata":{"execution":{"iopub.status.busy":"2024-05-11T10:18:42.640837Z","iopub.execute_input":"2024-05-11T10:18:42.641687Z","iopub.status.idle":"2024-05-11T10:18:43.210811Z","shell.execute_reply.started":"2024-05-11T10:18:42.641654Z","shell.execute_reply":"2024-05-11T10:18:43.209833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# set index\nsubmission.index = test['sample_id']","metadata":{"execution":{"iopub.status.busy":"2024-05-11T10:20:13.394617Z","iopub.execute_input":"2024-05-11T10:20:13.395003Z","iopub.status.idle":"2024-05-11T10:20:13.400209Z","shell.execute_reply.started":"2024-05-11T10:20:13.394973Z","shell.execute_reply":"2024-05-11T10:20:13.399118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get weights\nsamplesub = pd.read_csv(\"/kaggle/input/leap-atmospheric-physics-ai-climsim/sample_submission.csv\", chunksize=10)\nsample_batch = next(samplesub)\n\nweights = sample_batch[target_columns].iloc[0,:]","metadata":{"execution":{"iopub.status.busy":"2024-05-11T10:20:14.842269Z","iopub.execute_input":"2024-05-11T10:20:14.842632Z","iopub.status.idle":"2024-05-11T10:20:14.882316Z","shell.execute_reply.started":"2024-05-11T10:20:14.842604Z","shell.execute_reply":"2024-05-11T10:20:14.881484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# weight predictions\nfor key in tqdm(weights.to_dict()):\n    submission[key] = submission[key] * weights[key]","metadata":{"execution":{"iopub.status.busy":"2024-05-11T10:20:16.517499Z","iopub.execute_input":"2024-05-11T10:20:16.517881Z","iopub.status.idle":"2024-05-11T10:20:17.244454Z","shell.execute_reply.started":"2024-05-11T10:20:16.517854Z","shell.execute_reply":"2024-05-11T10:20:17.243501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# write submission\nsubmission = pl.from_pandas(submission.reset_index().rename({'index': 'sample_id'}, axis=1))\nsubmission.write_csv(\"/kaggle/working/submission.csv\",)","metadata":{"execution":{"iopub.status.busy":"2024-05-11T10:20:18.83696Z","iopub.execute_input":"2024-05-11T10:20:18.83783Z","iopub.status.idle":"2024-05-11T10:20:37.954825Z","shell.execute_reply.started":"2024-05-11T10:20:18.837795Z","shell.execute_reply":"2024-05-11T10:20:37.953706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}