{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":56537,"databundleVersionId":8015876,"sourceType":"competition"},{"sourceId":8280197,"sourceType":"datasetVersion","datasetId":4917313}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"In this notebook we'll try to estimate how many targets in the given task can be predicted with R2-score > 0, \n\nShare your findings as well!","metadata":{}},{"cell_type":"code","source":"from pathlib import Path\n\nimport tqdm\n\nimport polars as pl\nfrom sklearn.metrics import r2_score\n\nstorage_dir = Path('/kaggle/input/leap-atmospheric-physics-ai-climsim')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-03T12:50:45.487391Z","iopub.execute_input":"2024-05-03T12:50:45.48807Z","iopub.status.idle":"2024-05-03T12:50:47.354126Z","shell.execute_reply.started":"2024-05-03T12:50:45.488036Z","shell.execute_reply":"2024-05-03T12:50:47.353201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Total size of the dataset is 10,091,520 samples, in my experiments I'm using 8,000,000 rows to train and the rest 2kk+ to validate.","metadata":{}},{"cell_type":"code","source":"df = pl.read_parquet('/kaggle/input/leap-example-valid/example_valid.parquet')\ndf","metadata":{"execution":{"iopub.status.busy":"2024-05-03T12:50:47.357551Z","iopub.execute_input":"2024-05-03T12:50:47.358053Z","iopub.status.idle":"2024-05-03T12:51:19.104377Z","shell.execute_reply.started":"2024-05-03T12:50:47.35801Z","shell.execute_reply":"2024-05-03T12:51:19.101859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"single_targets = [\n    'cam_out_NETSW',\n    'cam_out_FLWDS',\n    'cam_out_PRECSC',\n    'cam_out_PRECC',\n    'cam_out_SOLS',\n    'cam_out_SOLL',\n    'cam_out_SOLSD',\n    'cam_out_SOLLD',\n]\n\nseq_targets = [\n    'ptend_t',\n    'ptend_q0001',\n    'ptend_q0002',\n    'ptend_q0003',\n    'ptend_u',\n    'ptend_v',\n]\n\ntarget_columns = []\nfor col in seq_targets:\n    for i in range(60):\n        target_columns.append(f\"{col}_{i}\")\ntarget_columns.extend(single_targets)\nlen(target_columns)","metadata":{"execution":{"iopub.status.busy":"2024-05-03T12:51:19.106923Z","iopub.execute_input":"2024-05-03T12:51:19.108648Z","iopub.status.idle":"2024-05-03T12:51:19.12257Z","shell.execute_reply.started":"2024-05-03T12:51:19.108604Z","shell.execute_reply":"2024-05-03T12:51:19.121048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"r2_scores = {}\nfor col in tqdm.tqdm(target_columns):\n    r2_scores[col] = r2_score(df[f\"{col}_target\"], df[f\"{col}_pred\"])","metadata":{"execution":{"iopub.status.busy":"2024-05-03T12:51:19.126139Z","iopub.execute_input":"2024-05-03T12:51:19.126652Z","iopub.status.idle":"2024-05-03T12:51:25.464187Z","shell.execute_reply.started":"2024-05-03T12:51:19.126612Z","shell.execute_reply":"2024-05-03T12:51:25.462911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Let's calculate the CV score of this sub","metadata":{}},{"cell_type":"code","source":"import numpy as np\n\nnp.mean(list(r2_scores.values()))","metadata":{"execution":{"iopub.status.busy":"2024-05-03T12:51:25.465396Z","iopub.execute_input":"2024-05-03T12:51:25.465908Z","iopub.status.idle":"2024-05-03T12:51:25.475144Z","shell.execute_reply.started":"2024-05-03T12:51:25.465879Z","shell.execute_reply":"2024-05-03T12:51:25.473569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Check single-value target performance\n\nIt appears that all of the single-value targets are predictable","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nplt.figure(figsize=(16, 5))\nplt.grid()\nfor ft in single_targets:\n    plt.bar(single_targets, [r2_scores[col] for col in single_targets])\nfor col in single_targets:\n    print(f\"{col} : {r2_scores[col]}\")","metadata":{"execution":{"iopub.status.busy":"2024-05-03T12:51:25.476877Z","iopub.execute_input":"2024-05-03T12:51:25.477338Z","iopub.status.idle":"2024-05-03T12:51:26.046608Z","shell.execute_reply.started":"2024-05-03T12:51:25.4773Z","shell.execute_reply":"2024-05-03T12:51:26.045244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ss = pl.scan_csv(storage_dir / 'sample_submission.csv').slice(0, 1)\nss = ss.select([x for x in ss.columns if x != 'sample_id']).collect()\nss","metadata":{"execution":{"iopub.status.busy":"2024-05-03T12:51:26.048493Z","iopub.execute_input":"2024-05-03T12:51:26.048945Z","iopub.status.idle":"2024-05-03T12:51:26.191688Z","shell.execute_reply.started":"2024-05-03T12:51:26.048898Z","shell.execute_reply":"2024-05-03T12:51:26.190365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Check sequential target performance \n\nIn these plots I highlight the performance of the model on sequential targets. \n\nSome columns have weight set to 0 by **sample_submission** weights, I plotted this as well","metadata":{}},{"cell_type":"code","source":"for col in seq_targets:\n    plt.figure(figsize=(8, 4))\n    plt.scatter(range(60), [r2_scores[f\"{col}_{i}\"] for i in range(60)])\n    plt.scatter(range(60), [(ss[f\"{col}_{i}\"].to_numpy()[0] != 0) for i in range(60)], alpha=0.5)\n    plt.title(f\"{col}, 60 values\")\n    plt.grid()\n    plt.legend([\"R2 scores\", \"ss_weight != 0\"])\n    plt.ylim(0, 1)","metadata":{"execution":{"iopub.status.busy":"2024-05-03T12:51:26.193043Z","iopub.execute_input":"2024-05-03T12:51:26.193364Z","iopub.status.idle":"2024-05-03T12:51:28.981739Z","shell.execute_reply.started":"2024-05-03T12:51:26.193338Z","shell.execute_reply":"2024-05-03T12:51:28.980296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It seems that most of the columns can be predicted with positive R2 score, however there is one concerning segment, - the start of **ptend_q0002**.\n\nFrom the position 10 up to until 30 my model can't predict **ptend_q0002** values at all.\n\nIf we check public baselines (e.g. [Giba's XGBoost notebook](https://www.kaggle.com/code/titericz/giba-baseline-xgboost)), it appears that GBDTs are also struggling with these columns.","metadata":{}},{"cell_type":"markdown","source":"### Lets check the least performant columns","metadata":{}},{"cell_type":"markdown","source":"As we already ruled out, the most problematic columns are the start of scored segment in **ptend_q0002** target, we need to analyze it closer to figure out if it can be predicted at all","metadata":{}},{"cell_type":"code","source":"r2_scores = {k:v for k,v in sorted(r2_scores.items(), key=lambda x : x[1])}\nfor k, v in r2_scores.items():\n    print(f\"{k} : {v:.5f}\")\n    if v > 0.3:\n        break","metadata":{"execution":{"iopub.status.busy":"2024-05-03T12:51:28.98362Z","iopub.execute_input":"2024-05-03T12:51:28.983993Z","iopub.status.idle":"2024-05-03T12:51:28.990907Z","shell.execute_reply.started":"2024-05-03T12:51:28.983964Z","shell.execute_reply":"2024-05-03T12:51:28.989634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import json\n\nwith open('baseline_r2_scores.json', 'w') as f:\n    json.dump(r2_scores, f, indent=4)","metadata":{"execution":{"iopub.status.busy":"2024-05-03T12:52:35.462681Z","iopub.execute_input":"2024-05-03T12:52:35.463186Z","iopub.status.idle":"2024-05-03T12:52:35.471171Z","shell.execute_reply.started":"2024-05-03T12:52:35.463151Z","shell.execute_reply":"2024-05-03T12:52:35.470174Z"},"trusted":true},"execution_count":null,"outputs":[]}]}