{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":56537,"databundleVersionId":8015876,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"%%capture\n\nimport numpy as np\nimport pandas as pd\n!pip install netCDF4\nimport netCDF4\nfrom huggingface_hub import HfFileSystem\nimport xarray as xr\nimport polars as pl\nfrom huggingface_hub import hf_hub_download\nimport pickle\nfrom IPython.utils import io","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-30T15:01:53.742685Z","iopub.execute_input":"2024-05-30T15:01:53.74339Z","iopub.status.idle":"2024-05-30T15:02:11.43878Z","shell.execute_reply.started":"2024-05-30T15:01:53.743352Z","shell.execute_reply":"2024-05-30T15:02:11.437461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pl.read_csv('/kaggle/input/leap-atmospheric-physics-ai-climsim/train.csv', n_rows = 10)\n#train_df = pd.read_csv('/kaggle/input/leap-atmospheric-physics-ai-climsim/train.csv', float_precision='round_trip', nrows = 384)\n\nFEAT_COLS = train_df.columns[1:557]\nTARGET_COLS = train_df.columns[557:]\nX_cols = FEAT_COLS\ny_cols = TARGET_COLS\n\nX_cols_series_parents = ['state_t', 'state_q0001', 'state_q0002', 'state_q0003', 'state_u', 'state_v',\n                         'pbuf_ozone', 'pbuf_CH4', 'pbuf_N2O']\nX_cols_series = [x for x in X_cols if np.mean([x.startswith(y) for y in X_cols_series_parents])>0]\nX_cols_series_NOT = [x for x in X_cols if x not in X_cols_series]\nprint(len(X_cols_series)+len(X_cols_series_NOT) == len(X_cols))\nX_cols_series_indices = np.asarray([np.where(np.asarray(X_cols) == x)[0][0] for x in X_cols_series])\nX_cols_series_NOT_indices = np.asarray([np.where(np.asarray(X_cols) == x)[0][0] for x in X_cols_series_NOT])\n\ny_cols_series_parents = ['ptend_t', 'ptend_q0001', 'ptend_q0002', 'ptend_q0003', 'ptend_u', 'ptend_v']\ny_cols_series = [x for x in y_cols if np.mean([x.startswith(y) for y in y_cols_series_parents])>0]\ny_cols_series_NOT = [x for x in y_cols if x not in y_cols_series]\nprint(len(y_cols_series)+len(y_cols_series_NOT) == len(y_cols))\ny_cols_series_indices = np.asarray([np.where(np.asarray(y_cols) == x)[0][0] for x in y_cols_series])\ny_cols_series_NOT_indices = np.asarray([np.where(np.asarray(y_cols) == x)[0][0] for x in y_cols_series_NOT])","metadata":{"execution":{"iopub.status.busy":"2024-05-30T15:02:11.440914Z","iopub.execute_input":"2024-05-30T15:02:11.441567Z","iopub.status.idle":"2024-05-30T15:02:12.131685Z","shell.execute_reply.started":"2024-05-30T15:02:11.441535Z","shell.execute_reply":"2024-05-30T15:02:12.130644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfs = HfFileSystem()\nall_files = fs.glob(\"datasets/LEAP/ClimSim_low-res/train/*/E3SM-MMF.mli.*.nc\")\nprint(len(all_files)*384/7)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T15:02:12.132779Z","iopub.execute_input":"2024-05-30T15:02:12.13305Z","iopub.status.idle":"2024-05-30T15:03:49.893389Z","shell.execute_reply.started":"2024-05-30T15:02:12.133028Z","shell.execute_reply":"2024-05-30T15:03:49.892209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_data_cols(nc):\n    data_cols = []\n    for col in X_cols_series_parents:\n        arr = np.asarray(nc[col])\n        arr = np.transpose(arr, [1,0])\n        data_cols.append(arr)\n    data_cols = np.concatenate(data_cols, axis = -1)\n    return data_cols\n\ndef get_data_cols_not(nc):\n    data_cols_not = []\n    for col in X_cols_series_NOT:\n        arr = np.asarray(nc[col])\n        data_cols_not.append(arr)\n    data_cols_not = np.stack(data_cols_not, axis = -1)\n    return data_cols_not\n\ndef get_data_cols_target(nc, nc_target):\n    data_cols_targets = []\n    for col in X_cols_series_parents[:6]:\n        arr_in = np.asarray(nc[col])\n        arr_out = np.asarray(nc_target[col])\n        arr = (arr_out-arr_in)/1200\n        arr = np.transpose(arr, [1,0])\n        data_cols_targets.append(arr)\n    data_cols_targets = np.concatenate(data_cols_targets, axis = -1)\n    return data_cols_targets\n\ndef get_data_cols_not_target(nc_target):\n    data_cols_not_target = []\n    for col in y_cols_series_NOT:\n        arr = np.asarray(nc_target[col])\n        data_cols_not_target.append(arr)\n    data_cols_not_target = np.stack(data_cols_not_target, axis = -1)\n    return data_cols_not_target","metadata":{"execution":{"iopub.status.busy":"2024-05-30T15:03:49.896395Z","iopub.execute_input":"2024-05-30T15:03:49.896845Z","iopub.status.idle":"2024-05-30T15:03:49.90816Z","shell.execute_reply.started":"2024-05-30T15:03:49.896806Z","shell.execute_reply":"2024-05-30T15:03:49.907008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"files = [all_files[i] for i in range(21, len(all_files), 32)]\nprint(len(files))","metadata":{"execution":{"iopub.status.busy":"2024-05-30T15:03:49.909491Z","iopub.execute_input":"2024-05-30T15:03:49.909809Z","iopub.status.idle":"2024-05-30T15:03:49.930926Z","shell.execute_reply.started":"2024-05-30T15:03:49.909783Z","shell.execute_reply":"2024-05-30T15:03:49.929855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nall_data_cols = np.zeros((384*len(files), 540))\nall_data_cols_not = np.zeros((384*len(files), 16))\nall_data_cols_targets = np.zeros((384*len(files), 360))\nall_data_cols_not_target = np.zeros((384*len(files), 8))\ntimes = []\nfails = []\nfor i, file_name in enumerate(files):\n    try:\n        if i%10 == 0:\n            print(i)\n        file_name = file_name[30:]\n        with io.capture_output() as captured:\n            file = hf_hub_download(repo_id=\"LEAP/ClimSim_low-res\", filename=file_name, repo_type=\"dataset\");\n            file_target = hf_hub_download(repo_id=\"LEAP/ClimSim_low-res\", filename=file_name.replace('.mli.','.mlo.'), repo_type=\"dataset\");\n        nc = netCDF4.Dataset(file)\n        nc_target = netCDF4.Dataset(file_target)\n\n        data_cols = get_data_cols(nc)\n        data_cols_not = get_data_cols_not(nc)\n        data_cols_targets = get_data_cols_target(nc, nc_target)\n        data_cols_not_target = get_data_cols_not_target(nc_target)\n\n        all_data_cols[i*384:(i+1)*384, :] = data_cols\n        all_data_cols_not[i*384:(i+1)*384, :] = data_cols_not\n        all_data_cols_targets[i*384:(i+1)*384, :] = data_cols_targets\n        all_data_cols_not_target[i*384:(i+1)*384, :] = data_cols_not_target\n    except:\n        print(f'FAIL: {str(i)}')\n        fails.append(i)\n        all_data_cols[i*384:(i+1)*384, :] = np.nan\n        all_data_cols_not[i*384:(i+1)*384, :] = np.nan\n        all_data_cols_targets[i*384:(i+1)*384, :] = np.nan\n        all_data_cols_not_target[i*384:(i+1)*384, :] = np.nan","metadata":{"execution":{"iopub.status.busy":"2024-05-30T15:08:42.246086Z","iopub.execute_input":"2024-05-30T15:08:42.246478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"times = [file_name[57:-3] for file_name in files]\nprint(len(times))\ntimes[0]","metadata":{"execution":{"iopub.status.busy":"2024-05-30T15:03:49.943511Z","iopub.status.idle":"2024-05-30T15:03:49.944411Z","shell.execute_reply.started":"2024-05-30T15:03:49.944211Z","shell.execute_reply":"2024-05-30T15:03:49.944228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(all_data_cols.shape)\nprint(all_data_cols_not.shape)\nprint(all_data_cols_targets.shape)\nprint(all_data_cols_not_target.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T15:03:49.945279Z","iopub.status.idle":"2024-05-30T15:03:49.946214Z","shell.execute_reply.started":"2024-05-30T15:03:49.945838Z","shell.execute_reply":"2024-05-30T15:03:49.945863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pickle.dump(times, open('times.p', 'bw'))\npickle.dump(fails, open('fails.p', 'bw'))\nnp.save('all_data_cols.npy', all_data_cols)\nnp.save('all_data_cols_not.npy', all_data_cols_not)\nnp.save('all_data_cols_targets.npy', all_data_cols_targets)\nnp.save('all_data_cols_not_target.npy', all_data_cols_not_target)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T15:03:49.947899Z","iopub.status.idle":"2024-05-30T15:03:49.948909Z","shell.execute_reply.started":"2024-05-30T15:03:49.948619Z","shell.execute_reply":"2024-05-30T15:03:49.948642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(fails))","metadata":{"execution":{"iopub.status.busy":"2024-05-30T15:08:36.354741Z","iopub.execute_input":"2024-05-30T15:08:36.355157Z","iopub.status.idle":"2024-05-30T15:08:37.190524Z","shell.execute_reply.started":"2024-05-30T15:08:36.355107Z","shell.execute_reply":"2024-05-30T15:08:37.188905Z"},"trusted":true},"execution_count":null,"outputs":[]}]}