{"metadata":{"kernelspec":{"name":"ir","display_name":"R","language":"R"},"language_info":{"name":"R","codemirror_mode":"r","pygments_lexer":"r","mimetype":"text/x-r-source","file_extension":".r","version":"4.0.5"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":56537,"databundleVersionId":8015876,"sourceType":"competition"}],"dockerImageVersionId":30618,"isInternetEnabled":true,"language":"r","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# <center>**LEAP - Atmospheric Physics using AI (ClimSim)**</center>\n### This competition features a large training data set, so this notebook will focus on helping R users read the competition material and begin to explore the data.","metadata":{}},{"cell_type":"markdown","source":"# Library load...📚 ","metadata":{}},{"cell_type":"code","source":"library(tidyverse) # metapackage of all tidyverse packages\nlibrary(tictoc) # time management\nlibrary(data.table) #fread function\nlibrary(ggthemes) # themes for ggplot\nlibrary(arrow) # parquet\nlist.files(path = \"../input/leap-atmospheric-physics-ai-climsim/\")","metadata":{"_uuid":"051d70d956493feee0c6d64651c6a088724dca2a","_execution_state":"idle","execution":{"iopub.status.busy":"2024-05-20T13:35:35.85053Z","iopub.execute_input":"2024-05-20T13:35:35.934306Z","iopub.status.idle":"2024-05-20T13:35:35.962738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"***","metadata":{}},{"cell_type":"markdown","source":"# Chunking **train.csv** into RDS files...🗄️\n## Making manageable chunks of data","metadata":{}},{"cell_type":"code","source":"# As the code below will show, the training data is large (181.72 GB) and has over 10 million rows.  \n# In order to train a dataset that is larger than memory, we need to separate into smaller chunks.  \n# The following code will show how to sample the training data and save into individual .RDS files for further processing.\n# In the example below, each file with all training columns & consisting of 50k rows, will consume ~ 257MB disk space in /kaggle/working.\n\n# Initialize an empty data table\ndt <- data.table()\n\n# Define the chunk size, number of rows to read, and starting position to read\nchunk_size <- 50000\nn_rows <- 500000\nstart_position <- 1\n\nk<-1 # for use in file naming & debugging\n\ntic(paste('Chunking data'))\n\n# Use fread to read in chunks\nfor (i in seq(start_position, n_rows, by=chunk_size)) {\n    dt <- fread('../input/leap-atmospheric-physics-ai-climsim/train.csv', skip = i-1, nrows = chunk_size)\n    saveRDS(dt, paste('/kaggle/working/',k,'_climsim_train.RDS',sep=''))\n    print(paste('writing file...','/kaggle/working/',k,'_climsim_train.RDS',sep=''))\n    k <- k+1\n}\ntoc()\n\nremove(dt) # cleanup memory usage","metadata":{"execution":{"iopub.status.busy":"2024-05-20T13:39:35.920357Z","iopub.execute_input":"2024-05-20T13:39:35.922137Z","iopub.status.idle":"2024-05-20T13:47:12.305321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"***","metadata":{}},{"cell_type":"markdown","source":"# Reading **train.csv**...🔍\n## Getting the number of columns in **train.csv**","metadata":{}},{"cell_type":"code","source":"# Because the train.csv file is large (181.72 GB), we need to break up the read process even for initial exploration.  \n# Below we'll read the first 10 rows, allowing us to see the columns and some sample data.\n\ntic('train.csv read first 10 rows')\ntest_10_rows <- fread('../input/leap-atmospheric-physics-ai-climsim/train.csv', nrows=10)\ntoc()\n\n# Initial look at train data\ntest_10_rows\n\n# 925 columns\ntest_10_rows %>% ncol()","metadata":{"execution":{"iopub.status.busy":"2024-04-29T13:39:12.969653Z","iopub.execute_input":"2024-04-29T13:39:12.971336Z","iopub.status.idle":"2024-04-29T13:39:13.448311Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Getting the number of rows in **train.csv**","metadata":{}},{"cell_type":"code","source":"# Since we have the column names, we will use the first column name, 'sample_id', and get all its rows.  \n# This will allow us to get the number of rows in that column, and thus the number of rows in the train data set.\n# This block took ~ 1606.614 sec\n\ntic('train.csv col sample_id')\ntest_first_col <- fread('../input/leap-atmospheric-physics-ai-climsim/train.csv', select = 'sample_id')\ntest_first_col %>% nrow()\ntoc()\n\n# 10091520 rows of data in train.csv","metadata":{"execution":{"iopub.status.busy":"2024-04-26T23:20:16.20894Z","iopub.execute_input":"2024-04-26T23:20:16.264141Z","iopub.status.idle":"2024-04-26T23:47:02.92705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Separating out target columns.","metadata":{}},{"cell_type":"code","source":"# From the competition information, we know there are 6 target columns with 60 dimensions, in addition to 8 target columns with no additional dimensions, so 368 target columns in total.  \n# Instead of writing out the dimensions by hand, we'll write them as given in the competition instructions, \n# then match for the full list of target columns, including all the dimensions, from the test_10_rows df we used earlier.\n\n# From the competition data, we know these are the target columns (first 6 have 60 dimensions)\ntarget_cols <- c('ptend_t', 'ptend_q0001', 'ptend_q0002', 'ptend_q0003', 'ptend_u', 'ptend_v', 'cam_out_NETSW', 'cam_out_FLWDS', \n                 'cam_out_PRECSC', 'cam_out_PRECC', 'cam_out_SOLS', 'cam_out_SOLL', 'cam_out_SOLSD', 'cam_out_SOLLD')\n\n# grab the column names from the matches\ntarget_cols <- test_10_rows %>% select(matches(target_cols)) %>% colnames()\n\n# 368, that's what we want!\nlength(target_cols)\n\n# Quick view of the dimenions associated with the first target column (ptend_t).\ntarget_cols %>% head()","metadata":{"execution":{"iopub.status.busy":"2024-04-29T13:39:27.872665Z","iopub.execute_input":"2024-04-29T13:39:27.87437Z","iopub.status.idle":"2024-04-29T13:39:27.954497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Reading **test.csv**...🔍","metadata":{}},{"cell_type":"code","source":"# The test data is significantly smaller (~7Gb)\n\ntic('test.csv read')\ntest <- fread('../input/leap-atmospheric-physics-ai-climsim/test.csv')\ntoc()\n\n# 625000 rows of sample data\ntest %>% nrow()\n\n# 557 test columns\ntest %>% ncol()\n\n# 0 NAs\nsum(is.na(test))","metadata":{"execution":{"iopub.status.busy":"2024-04-29T13:43:21.888879Z","iopub.execute_input":"2024-04-29T13:43:21.890807Z","iopub.status.idle":"2024-04-29T13:44:24.025773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Initial data exploration...🔬","metadata":{}},{"cell_type":"code","source":"# Let's look at the distribution of one of the target columns with its dimensions.\n\n# Adjust plot size\noptions(repr.plot.width = 18, repr.plot.height =8)\n\n# Read in 5 dimensions of 'ptend_t', with 1000 rows each.\ntic('1k read of ptend_t_0')\nptend_t_sample <- fread('../input/leap-atmospheric-physics-ai-climsim/train.csv', select = c('ptend_t_0', 'ptend_t_1', 'ptend_t_2', 'ptend_t_3', 'ptend_t_4'), nrows = 1000)\ntoc()\n\n# Plot & compare the distributions.\nptend_t_sample %>% pivot_longer(cols = starts_with('ptend_t')) %>% ggplot(aes(value)) + geom_histogram(fill = 'steelblue') + facet_grid(cols=vars(name)) + theme_clean()","metadata":{"execution":{"iopub.status.busy":"2024-04-29T14:54:39.707812Z","iopub.execute_input":"2024-04-29T14:54:39.709699Z","iopub.status.idle":"2024-04-29T14:54:40.719038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let's look at the distribution of the 8 target columns that do not have any additional dimensions.\n\n# Adjust plot size\noptions(repr.plot.width = 22, repr.plot.height =8)\n\n# Read in 5 dimensions of 'ptend_t', with 1000 rows each.\ntic('1k read of ptend_t_0')\ncam_out_sample <- fread('../input/leap-atmospheric-physics-ai-climsim/train.csv', select = c('cam_out_NETSW', 'cam_out_FLWDS', 'cam_out_PRECSC', 'cam_out_PRECC', 'cam_out_SOLS', 'cam_out_SOLL', 'cam_out_SOLSD', 'cam_out_SOLLD'), nrows = 1000)\ntoc()\n\n# Plot & compare the distributions.\ncam_out_sample %>% pivot_longer(cols = starts_with('cam_out')) %>% ggplot(aes(value)) + geom_histogram(fill = 'steelblue') + facet_grid(cols=vars(name)) + theme_clean()","metadata":{"execution":{"iopub.status.busy":"2024-04-29T14:59:36.759785Z","iopub.execute_input":"2024-04-29T14:59:36.76245Z","iopub.status.idle":"2024-04-29T14:59:37.947267Z"},"trusted":true},"execution_count":null,"outputs":[]}]}