{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":56537,"databundleVersionId":8015876,"sourceType":"competition"}],"dockerImageVersionId":30699,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Summary\n- This code originates from previosly presented notebook, [Pytorch NN](https://www.kaggle.com/code/airazusta014/pytorch-nn), with some modifications.\n- Postprocessing was added according to https://www.kaggle.com/competitions/leap-atmospheric-physics-ai-climsim/discussion/502484.\n- Calculation of R2Score was added.\n- Targets are scaled before training according to submission weights.","metadata":{"_uuid":"753d95f2-ae01-43bb-a2c6-17c85501a290","_cell_guid":"398761aa-424e-4ad3-944b-caff61881900","jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"import gc\nimport os\nimport random\nimport time\nimport torch\nimport datetime\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchmetrics.regression import R2Score","metadata":{"execution":{"iopub.status.busy":"2024-05-16T01:00:04.677974Z","iopub.execute_input":"2024-05-16T01:00:04.678841Z","iopub.status.idle":"2024-05-16T01:00:13.103768Z","shell.execute_reply.started":"2024-05-16T01:00:04.678811Z","shell.execute_reply":"2024-05-16T01:00:13.102904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATA_PATH = \"/kaggle/input/\"\nBATCH_SIZE = 12288\nMIN_STD = 1e-8\nSCHEDULER_PATIENCE = 3\nSCHEDULER_FACTOR = 10**(-0.5)\nEPOCHS = 70\nPATIENCE = 6\nPRINT_FREQ = 50","metadata":{"execution":{"iopub.status.busy":"2024-05-16T01:00:13.105673Z","iopub.execute_input":"2024-05-16T01:00:13.106149Z","iopub.status.idle":"2024-05-16T01:00:13.111515Z","shell.execute_reply.started":"2024-05-16T01:00:13.106121Z","shell.execute_reply":"2024-05-16T01:00:13.110482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def format_time(elapsed):\n    \"\"\"Take a time in seconds and return a string hh:mm:ss.\"\"\"\n    elapsed_rounded = int(round((elapsed)))\n    return str(datetime.timedelta(seconds=elapsed_rounded))","metadata":{"execution":{"iopub.status.busy":"2024-05-16T01:00:13.112541Z","iopub.execute_input":"2024-05-16T01:00:13.112783Z","iopub.status.idle":"2024-05-16T01:00:13.123152Z","shell.execute_reply.started":"2024-05-16T01:00:13.112763Z","shell.execute_reply":"2024-05-16T01:00:13.122307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def seed_everything(seed_val=1325):\n    \"\"\"Seed everything.\"\"\"\n    random.seed(seed_val)\n    np.random.seed(seed_val)\n    torch.manual_seed(seed_val)\n    torch.cuda.manual_seed_all(seed_val)","metadata":{"execution":{"iopub.status.busy":"2024-05-16T01:00:13.124329Z","iopub.execute_input":"2024-05-16T01:00:13.124731Z","iopub.status.idle":"2024-05-16T01:00:13.13374Z","shell.execute_reply.started":"2024-05-16T01:00:13.124697Z","shell.execute_reply":"2024-05-16T01:00:13.132758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ts = time.time()\n\nweights = pd.read_csv(DATA_PATH + \"leap-atmospheric-physics-ai-climsim/sample_submission.csv\", nrows=1)\ndel weights['sample_id']\nweights = weights.T\nweights = weights.to_dict()[0]\n\ndf_train = pl.read_csv(DATA_PATH + \"leap-atmospheric-physics-ai-climsim/train.csv\", n_rows=2_500_000)\n\nfor target in weights:\n    df_train = df_train.with_columns(pl.col(target).mul(weights[target]))\n\nprint(\"Time to read dataset:\", format_time(time.time()-ts), flush=True)\n\nFEAT_COLS = df_train.columns[1:557]\nTARGET_COLS = df_train.columns[557:]\n\nfor col in FEAT_COLS:\n    df_train = df_train.with_columns(pl.col(col).cast(pl.Float32))\n\nfor col in TARGET_COLS:\n    df_train = df_train.with_columns(pl.col(col).cast(pl.Float32))\n\nx_train = df_train.select(FEAT_COLS).to_numpy()\ny_train = df_train.select(TARGET_COLS).to_numpy()\n\ndel df_train\ngc.collect()\n\n# Normalization of training data\nmx = x_train.mean(axis=0)\nsx = np.maximum(x_train.std(axis=0), MIN_STD)\nx_train = (x_train - mx.reshape(1,-1)) / sx.reshape(1,-1)\n\nmy = y_train.mean(axis=0)\nsy = np.maximum(np.sqrt((y_train*y_train).mean(axis=0)), MIN_STD)\ny_train = (y_train - my.reshape(1,-1)) / sy.reshape(1,-1)\n\nprint(\"Time after processing data:\", format_time(time.time()-ts), flush=True)","metadata":{"_kg_hide-input":false,"_kg_hide-output":false,"execution":{"iopub.status.busy":"2024-05-16T01:00:13.136056Z","iopub.execute_input":"2024-05-16T01:00:13.136314Z","iopub.status.idle":"2024-05-16T01:04:09.505483Z","shell.execute_reply.started":"2024-05-16T01:00:13.136292Z","shell.execute_reply":"2024-05-16T01:04:09.504415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"seed_everything()\ndevice = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")","metadata":{"_kg_hide-output":false,"_kg_hide-input":false,"execution":{"iopub.status.busy":"2024-05-16T01:04:09.506824Z","iopub.execute_input":"2024-05-16T01:04:09.507117Z","iopub.status.idle":"2024-05-16T01:04:09.617413Z","shell.execute_reply.started":"2024-05-16T01:04:09.507092Z","shell.execute_reply":"2024-05-16T01:04:09.616374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class NumpyDataset(Dataset):\n    def __init__(self, x, y):\n        \"\"\"\n        Initialize with NumPy arrays.\n        \"\"\"\n        assert x.shape[0] == y.shape[0], \"Features and labels must have the same number of samples\"\n        self.x = x\n        self.y = y\n\n    def __len__(self):\n        \"\"\"\n        Total number of samples.\n        \"\"\"\n        return self.x.shape[0]\n\n    def __getitem__(self, index):\n        \"\"\"\n        Generate one sample of data.\n        \"\"\"\n        # Convert the data to tensors when requested\n        return torch.from_numpy(self.x[index]).float().to(device), torch.from_numpy(self.y[index]).float().to(device)","metadata":{"execution":{"iopub.status.busy":"2024-05-16T01:04:09.618855Z","iopub.execute_input":"2024-05-16T01:04:09.619612Z","iopub.status.idle":"2024-05-16T01:04:09.626431Z","shell.execute_reply.started":"2024-05-16T01:04:09.619577Z","shell.execute_reply":"2024-05-16T01:04:09.6254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset = NumpyDataset(x_train, y_train)\n\ntrain_size = int(0.9 * len(dataset))\nval_size = len(dataset) - train_size\ntrain_dataset, val_dataset = torch.utils.data.random_split(dataset, [train_size, val_size])\n\ntrain_loader = DataLoader(train_dataset, batch_size=BATCH_SIZE, shuffle=True)\nval_loader = DataLoader(val_dataset, batch_size=BATCH_SIZE, shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2024-05-16T01:04:09.62759Z","iopub.execute_input":"2024-05-16T01:04:09.627909Z","iopub.status.idle":"2024-05-16T01:04:09.925065Z","shell.execute_reply.started":"2024-05-16T01:04:09.627872Z","shell.execute_reply":"2024-05-16T01:04:09.924214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class FFNN(nn.Module):\n    def __init__(self, input_size, hidden_sizes, output_size):\n        super(FFNN, self).__init__()\n        \n        layers = []\n        previous_size = input_size\n        for hidden_size in hidden_sizes:\n            layers.append(nn.Linear(previous_size, hidden_size))\n            layers.append(nn.LayerNorm(hidden_size))\n            layers.append(nn.LeakyReLU(inplace=True))\n            layers.append(nn.Dropout(p=0.1))\n            previous_size = hidden_size\n        \n        layers.append(nn.Linear(previous_size, output_size))\n        \n        self.layers = nn.Sequential(*layers)\n\n    def forward(self, x):\n        return self.layers(x)","metadata":{"execution":{"iopub.status.busy":"2024-05-16T01:04:09.92629Z","iopub.execute_input":"2024-05-16T01:04:09.926655Z","iopub.status.idle":"2024-05-16T01:04:09.934685Z","shell.execute_reply.started":"2024-05-16T01:04:09.926622Z","shell.execute_reply":"2024-05-16T01:04:09.933604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input_size = x_train.shape[1]\noutput_size = y_train.shape[1]\nhidden_size = input_size + output_size\nmodel = FFNN(input_size, [3*hidden_size, 2*hidden_size, 2*hidden_size, 2*hidden_size, 3*hidden_size], output_size).to(device)\ncriterion = nn.MSELoss()\noptimizer = optim.AdamW(model.parameters(), lr=0.001, weight_decay=0.01)\nscheduler = optim.lr_scheduler.ReduceLROnPlateau(optimizer, mode='min', factor=SCHEDULER_FACTOR, patience=SCHEDULER_PATIENCE)\n\nprint(\"Time after all preparations:\", format_time(time.time()-ts), flush=True)","metadata":{"execution":{"iopub.status.busy":"2024-05-16T01:04:09.935904Z","iopub.execute_input":"2024-05-16T01:04:09.936195Z","iopub.status.idle":"2024-05-16T01:04:11.096454Z","shell.execute_reply.started":"2024-05-16T01:04:09.936167Z","shell.execute_reply":"2024-05-16T01:04:11.095556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_val_loss = float('inf')\nbest_model_state = None\npatience_count = 0\nr2score = R2Score(num_outputs=len(TARGET_COLS)).to(device)\nfor epoch in range(EPOCHS):\n    print(\"\")\n    model.train()\n    total_loss = 0\n    steps = 0\n    for batch_idx, (inputs, labels) in enumerate(train_loader):\n        optimizer.zero_grad()\n        outputs = model(inputs)\n        loss = criterion(outputs, labels)\n        loss.backward()\n        optimizer.step()\n\n        total_loss += loss.item()\n        steps += 1\n\n        if (batch_idx + 1) % PRINT_FREQ == 0:\n            current_lr = optimizer.param_groups[0][\"lr\"]\n            elapsed_time = format_time(time.time() - ts)\n            print(f'  Epoch: {epoch+1}',\\\n                  f'  Batch: {batch_idx + 1}/{len(train_loader)}',\\\n                  f'  Train Loss: {total_loss / steps:.4f}',\\\n                  f'  LR: {current_lr:.1e}',\\\n                  f'  Time: {elapsed_time}', flush=True)\n            total_loss = 0\n            steps = 0\n    \n\n    model.eval()\n    val_loss = 0\n    y_true = torch.tensor([], device=device)\n    all_outputs = torch.tensor([], device=device)\n    with torch.no_grad():\n        for inputs, labels in val_loader:\n            outputs = model(inputs)\n            val_loss += criterion(outputs, labels).item()\n            y_true = torch.cat((y_true, labels), 0)\n            all_outputs = torch.cat((all_outputs, outputs), 0)\n    r2=0\n    r2_broken = []\n    r2_broken_names = []\n    for i in range(368):\n        r2_i = r2score(all_outputs[:, i], y_true[:, i])\n        if r2_i > 1e-6:\n            r2 += r2_i\n        else:\n            r2_broken.append(i)\n            r2_broken_names.append(FEAT_COLS[i])\n    r2 /= 368\n\n    avg_val_loss = val_loss / len(val_loader)\n    print(f'\\nEpoch: {epoch+1}  Val Loss: {avg_val_loss:.4f}  R2 score: {r2:.4f}')\n    print(f'{len(r2_broken)} targets were excluded during evaluation of R2 score.')\n    # print(r2_broken)\n    # print(r2_broken_names, flush=True)\n   \n    scheduler.step(avg_val_loss)\n\n    if avg_val_loss < best_val_loss:\n        best_val_loss = avg_val_loss\n        best_model_state = model.state_dict()\n        patience_count = 0\n        print(\"Validation loss decreased, saving new best model and resetting patience counter.\")\n    else:\n        patience_count += 1\n        print(f\"No improvement in validation loss for {patience_count} epochs.\")\n        \n    if patience_count >= PATIENCE:\n        print(\"Stopping early due to no improvement in validation loss.\")\n        break\n\ndel x_train, y_train\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-05-16T01:04:11.097808Z","iopub.execute_input":"2024-05-16T01:04:11.098589Z","iopub.status.idle":"2024-05-16T01:13:17.992041Z","shell.execute_reply.started":"2024-05-16T01:04:11.098552Z","shell.execute_reply":"2024-05-16T01:13:17.991141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.load_state_dict(best_model_state)\nmodel.eval()\n\ndf_test = pl.read_csv(DATA_PATH + \"leap-atmospheric-physics-ai-climsim/test.csv\")\n\nfor col in FEAT_COLS:\n    df_test = df_test.with_columns(pl.col(col).cast(pl.Float32))\n\nx_test = df_test.select(FEAT_COLS).to_numpy()\n\nx_test = (x_test - mx.reshape(1,-1)) / sx.reshape(1,-1)\n\npredt = np.zeros([x_test.shape[0], output_size], dtype=np.float32)\n\ni1 = 0\nfor i in range(10000):\n    i2 = np.minimum(i1 + BATCH_SIZE, x_test.shape[0])\n    if i1 == i2:  # Break the loop if range does not change\n        break\n\n    # Convert the current slice of xt to a PyTorch tensor\n    inputs = torch.from_numpy(x_test[i1:i2, :]).float().to(device)\n\n    # No need to track gradients for inference\n    with torch.no_grad():\n        outputs = model(inputs)  # Get model predictions\n        predt[i1:i2, :] = outputs.cpu().numpy()  # Store predictions in predt\n\n    i1 = i2  # Update i1 to the end of the current batch\n\n    if i2 >= x_test.shape[0]:\n        break\n\nfor i in range(sy.shape[0]):\n    if sy[i] < MIN_STD * 1.1:\n        predt[:,i] = 0\n\npredt = predt * sy.reshape(1,-1) + my.reshape(1,-1)\n\nss = pd.read_csv(DATA_PATH + \"leap-atmospheric-physics-ai-climsim/sample_submission.csv\")\nss.iloc[:,1:] = predt\n\ndel predt\ngc.collect()\n\nuse_cols = []\nfor i in range(27):\n    use_cols.append(f\"ptend_q0002_{i}\")\n\nss2 = pd.read_csv(DATA_PATH + \"leap-atmospheric-physics-ai-climsim/sample_submission.csv\")\ndf_test = df_test.to_pandas()\nfor col in use_cols:\n    ss[col] = -df_test[col.replace(\"ptend\", \"state\")]*ss2[col]/1200.\n\ntest_polars = pl.from_pandas(ss[[\"sample_id\"]+TARGET_COLS])\ntest_polars.write_csv(\"submission.csv\")\n\nprint(\"Total time:\", format_time(time.time()-ts))","metadata":{"execution":{"iopub.status.busy":"2024-05-16T01:17:44.311092Z","iopub.execute_input":"2024-05-16T01:17:44.312075Z","iopub.status.idle":"2024-05-16T01:21:34.355051Z","shell.execute_reply.started":"2024-05-16T01:17:44.312033Z","shell.execute_reply":"2024-05-16T01:21:34.354024Z"},"trusted":true},"execution_count":null,"outputs":[]}]}