{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Importing the necessary libraries","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-02T14:14:21.118324Z","iopub.execute_input":"2023-06-02T14:14:21.118618Z","iopub.status.idle":"2023-06-02T14:14:25.863846Z","shell.execute_reply.started":"2023-06-02T14:14:21.118585Z","shell.execute_reply":"2023-06-02T14:14:25.862886Z"}}},{"cell_type":"code","source":"import numpy as np\nimport torch \nimport torch.nn as nn\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport os ","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:12:11.504743Z","iopub.execute_input":"2023-06-02T15:12:11.505097Z","iopub.status.idle":"2023-06-02T15:12:14.488721Z","shell.execute_reply.started":"2023-06-02T15:12:11.505071Z","shell.execute_reply":"2023-06-02T15:12:14.487107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_dir = \"/kaggle/input/google-research-identify-contrails-reduce-global-warming\"","metadata":{"execution":{"iopub.status.busy":"2023-06-02T14:14:25.866876Z","iopub.execute_input":"2023-06-02T14:14:25.868234Z","iopub.status.idle":"2023-06-02T14:14:25.874435Z","shell.execute_reply.started":"2023-06-02T14:14:25.868198Z","shell.execute_reply":"2023-06-02T14:14:25.872274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device = \"cuda\" if torch.cuda.is_available() else \"cpu\"","metadata":{"execution":{"iopub.status.busy":"2023-06-02T14:14:25.877212Z","iopub.execute_input":"2023-06-02T14:14:25.877557Z","iopub.status.idle":"2023-06-02T14:14:25.955969Z","shell.execute_reply.started":"2023-06-02T14:14:25.877525Z","shell.execute_reply":"2023-06-02T14:14:25.954937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Getting The dataset Class","metadata":{}},{"cell_type":"code","source":"_T11_BOUNDS = (243, 303)\n_CLOUD_TOP_TDIFF_BOUNDS = (-4, 5)\n_TDIFF_BOUNDS = (-4, 2)\n\ndef normalize_range(data, bounds):\n    \"\"\"Maps data to the range [0, 1].\"\"\"\n    return (data - bounds[0]) / (bounds[1] - bounds[0])\n\ndef normalize_std(spec):\n    return (spec- np.mean(spec))/np.std(spec)\n\nclass Dataset(torch.utils.data.Dataset):\n    def __init__(self, data_path, mode='train'):\n        self.data_path = data_path\n        self.file_name = os.listdir(data_path)\n        self.mode = mode\n        \n\n    def __len__(self):\n        return len(self.file_name)\n\n    def __getitem__(self, i):\n        \n        band11 = np.load(os.path.join(self.data_path, self.file_name[i], 'band_11.npy'))\n        band14 = np.load(os.path.join(self.data_path, self.file_name[i], 'band_14.npy'))\n        band15 = np.load(os.path.join(self.data_path, self.file_name[i], 'band_15.npy'))\n        \n        r = normalize_range(band15 - band14, _TDIFF_BOUNDS)\n        g = normalize_range(band14 - band11, _CLOUD_TOP_TDIFF_BOUNDS)\n        b = normalize_range(band14, _T11_BOUNDS)\n        x = np.transpose(np.clip(np.stack([r, g, b], axis=2), 0, 1)[:,:,:,4],(2,0,1))\n        x = normalize_std(x)\n        \n        if self.mode == 'train':\n            y = np.load(os.path.join(self.data_path, self.file_name[i], 'human_pixel_masks.npy')).astype(np.float32).transpose(2,0,1)\n        elif self.mode == 'test':\n            y = self.file_name[i]\n        else:\n            y = None\n        \n        return x, y","metadata":{"execution":{"iopub.status.busy":"2023-06-02T14:14:25.960551Z","iopub.execute_input":"2023-06-02T14:14:25.960888Z","iopub.status.idle":"2023-06-02T14:14:25.975826Z","shell.execute_reply.started":"2023-06-02T14:14:25.960863Z","shell.execute_reply":"2023-06-02T14:14:25.974459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Normal Encoder to Rle encoder","metadata":{}},{"cell_type":"code","source":"def rle_encode(x, fg_val=1):\n    \"\"\"\n    Args:\n        x:  numpy array of shape (height, width), 1 - mask, 0 - background\n    Returns: run length encoding as list\n    \"\"\"\n\n    dots = np.where(\n        x.T.flatten() == fg_val)[0]  # .T sets Fortran order down-then-right\n    run_lengths = []\n    prev = -2\n    for b in dots:\n        if b > prev + 1:\n            run_lengths.extend((b + 1, 0))\n        run_lengths[-1] += 1\n        prev = b\n    return run_lengths\n\n\ndef list_to_string(x):\n    \"\"\"\n    Converts list to a string representation\n    Empty list returns '-'\n    \"\"\"\n    if x: # non-empty list\n        s = str(x).replace(\"[\", \"\").replace(\"]\", \"\").replace(\",\", \"\")\n    else:\n        s = '-'\n    return s\n\n\ndef rle_decode(mask_rle, shape=(256, 256)):\n    '''\n    mask_rle: run-length as string formatted (start length)\n              empty predictions need to be encoded with '-'\n    shape: (height, width) of array to return \n    Returns numpy array, 1 - mask, 0 - background\n    '''\n\n    img = np.zeros(shape[0]*shape[1], dtype=np.uint8)\n    if mask_rle != '-': \n        s = mask_rle.split()\n        starts, lengths = [np.asarray(x, dtype=int) for x in (s[0:][::2], s[1:][::2])]\n        starts -= 1\n        ends = starts + lengths\n        for lo, hi in zip(starts, ends):\n            img[lo:hi] = 1\n    return img.reshape(shape, order='F')  # Needed to align to RLE direction","metadata":{"execution":{"iopub.status.busy":"2023-06-02T14:14:25.977235Z","iopub.execute_input":"2023-06-02T14:14:25.977665Z","iopub.status.idle":"2023-06-02T14:14:25.989938Z","shell.execute_reply.started":"2023-06-02T14:14:25.977627Z","shell.execute_reply":"2023-06-02T14:14:25.988852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# A history of this model\n \n\n1. It's a deep_labv3 model taken straightaway from pytorch official Repo \nThis model produces 21 output and we have to select the first output as out answer and train the model on that pred[0,:,:]\n2. The model used here is trained on 40 epochs with each epoch taking 17 mins \n3. first 20 epoch 0.1 lr , next 20 epoch 0.001 epoch , loss goes down to 24 something\n\n4. The first model with 20 epochs can be foud in segment-anything-model, next 20 (fork of segment) ","metadata":{}},{"cell_type":"code","source":"# model  = torch.load(\"/kaggle/input/segment-anything-model/deep_labv3_final.pth\")\nmodel = torch.load(\"/kaggle/input/fork-of-segment-anything-model/deep_labv3_new.pth\")","metadata":{"execution":{"iopub.status.busy":"2023-06-02T14:14:25.99138Z","iopub.execute_input":"2023-06-02T14:14:25.991922Z","iopub.status.idle":"2023-06-02T14:14:32.18425Z","shell.execute_reply.started":"2023-06-02T14:14:25.991891Z","shell.execute_reply":"2023-06-02T14:14:32.183283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Defining the test dataset","metadata":{}},{"cell_type":"code","source":"batch_size = 1\nnum_workers = 2\n\ntest_dataset = Dataset('/kaggle/input/google-research-identify-contrails-reduce-global-warming/test', mode='test')\n\ntest_loader = torch.utils.data.DataLoader(test_dataset, batch_size=batch_size, shuffle=False, num_workers=num_workers, drop_last=True, pin_memory=True)","metadata":{"execution":{"iopub.status.busy":"2023-06-02T14:14:32.185682Z","iopub.execute_input":"2023-06-02T14:14:32.186022Z","iopub.status.idle":"2023-06-02T14:14:32.197972Z","shell.execute_reply.started":"2023-06-02T14:14:32.185992Z","shell.execute_reply":"2023-06-02T14:14:32.196932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_recs = os.listdir(os.path.join(base_dir,\"test\"))\nprint(test_recs)","metadata":{"execution":{"iopub.status.busy":"2023-06-02T14:14:32.200293Z","iopub.execute_input":"2023-06-02T14:14:32.200964Z","iopub.status.idle":"2023-06-02T14:14:32.207194Z","shell.execute_reply.started":"2023-06-02T14:14:32.200934Z","shell.execute_reply":"2023-06-02T14:14:32.206175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model.module(X) # this is used when Model parallalization was used ( when training with 2 GPUs simultaniously)","metadata":{"execution":{"iopub.status.busy":"2023-06-02T14:14:32.209182Z","iopub.execute_input":"2023-06-02T14:14:32.210253Z","iopub.status.idle":"2023-06-02T14:14:32.215447Z","shell.execute_reply.started":"2023-06-02T14:14:32.210222Z","shell.execute_reply":"2023-06-02T14:14:32.214491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Actual predictions Tweaking the threshold","metadata":{}},{"cell_type":"code","source":"# making the dataset for the imbalance \nhave_mask = 0 \nno_mask = 0\n\ntrain_dataset =os.listdir(\"/kaggle/input/google-research-identify-contrails-reduce-global-warming/train\")\n\nfor items in train_dataset:\n    mask = np.load(os.path.join(\"/kaggle/input/google-research-identify-contrails-reduce-global-warming/train\",items,\"human_pixel_masks.npy\"))\n    if (np.sum(mask)) == 0:\n        no_mask =no_mask+ 1\n    else:\n        have_mask=have_mask+1\nprint(have_mask)","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:16:14.582376Z","iopub.execute_input":"2023-06-02T15:16:14.582717Z","iopub.status.idle":"2023-06-02T15:18:47.5203Z","shell.execute_reply.started":"2023-06-02T15:16:14.582694Z","shell.execute_reply":"2023-06-02T15:18:47.518873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nsns.barplot(x = [\"Contrails\",\"No Contrails\"],y =[have_mask,no_mask] )\nplt.title(\"Imbalanced Training set\")","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:28:32.065209Z","iopub.execute_input":"2023-06-02T15:28:32.066492Z","iopub.status.idle":"2023-06-02T15:28:32.296948Z","shell.execute_reply.started":"2023-06-02T15:28:32.066437Z","shell.execute_reply":"2023-06-02T15:28:32.296016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# This graph is the Proof that -ve masks > +ve mask => biased data\n\n### Although the baseline for me was 0.34 when threshold was 0.5 , but the I read this article mentioned in the notebook  : https://www.kaggle.com/code/mnokno/getting-started-eda-model-train-submit?kernelSessionId=131699711\n\n''\nThere is a significant imbalance in the number of pixels between the negative class (non-contrail) and the positive class (contrail). In the training and validation datasets, the ratio of negative to positive class pixels is 85:1 and 164:1 respectively. This severe class imbalance can lead to biased models that prioritize the majority class, ultimately reducing the overall prediction quality.\n\nTo address this issue, we can employ two strategies: optimizing the confidence threshold during post-processing or incorporating class weights into the loss function during training.\n\nOptimizing confidence threshold during post-processing: After training the model, we can adjust the confidence threshold used to determine class predictions. By carefully selecting the threshold, we can increase the sensitivity to the positive class, thereby improving the detection of contrails. This approach allows us to fine-tune the model's predictions without retraining it.\n\nAdding class weights to the loss function during training: Another way to handle the class imbalance is by assigning appropriate weights to the different classes during model training. By assigning higher weights to the minority class (contrail), we can increase its influence on the loss function. This adjustment ensures that the model pays more attention to the positive class and helps mitigate the bias towards the majority class (non-contrail).\n\nThis notebook will utilize both strategies.''\n\n# due to this I changed the threshold from 0.5 to 0.3 which gives an improvement of almost 0.1","metadata":{}},{"cell_type":"code","source":"m = nn.Sigmoid()\nsubmission = pd.read_csv('/kaggle/input/google-research-identify-contrails-reduce-global-warming/sample_submission.csv', index_col='record_id')\nmodel.eval()\nwith torch.no_grad():\n    for X, rec in test_loader:\n        X = X.to(device)\n#         pred = m(model.module(X)['out']).cpu().detach().numpy().copy()[0,0,:,:] \n        pred = m(model(X)['out']).cpu().detach().numpy().copy()[0,0,:,:] \n        print(pred.shape)\n        mask = np.zeros((256, 256))\n        mask[pred<=0.3] = 0\n        mask[pred>0.3] = 1\n        \n        submission.loc[int(rec[0]), 'encoded_pixels'] = list_to_string(rle_encode(mask))\nsubmission.head()\n","metadata":{"execution":{"iopub.status.busy":"2023-06-02T14:14:32.218876Z","iopub.execute_input":"2023-06-02T14:14:32.219135Z","iopub.status.idle":"2023-06-02T14:14:39.48119Z","shell.execute_reply.started":"2023-06-02T14:14:32.219108Z","shell.execute_reply":"2023-06-02T14:14:39.479995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Final Submission","metadata":{}},{"cell_type":"code","source":"submission.to_csv('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-06-02T14:14:39.48349Z","iopub.execute_input":"2023-06-02T14:14:39.484319Z","iopub.status.idle":"2023-06-02T14:14:39.494318Z","shell.execute_reply.started":"2023-06-02T14:14:39.484264Z","shell.execute_reply":"2023-06-02T14:14:39.493254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}