{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Summary\n- Baseline written using Pytorch Lightning\n- resnest26d as encoder\n- Dataset is taken from: https://www.kaggle.com/datasets/shashwatraman/contrails-images-ash-color","metadata":{}},{"cell_type":"markdown","source":"### Training part","metadata":{}},{"cell_type":"code","source":"import sys\nsys.path.append(\"../input/pretrained-models-pytorch\")\nsys.path.append(\"../input/efficientnet-pytorch\")\nsys.path.append(\"/kaggle/input/smp-github/segmentation_models.pytorch-master\")\nsys.path.append(\"/kaggle/input/timm-pretrained-resnest/resnest/\")\nimport segmentation_models_pytorch as smp","metadata":{"execution":{"iopub.status.busy":"2023-06-12T13:44:48.314558Z","iopub.execute_input":"2023-06-12T13:44:48.314825Z","iopub.status.idle":"2023-06-12T13:44:55.957592Z","shell.execute_reply.started":"2023-06-12T13:44:48.314795Z","shell.execute_reply":"2023-06-12T13:44:55.956685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir -p /root/.cache/torch/hub/checkpoints/\n!cp /kaggle/input/timm-pretrained-resnest/resnest/gluon_resnest26-50eb607c.pth /root/.cache/torch/hub/checkpoints/gluon_resnest26-50eb607c.pth","metadata":{"execution":{"iopub.status.busy":"2023-06-12T13:45:01.288805Z","iopub.execute_input":"2023-06-12T13:45:01.289151Z","iopub.status.idle":"2023-06-12T13:45:02.232885Z","shell.execute_reply.started":"2023-06-12T13:45:01.289124Z","shell.execute_reply":"2023-06-12T13:45:02.231706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# config\nconfig = {\n    \"data_path\": \"/kaggle/input/contrails-images-ash-color\",\n    \"model\": {\n        \"encoder_name\": \"timm-resnest26d\",\n        \"loss_smooth\": 1.0,\n        \"optimizer_params\": {\"lr\": 0.003, \"weight_decay\": 0.0},\n        \"scheduler\": {\n            \"name\": \"CosineAnnealingLR\",\n            \"params\": {\n                \"CosineAnnealingLR\": {\"T_max\": 500, \"eta_min\": 1e-06, \"last_epoch\": -1},\n                \"ReduceLROnPlateau\": {\n                    \"factor\": 0.31622776601,\n                    \"mode\": \"min\",\n                    \"patience\": 4,\n                    \"verbose\": True,\n                },\n            },\n        },\n        \"seg_model\": \"Unet\",\n    },\n    \"output_dir\": \"models\",\n    \"progress_bar_refresh_rate\": 50,\n    \"seed\": 42,\n    \"train_bs\": 32,\n    \"trainer\": {\n        \"enable_progress_bar\": True,\n        \"max_epochs\": 30,\n        \"min_epochs\": 30,\n    },\n    \"valid_bs\": 64,\n    \"workers\": 2,\n}","metadata":{"execution":{"iopub.status.busy":"2023-06-12T13:40:01.699252Z","iopub.execute_input":"2023-06-12T13:40:01.700267Z","iopub.status.idle":"2023-06-12T13:40:01.70821Z","shell.execute_reply.started":"2023-06-12T13:40:01.700228Z","shell.execute_reply":"2023-06-12T13:40:01.706938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Dataset\n\nimport torch\nimport numpy as np\nimport torchvision.transforms as T\n\nclass ContrailsDataset(torch.utils.data.Dataset):\n    def __init__(self, df, train=True):\n\n        self.df = df\n        self.trn = train\n        self.normalize_image = T.Normalize((0.485, 0.456, 0.406), (0.229, 0.224, 0.225))\n\n    def __getitem__(self, index):\n        row = self.df.iloc[index]\n        con_path = row.path\n        con = np.load(str(con_path))\n\n        img = con[..., :-1]\n        label = con[..., -1]\n\n        label = torch.tensor(label)\n\n        img = torch.tensor(np.reshape(img, (256, 256, 3))).to(torch.float32).permute(2, 0, 1)\n\n        img = self.normalize_image(img)\n\n        return img.float(), label.float()\n\n    def __len__(self):\n        return len(self.df)","metadata":{"execution":{"iopub.status.busy":"2023-06-12T13:40:01.710075Z","iopub.execute_input":"2023-06-12T13:40:01.71046Z","iopub.status.idle":"2023-06-12T13:40:01.724054Z","shell.execute_reply.started":"2023-06-12T13:40:01.710432Z","shell.execute_reply":"2023-06-12T13:40:01.723179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Lightning module\n\nimport torch\nimport pytorch_lightning as pl\nimport segmentation_models_pytorch as smp\nfrom torch.optim.lr_scheduler import CosineAnnealingLR, ReduceLROnPlateau\nfrom torch.optim import AdamW\nimport torch.nn as nn\nfrom torchmetrics.functional import dice\n\nseg_models = {\n    \"Unet\": smp.Unet,\n    \"Unet++\": smp.UnetPlusPlus,\n    \"MAnet\": smp.MAnet,\n    \"Linknet\": smp.Linknet,\n    \"FPN\": smp.FPN,\n    \"PSPNet\": smp.PSPNet,\n    \"PAN\": smp.PAN,\n    \"DeepLabV3\": smp.DeepLabV3,\n    \"DeepLabV3+\": smp.DeepLabV3Plus,\n}\n\n\nclass LightningModule(pl.LightningModule):\n    def __init__(self, config):\n        super().__init__()\n        self.config = config\n        self.model = model = seg_models[config[\"seg_model\"]](\n            encoder_name=config[\"encoder_name\"],\n            encoder_weights=\"imagenet\",\n            in_channels=3,\n            classes=1,\n            activation=None,\n        )\n        self.loss_module = smp.losses.DiceLoss(mode=\"binary\", smooth=config[\"loss_smooth\"])\n        self.val_step_outputs = []\n        self.val_step_labels = []\n\n    def forward(self, batch):\n        imgs = batch\n        preds = self.model(imgs)\n        return preds\n\n    def configure_optimizers(self):\n        optimizer = AdamW(self.parameters(), **self.config[\"optimizer_params\"])\n\n        if self.config[\"scheduler\"][\"name\"] == \"CosineAnnealingLR\":\n            scheduler = CosineAnnealingLR(\n                optimizer,\n                **self.config[\"scheduler\"][\"params\"][self.config[\"scheduler\"][\"name\"]],\n            )\n            lr_scheduler_dict = {\"scheduler\": scheduler, \"interval\": \"step\"}\n            return {\"optimizer\": optimizer, \"lr_scheduler\": lr_scheduler_dict}\n        elif self.config[\"scheduler\"][\"name\"] == \"ReduceLROnPlateau\":\n            scheduler = ReduceLROnPlateau(\n                optimizer,\n                **self.config[\"scheduler\"][\"params\"][self.config[\"scheduler\"][\"name\"]],\n            )\n            lr_scheduler = {\"scheduler\": scheduler, \"monitor\": \"val_loss\"}\n            return {\"optimizer\": optimizer, \"lr_scheduler\": lr_scheduler}\n\n    def training_step(self, batch, batch_idx):\n        imgs, labels = batch\n        preds = self.model(imgs)\n        loss = self.loss_module(preds, labels)\n        self.log(\"train_loss\", loss, on_step=True, on_epoch=True, prog_bar=True, batch_size=16)\n\n        for param_group in self.trainer.optimizers[0].param_groups:\n            lr = param_group[\"lr\"]\n        self.log(\"lr\", lr, on_step=True, on_epoch=False, prog_bar=True)\n\n        return loss\n\n    def validation_step(self, batch, batch_idx):\n        imgs, labels = batch\n        preds = self.model(imgs)\n        loss = self.loss_module(preds, labels)\n        self.log(\"val_loss\", loss, on_step=False, on_epoch=True, prog_bar=True)\n        self.val_step_outputs.append(preds)\n        self.val_step_labels.append(labels)\n\n    def on_validation_epoch_end(self):\n        all_preds = torch.cat(self.val_step_outputs)\n        all_labels = torch.cat(self.val_step_labels)\n        self.val_step_outputs.clear()\n        self.val_step_labels.clear()\n        val_dice = dice(all_preds, all_labels.long())\n        self.log(\"val_dice\", val_dice, on_step=False, on_epoch=True, prog_bar=True)","metadata":{"execution":{"iopub.status.busy":"2023-06-12T13:40:01.726417Z","iopub.execute_input":"2023-06-12T13:40:01.726782Z","iopub.status.idle":"2023-06-12T13:40:11.865192Z","shell.execute_reply.started":"2023-06-12T13:40:01.726754Z","shell.execute_reply":"2023-06-12T13:40:11.864293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\n\nwarnings.filterwarnings(\"ignore\")\n\nimport os\nimport torch\nimport yaml\nimport pandas as pd\nimport pytorch_lightning as pl\nfrom pprint import pprint\nfrom pytorch_lightning.callbacks import ModelCheckpoint, EarlyStopping, TQDMProgressBar\nfrom torch.utils.data import DataLoader\n\ncontrails = os.path.join(config[\"data_path\"], \"contrails/\")\ntrain_path = os.path.join(config[\"data_path\"], \"train_df.csv\")\nvalid_path = os.path.join(config[\"data_path\"], \"valid_df.csv\")\n\ntrain_df = pd.read_csv(train_path)\nvalid_df = pd.read_csv(valid_path)\n\ntrain_df[\"path\"] = contrails + train_df[\"record_id\"].astype(str) + \".npy\"\nvalid_df[\"path\"] = contrails + valid_df[\"record_id\"].astype(str) + \".npy\"\n\ndataset_train = ContrailsDataset(train_df, train=True)\ndataset_validation = ContrailsDataset(valid_df, train=False)\n\ndata_loader_train = DataLoader(\n    dataset_train, batch_size=config[\"train_bs\"], shuffle=True, num_workers=config[\"workers\"]\n)\ndata_loader_validation = DataLoader(\n    dataset_validation, batch_size=config[\"valid_bs\"], shuffle=False, num_workers=config[\"workers\"]\n)\n\npl.seed_everything(config[\"seed\"])\n\nfilename = f\"model\"\n\ncheckpoint_callback = ModelCheckpoint(\n    monitor=\"val_dice\",\n    dirpath=config[\"output_dir\"],\n    mode=\"max\",\n    filename=filename,\n    save_top_k=1,\n    verbose=1,\n)\n\nprogress_bar_callback = TQDMProgressBar(refresh_rate=config[\"progress_bar_refresh_rate\"])\n\nearly_stop_callback = EarlyStopping(monitor=\"val_loss\", mode=\"min\", patience=5, verbose=1)\n\ntrainer = pl.Trainer(\n    callbacks=[checkpoint_callback, early_stop_callback, progress_bar_callback], logger=None, **config[\"trainer\"]\n)\n\nmodel = LightningModule(config[\"model\"])\n\ntrainer.fit(model, data_loader_train, data_loader_validation)","metadata":{"execution":{"iopub.status.busy":"2023-06-12T13:40:11.866646Z","iopub.execute_input":"2023-06-12T13:40:11.867331Z","iopub.status.idle":"2023-06-12T13:40:35.122489Z","shell.execute_reply.started":"2023-06-12T13:40:11.867299Z","shell.execute_reply":"2023-06-12T13:40:35.120976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Submission part","metadata":{}},{"cell_type":"code","source":"batch_size = 16\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\ndata = '/kaggle/input/google-research-identify-contrails-reduce-global-warming'\ndata_root = '/kaggle/input/google-research-identify-contrails-reduce-global-warming/test/'","metadata":{"execution":{"iopub.status.busy":"2023-06-12T13:36:24.019622Z","iopub.execute_input":"2023-06-12T13:36:24.020013Z","iopub.status.idle":"2023-06-12T13:36:24.026788Z","shell.execute_reply.started":"2023-06-12T13:36:24.01998Z","shell.execute_reply":"2023-06-12T13:36:24.025753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filenames = os.listdir(data_root)\ntest_df = pd.DataFrame(filenames, columns=['record_id'])\ntest_df['path'] = data_root + test_df['record_id'].astype(str)","metadata":{"execution":{"iopub.status.busy":"2023-06-12T13:36:26.141848Z","iopub.execute_input":"2023-06-12T13:36:26.142219Z","iopub.status.idle":"2023-06-12T13:36:26.154046Z","shell.execute_reply.started":"2023-06-12T13:36:26.142189Z","shell.execute_reply":"2023-06-12T13:36:26.153045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class ContrailsDataset(torch.utils.data.Dataset):\n    def __init__(self, df, train=True):\n        \n        self.df = df\n        self.trn = train\n        self.df_idx: pd.DataFrame = pd.DataFrame({'idx': os.listdir(f'/kaggle/input/google-research-identify-contrails-reduce-global-warming/test')})\n        self.normalize_image = T.Normalize((0.485, 0.456, 0.406), (0.229, 0.224, 0.225))\n    \n    def read_record(self, directory):\n        record_data = {}\n        for x in [\n            \"band_11\", \n            \"band_14\", \n            \"band_15\"\n        ]:\n\n            record_data[x] = np.load(os.path.join(directory, x + \".npy\"))\n\n        return record_data\n\n    def normalize_range(self, data, bounds):\n        \"\"\"Maps data to the range [0, 1].\"\"\"\n        return (data - bounds[0]) / (bounds[1] - bounds[0])\n    \n    def get_false_color(self, record_data):\n        _T11_BOUNDS = (243, 303)\n        _CLOUD_TOP_TDIFF_BOUNDS = (-4, 5)\n        _TDIFF_BOUNDS = (-4, 2)\n        \n        N_TIMES_BEFORE = 4\n\n        r = self.normalize_range(record_data[\"band_15\"] - record_data[\"band_14\"], _TDIFF_BOUNDS)\n        g = self.normalize_range(record_data[\"band_14\"] - record_data[\"band_11\"], _CLOUD_TOP_TDIFF_BOUNDS)\n        b = self.normalize_range(record_data[\"band_14\"], _T11_BOUNDS)\n        false_color = np.clip(np.stack([r, g, b], axis=2), 0, 1)\n        img = false_color[..., N_TIMES_BEFORE]\n\n        return img\n    \n    def __getitem__(self, index):\n        row = self.df.iloc[index]\n        con_path = row.path\n        data = self.read_record(con_path)    \n        \n        img = self.get_false_color(data)\n        \n        img = torch.tensor(np.reshape(img, (256, 256, 3))).to(torch.float32).permute(2, 0, 1)\n        \n        img = self.normalize_image(img)\n        \n        image_id = int(self.df_idx.iloc[index]['idx'])\n            \n        return img.float(), torch.tensor(image_id)\n    \n    def __len__(self):\n        return len(self.df)","metadata":{"execution":{"iopub.status.busy":"2023-06-12T13:36:28.399324Z","iopub.execute_input":"2023-06-12T13:36:28.400393Z","iopub.status.idle":"2023-06-12T13:36:28.416742Z","shell.execute_reply.started":"2023-06-12T13:36:28.400359Z","shell.execute_reply":"2023-06-12T13:36:28.415422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_ds = ContrailsDataset(\n        test_df,\n        train = False\n    )\n \ntest_dl = DataLoader(test_ds, batch_size=batch_size, num_workers = 1)","metadata":{"execution":{"iopub.status.busy":"2023-06-12T13:36:39.236413Z","iopub.execute_input":"2023-06-12T13:36:39.237105Z","iopub.status.idle":"2023-06-12T13:36:39.245881Z","shell.execute_reply.started":"2023-06-12T13:36:39.237073Z","shell.execute_reply":"2023-06-12T13:36:39.244866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class LightningModule(pl.LightningModule):\n\n    def __init__(self):\n        super().__init__()\n        self.model = smp.Unet(encoder_name=\"timm-resnest26d\",\n                              encoder_weights=None,\n                              in_channels=3,\n                              classes=1,\n                              activation=None,\n                              )\n\n    def forward(self, batch):\n        return self.model(batch)","metadata":{"execution":{"iopub.status.busy":"2023-06-12T13:36:41.482631Z","iopub.execute_input":"2023-06-12T13:36:41.483195Z","iopub.status.idle":"2023-06-12T13:36:41.490028Z","shell.execute_reply.started":"2023-06-12T13:36:41.483163Z","shell.execute_reply":"2023-06-12T13:36:41.48889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = LightningModule().load_from_checkpoint(\"/kaggle/working/models/model.ckpt\")\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nmodel.to(device)\nmodel.eval()\nmodel.zero_grad()","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-06-12T13:36:43.683338Z","iopub.execute_input":"2023-06-12T13:36:43.684036Z","iopub.status.idle":"2023-06-12T13:36:44.872486Z","shell.execute_reply.started":"2023-06-12T13:36:43.684002Z","shell.execute_reply":"2023-06-12T13:36:44.871493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def rle_encode(x, fg_val=1):\n    \"\"\"\n    Args:\n        x:  numpy array of shape (height, width), 1 - mask, 0 - background\n    Returns: run length encoding as list\n    \"\"\"\n\n    dots = np.where(\n        x.T.flatten() == fg_val)[0]  # .T sets Fortran order down-then-right\n    run_lengths = []\n    prev = -2\n    for b in dots:\n        if b > prev + 1:\n            run_lengths.extend((b + 1, 0))\n        run_lengths[-1] += 1\n        prev = b\n    return run_lengths\n\ndef list_to_string(x):\n    \"\"\"\n    Converts list to a string representation\n    Empty list returns '-'\n    \"\"\"\n    if x: # non-empty list\n        s = str(x).replace(\"[\", \"\").replace(\"]\", \"\").replace(\",\", \"\")\n    else:\n        s = '-'\n    return s","metadata":{"execution":{"iopub.status.busy":"2023-06-12T13:36:48.025542Z","iopub.execute_input":"2023-06-12T13:36:48.025904Z","iopub.status.idle":"2023-06-12T13:36:48.033998Z","shell.execute_reply.started":"2023-06-12T13:36:48.025875Z","shell.execute_reply":"2023-06-12T13:36:48.033096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv('/kaggle/input/google-research-identify-contrails-reduce-global-warming/sample_submission.csv', index_col='record_id')","metadata":{"execution":{"iopub.status.busy":"2023-06-12T13:36:55.098015Z","iopub.execute_input":"2023-06-12T13:36:55.09839Z","iopub.status.idle":"2023-06-12T13:36:55.117389Z","shell.execute_reply.started":"2023-06-12T13:36:55.098359Z","shell.execute_reply":"2023-06-12T13:36:55.116497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i, data in enumerate(test_dl):\n    images, image_id = data\n    \n    # Predict mask for this instance\n    images = images.to(device)\n    predicated_mask = model.forward(images[:, :, :, :])\n    predicated_mask = torch.sigmoid(predicated_mask).cpu().detach().numpy()\n    \n    # Apply threshold\n    predicated_mask_with_threshold = np.zeros((images.shape[0], 256, 256))\n    predicated_mask_with_threshold[predicated_mask[:, 0, :, :] < 0.5] = 0\n    predicated_mask_with_threshold[predicated_mask[:, 0, :, :] > 0.5] = 1\n    \n    for img_num in range(0, images.shape[0]):\n        current_mask = predicated_mask_with_threshold[img_num, :, :]\n        current_image_id = image_id[img_num].item()\n        \n        submission.loc[int(current_image_id), 'encoded_pixels'] = list_to_string(rle_encode(current_mask))","metadata":{"execution":{"iopub.status.busy":"2023-06-12T13:36:57.04915Z","iopub.execute_input":"2023-06-12T13:36:57.049563Z","iopub.status.idle":"2023-06-12T13:36:57.339749Z","shell.execute_reply.started":"2023-06-12T13:36:57.049535Z","shell.execute_reply":"2023-06-12T13:36:57.338423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2023-06-12T13:37:00.212082Z","iopub.execute_input":"2023-06-12T13:37:00.212723Z","iopub.status.idle":"2023-06-12T13:37:00.241416Z","shell.execute_reply.started":"2023-06-12T13:37:00.212677Z","shell.execute_reply":"2023-06-12T13:37:00.240505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-06-12T13:37:02.8172Z","iopub.execute_input":"2023-06-12T13:37:02.817591Z","iopub.status.idle":"2023-06-12T13:37:02.827653Z","shell.execute_reply.started":"2023-06-12T13:37:02.817563Z","shell.execute_reply":"2023-06-12T13:37:02.826502Z"},"trusted":true},"execution_count":null,"outputs":[]}]}