{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"},{"sourceId":7160316,"sourceType":"datasetVersion","datasetId":3892404},{"sourceId":7183944,"sourceType":"datasetVersion","datasetId":4152754},{"sourceId":7216256,"sourceType":"datasetVersion","datasetId":4141310},{"sourceId":7307506,"sourceType":"datasetVersion","datasetId":4186203},{"sourceId":154780340,"sourceType":"kernelVersion"}],"dockerImageVersionId":30588,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!python -m pip install --no-index --find-links=/kaggle/input/ubc-ocean-download segmentation-models-pytorch","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2024-01-01T22:20:36.363439Z","iopub.execute_input":"2024-01-01T22:20:36.363794Z","iopub.status.idle":"2024-01-01T22:20:54.658196Z","shell.execute_reply.started":"2024-01-01T22:20:36.363764Z","shell.execute_reply":"2024-01-01T22:20:54.657104Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#     public score   private score\n# v1  0.55            0.58\n# v3  0.51            0.52    no WSI and TMA outlier pred\n# v4  0.55            0.57    no WSI outlier pred\n# v5  0.55            0.59    Lower the outlier prediction threshold for TMA (from aux label <0.5 being considered as outlier to aux label <0.55 being considered as outlier)\n# v6  0.54            0.57    Add segmentation results into patch extraction\n# v7  0.56            0.61       Lower the outlier prediction threshold for WSI and TMA ","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nos.environ[\"OPENCV_IO_MAX_IMAGE_PIXELS\"] = str(pow(2,40))\n\nimport cv2\nimport time\n\n\nfrom joblib import Parallel, delayed\nimport glob\nimport random\n\n\n# Pytorch Imports\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nimport torch.nn.functional as F\nfrom torch.optim import lr_scheduler\nfrom torch.utils.data import Dataset, DataLoader\nfrom torch.cuda import amp\nimport torchvision\n\n# Albumentations for augmentations\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\n\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.metrics import f1_score,roc_auc_score\n\n\nimport timm\nfrom timm.models.efficientnet import *\nfrom timm.models.maxxvit import maxvit_tiny_tf_512\n\n# Utils\nimport joblib\nfrom tqdm.notebook import tqdm\nfrom collections import defaultdict\n\nimport gc\n\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:20:54.660365Z","iopub.execute_input":"2024-01-01T22:20:54.660741Z","iopub.status.idle":"2024-01-01T22:21:00.641873Z","shell.execute_reply.started":"2024-01-01T22:20:54.660704Z","shell.execute_reply":"2024-01-01T22:21:00.641098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from PIL import Image","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:21:00.643046Z","iopub.execute_input":"2024-01-01T22:21:00.643371Z","iopub.status.idle":"2024-01-01T22:21:00.647639Z","shell.execute_reply.started":"2024-01-01T22:21:00.64334Z","shell.execute_reply":"2024-01-01T22:21:00.646739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inference=\"test\" #test ,train","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:21:00.649703Z","iopub.execute_input":"2024-01-01T22:21:00.650007Z","iopub.status.idle":"2024-01-01T22:21:00.65672Z","shell.execute_reply.started":"2024-01-01T22:21:00.649974Z","shell.execute_reply":"2024-01-01T22:21:00.655766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"root_dir = \"/kaggle/input/UBC-OCEAN\"","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:21:00.657996Z","iopub.execute_input":"2024-01-01T22:21:00.658295Z","iopub.status.idle":"2024-01-01T22:21:00.665147Z","shell.execute_reply.started":"2024-01-01T22:21:00.658265Z","shell.execute_reply":"2024-01-01T22:21:00.664316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if inference == \"test\":\n    df = pd.read_csv(\"/kaggle/input/UBC-OCEAN/test.csv\")\nelse:\n    df = pd.read_csv(\"/kaggle/input/UBC-OCEAN/train.csv\")\n    df = df[[\"image_id\",\"image_width\",\"image_height\"]][:10]","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:21:00.666152Z","iopub.execute_input":"2024-01-01T22:21:00.666406Z","iopub.status.idle":"2024-01-01T22:21:00.685041Z","shell.execute_reply.started":"2024-01-01T22:21:00.666383Z","shell.execute_reply":"2024-01-01T22:21:00.684203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.makedirs(\"/kaggle/tmp\",exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:21:00.686113Z","iopub.execute_input":"2024-01-01T22:21:00.686368Z","iopub.status.idle":"2024-01-01T22:21:00.690412Z","shell.execute_reply.started":"2024-01-01T22:21:00.686346Z","shell.execute_reply":"2024-01-01T22:21:00.689593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['tma']=1\nfor image_id in df[\"image_id\"].values:\n    if os.path.exists(f\"{root_dir}/{inference}_thumbnails/{image_id}_thumbnail.png\"):\n        df.loc[df[\"image_id\"]==image_id,\"tma\"]=0\n    if (df.loc[df[\"image_id\"]==image_id,\"image_width\"].values[0]<6000) & (df.loc[df[\"image_id\"]==image_id,\"image_width\"].values[0]<6000):\n        df.loc[df[\"image_id\"]==image_id,\"tma\"]=1\n    if (df.loc[df[\"image_id\"]==image_id,\"tma\"].values[0]==1)&(os.path.exists(f\"{root_dir}/{inference}_thumbnails/{image_id}_thumbnail.png\")):\n        img=cv2.imread(f\"/kaggle/input/UBC-OCEAN/{inference}_thumbnails/{image_id}_thumbnail.png\")\n        if (np.sum(np.sum(img, axis=2) == 0)/(img.shape[0]*img.shape[1]))>0.05:\n            df.loc[df[\"image_id\"]==image_id,\"tma\"]=0\n#df['tma'] = ((df[\"image_width\"]<6000)&(df[\"image_height\"]<6000)).astype(int)","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:21:00.691604Z","iopub.execute_input":"2024-01-01T22:21:00.691865Z","iopub.status.idle":"2024-01-01T22:21:00.709145Z","shell.execute_reply.started":"2024-01-01T22:21:00.691842Z","shell.execute_reply":"2024-01-01T22:21:00.708257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_wsi=df[df[\"tma\"]==0].reset_index(drop=True)\ndf_tma=df[df[\"tma\"]==1].reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:21:00.710176Z","iopub.execute_input":"2024-01-01T22:21:00.710464Z","iopub.status.idle":"2024-01-01T22:21:00.71907Z","shell.execute_reply.started":"2024-01-01T22:21:00.710416Z","shell.execute_reply":"2024-01-01T22:21:00.718111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open(\"/kaggle/input/ubc-ocean-weights-fate/le.pkl\", \"rb\") as fp:\n    le = joblib.load(fp)\ndict(zip(le.inverse_transform(list(range(len(le.classes_)))),list(range(len(le.classes_)))))","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:21:00.723934Z","iopub.execute_input":"2024-01-01T22:21:00.724223Z","iopub.status.idle":"2024-01-01T22:21:00.741648Z","shell.execute_reply.started":"2024-01-01T22:21:00.724199Z","shell.execute_reply":"2024-01-01T22:21:00.740817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# seg","metadata":{}},{"cell_type":"code","source":"import segmentation_models_pytorch as smp","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:21:00.742683Z","iopub.execute_input":"2024-01-01T22:21:00.743019Z","iopub.status.idle":"2024-01-01T22:21:01.872061Z","shell.execute_reply.started":"2024-01-01T22:21:00.742988Z","shell.execute_reply":"2024-01-01T22:21:01.87108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class TestDataset(Dataset):\n    def __init__(self, df, size):\n        self.df = df\n        self.size = size\n\n        # self.df = self.df.sample(frac=1).reset_index(drop=True)       \n\n        self.transform = torchvision.transforms.Compose([\n            torchvision.transforms.Resize((size, size)),\n            torchvision.transforms.ToTensor(),\n        ])\n\n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, idx):\n        image_id = self.df['image_id'][idx]\n        image_path=f\"/kaggle/input/UBC-OCEAN/{inference}_thumbnails/{image_id}_thumbnail.png\"\n\n#         image = cv2.imread(image_path)\n#         image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        image = Image.open(image_path)\n        image = self.transform(image)\n# \n#         mask = mask.permute(2, 0, 1)\n        return {'image': image, 'image_id': image_id}","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:21:01.873253Z","iopub.execute_input":"2024-01-01T22:21:01.873607Z","iopub.status.idle":"2024-01-01T22:21:01.880649Z","shell.execute_reply.started":"2024-01-01T22:21:01.873575Z","shell.execute_reply":"2024-01-01T22:21:01.879808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch.nn as nn\nclass testSegModel(nn.Module):\n\n    def __init__(self, arch, encoder_name, in_channels, out_classes, **kwargs):\n        super().__init__()\n        self.model = smp.create_model(\n            arch, encoder_name=encoder_name, in_channels=in_channels, classes=out_classes, encoder_weights=None,**kwargs\n        )\n        params = smp.encoders.get_preprocessing_params(encoder_name)\n        self.register_buffer(\"std\", torch.tensor(params[\"std\"]).view(1, 3, 1, 1))\n        self.register_buffer(\"mean\", torch.tensor(params[\"mean\"]).view(1, 3, 1, 1))\n\n#         self.loss_fn = smp.losses.DiceLoss(smp.losses.MULTICLASS_MODE, from_logits=True)\n        \n#         self.train_outputs = []\n#         self.valid_outputs = []\n\n    def forward(self, image):\n        image = (image - self.mean) / self.std\n        mask = self.model(image)\n        return mask","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:21:01.881991Z","iopub.execute_input":"2024-01-01T22:21:01.882326Z","iopub.status.idle":"2024-01-01T22:21:01.891678Z","shell.execute_reply.started":"2024-01-01T22:21:01.882289Z","shell.execute_reply":"2024-01-01T22:21:01.890805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = testSegModel(\"unetplusplus\", \"efficientnet-b3\", in_channels=3, out_classes=4)\nmodel.load_state_dict(torch.load(\"/kaggle/input/ubc-ocean-seg-model/unetpp_b3_epoch99.ckpt\", map_location=lambda storage, loc: storage)[\"state_dict\"])\nmodel.cuda()\nmodel.eval()","metadata":{"_kg_hide-input":false,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-01-01T22:21:01.892724Z","iopub.execute_input":"2024-01-01T22:21:01.893018Z","iopub.status.idle":"2024-01-01T22:21:06.429863Z","shell.execute_reply.started":"2024-01-01T22:21:01.892989Z","shell.execute_reply":"2024-01-01T22:21:06.428935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset = TestDataset(df_wsi, 512)\n\ntest_loader = DataLoader(test_dataset, batch_size=8, shuffle=False, num_workers= 3)","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:21:06.431268Z","iopub.execute_input":"2024-01-01T22:21:06.43165Z","iopub.status.idle":"2024-01-01T22:21:06.436894Z","shell.execute_reply.started":"2024-01-01T22:21:06.431615Z","shell.execute_reply":"2024-01-01T22:21:06.436063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_mask_512_folder=\"/kaggle/pred_mask_512\"\nos.makedirs(pred_mask_512_folder,exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:21:06.438174Z","iopub.execute_input":"2024-01-01T22:21:06.438508Z","iopub.status.idle":"2024-01-01T22:21:06.445902Z","shell.execute_reply.started":"2024-01-01T22:21:06.438476Z","shell.execute_reply":"2024-01-01T22:21:06.445015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bar = tqdm(enumerate(test_loader), total=len(test_loader))\nfor step, batch in bar:\n    with torch.no_grad():\n        model.eval()\n        \n        images = batch['image'].to(\"cuda:0\", dtype=torch.float)\n        ids = batch['image_id'].numpy()\n        logits = model(images)\n\n    #pr_masks = logits.softmax(dim = 1)\n    pr_masks = torch.argmax(logits.softmax(dim = 1), dim = 1)\n    for image_id,pr_mask in  zip(ids,pr_masks):\n        mask_pred=pr_mask.cpu().numpy()\n        mask_pred=(mask_pred==1).astype(int)\n#         print(image_id)\n        \n#         plt.imshow(mask_pred)\n#         plt.show()\n        np.save(f\"{pred_mask_512_folder}/{image_id}.npy\",mask_pred)","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:21:06.447015Z","iopub.execute_input":"2024-01-01T22:21:06.447297Z","iopub.status.idle":"2024-01-01T22:21:10.944221Z","shell.execute_reply.started":"2024-01-01T22:21:06.447264Z","shell.execute_reply":"2024-01-01T22:21:10.943303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del model,bar\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:21:10.94573Z","iopub.execute_input":"2024-01-01T22:21:10.946206Z","iopub.status.idle":"2024-01-01T22:21:11.140528Z","shell.execute_reply.started":"2024-01-01T22:21:10.946166Z","shell.execute_reply":"2024-01-01T22:21:11.139538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# glob.glob(f\"{pred_mask_512_folder}/*\")\n# plt.imshow(np.load(glob.glob(f\"{pred_mask_512_folder}/*\")[0]))","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:21:11.141615Z","iopub.execute_input":"2024-01-01T22:21:11.141863Z","iopub.status.idle":"2024-01-01T22:21:11.152361Z","shell.execute_reply.started":"2024-01-01T22:21:11.141842Z","shell.execute_reply":"2024-01-01T22:21:11.15161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# WSI tile model","metadata":{}},{"cell_type":"code","source":"df_wsi_=df_wsi.copy()\ndf_wsi_[\"area\"]=df_wsi_[\"image_height\"]*df_wsi_[\"image_width\"]","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:21:11.153604Z","iopub.execute_input":"2024-01-01T22:21:11.153978Z","iopub.status.idle":"2024-01-01T22:21:11.162692Z","shell.execute_reply.started":"2024-01-01T22:21:11.153929Z","shell.execute_reply":"2024-01-01T22:21:11.161909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df_wsi_=df_wsi_[df_wsi_[\"area\"]<3e9].reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:21:11.163836Z","iopub.execute_input":"2024-01-01T22:21:11.164191Z","iopub.status.idle":"2024-01-01T22:21:11.170982Z","shell.execute_reply.started":"2024-01-01T22:21:11.164161Z","shell.execute_reply":"2024-01-01T22:21:11.170197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def resize_image_and_make_tile(name,out_path,scale):\n    path=f\"/kaggle/input/UBC-OCEAN/{inference}_images/{name}.png\"\n    p_mask=f\"{pred_mask_512_folder}/{name}.npy\"\n    image=cv2.imread(path)\n    image=cv2.resize(image,(0,0),fx=scale,fy=scale,interpolation=cv2.INTER_AREA)\n    mask=np.load(p_mask)\n    #cv2.imwrite(f\"{save_folder}/{name}.png\",image)\n    os.makedirs(f\"{out_path}/{name}\",exist_ok=True)\n    count=0\n    #############\n#     #seg mask\n#     if np.sum(mask)>50:\n#         mask=cv2.resize(mask,(image.shape[1], image.shape[0]),interpolation=cv2.INTER_NEAREST)\n#         idxs=[(y,x) for y in range(0,image.shape[0]//512) for x in range(0,image.shape[1]//512)]\n\n#         for k, (y, x) in enumerate(idxs):\n#             tile=image[y*512:(y+1)*512,x*512:(x+1)*512,:]\n#             tile_mask=mask[y*512:(y+1)*512,x*512:(x+1)*512]\n# #             b_count = np.sum(np.sum(tile, axis=2) == 0)\n# #             if b_count>tile.shape[0]*tile.shape[1]*0.25: #0.2?\n# #                 continue\n                \n#             mask_count=np.sum(tile_mask)\n\n#             if (mask_count)>0:\n#                 cv2.imwrite(f\"{out_path}/{name}/{x}_{y}.png\",tile)\n#                 count+=1\n    ##########\n    \n    if (count<20):\n        idxs=[(y,x) for y in range(0,image.shape[0]//512) for x in range(0,image.shape[1]//512)]\n        random.shuffle(idxs)\n        for k, (y, x) in enumerate(idxs):\n            tile=image[y*512:(y+1)*512,x*512:(x+1)*512,:]\n    #             b_count = np.sum(np.sum(tile, axis=2) == 0)\n    #             if b_count>tile.shape[0]*tile.shape[1]*0.25: #0.2?\n    #                 continue\n            #bg_count=np.sum((tile.max(axis=2)-tile.min(axis=2))<20)\n            bg_count=np.sum(np.ptp(tile,axis=2)<20)\n            if ((bg_count/(512*512))<=0.5):\n                cv2.imwrite(f\"{out_path}/{name}/{x}_{y}.png\",tile)\n                count+=1\n\n            if count>=60: #60\n                break\n\n\n        if count<20:\n            idxs=[(y,x) for y in range(0,image.shape[0]//512) for x in range(0,image.shape[1]//512)]\n            random.shuffle(idxs)\n            for k, (y, x) in enumerate(idxs):\n                tile=image[y*512:(y+1)*512,x*512:(x+1)*512,:]\n    #             b_count = np.sum(np.sum(tile, axis=2) == 0)\n    #             if b_count>tile.shape[0]*tile.shape[1]*0.25: #0.2?\n    #                 continue\n                #bg_count=np.sum((tile.max(axis=2)-tile.min(axis=2))<20)\n                bg_count=np.sum(np.ptp(tile,axis=2)<20)\n\n                if ((bg_count/(512*512))<=0.65)&((bg_count/(512*512))>0.5):\n                    cv2.imwrite(f\"{out_path}/{name}/{x}_{y}.png\",tile)\n                    count+=1\n\n                if count>=40:\n                    break\n        if count<10:\n            idxs=[(y,x) for y in range(0,image.shape[0]//512) for x in range(0,image.shape[1]//512)]\n            random.shuffle(idxs)\n            for k, (y, x) in enumerate(idxs):\n                tile=image[y*512:(y+1)*512,x*512:(x+1)*512,:]\n    #             b_count = np.sum(np.sum(tile, axis=2) == 0)\n    #             if b_count>tile.shape[0]*tile.shape[1]*0.25: #0.2?\n    #                 continue\n                #bg_count=np.sum((tile.max(axis=2)-tile.min(axis=2))<20)\n                bg_count=np.sum(np.ptp(tile,axis=2)<20)\n\n                if ((bg_count/(512*512))<=0.75)&((bg_count/(512*512))>0.65):\n                    cv2.imwrite(f\"{out_path}/{name}/{x}_{y}.png\",tile)\n                    count+=1\n\n                if count>=10:\n                    break","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:21:11.18112Z","iopub.execute_input":"2024-01-01T22:21:11.181634Z","iopub.status.idle":"2024-01-01T22:21:11.198831Z","shell.execute_reply.started":"2024-01-01T22:21:11.181608Z","shell.execute_reply":"2024-01-01T22:21:11.197878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfor name in tqdm(df_wsi_[\"image_id\"].values):\n    resize_image_and_make_tile(name,\"/kaggle/tmp\",0.33)","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:21:11.200196Z","iopub.execute_input":"2024-01-01T22:21:11.20049Z","iopub.status.idle":"2024-01-01T22:21:30.526859Z","shell.execute_reply.started":"2024-01-01T22:21:11.200466Z","shell.execute_reply":"2024-01-01T22:21:30.525943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(glob.glob(\"/kaggle/tmp/*/*\"))","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:21:30.527931Z","iopub.execute_input":"2024-01-01T22:21:30.528226Z","iopub.status.idle":"2024-01-01T22:21:30.535064Z","shell.execute_reply.started":"2024-01-01T22:21:30.528201Z","shell.execute_reply":"2024-01-01T22:21:30.534181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class UBCDataset_valid_batch(Dataset):\n    def __init__(self, df, transforms=None):\n        self.df = df\n        self.image_path = df['path'].values\n        \n        self.transforms = transforms\n        \n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, index):\n        \n        img_path=self.image_path[index] \n        \n        img = cv2.imread(img_path)\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        \n        \n        ###\n        img = img.astype(np.float32)/255\n        ######\n        \n        if self.transforms is not None:\n            img = self.transforms(image=img)[\"image\"]\n            \n        return {\n            'image': img,\n            \n        }","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:21:30.536468Z","iopub.execute_input":"2024-01-01T22:21:30.536724Z","iopub.status.idle":"2024-01-01T22:21:30.544781Z","shell.execute_reply.started":"2024-01-01T22:21:30.536702Z","shell.execute_reply":"2024-01-01T22:21:30.543771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_transforms_tile = {\n\n    \n    \"valid\": A.Compose([\n        A.Resize(512, 512),\n\n        #A.Normalize(),\n        ToTensorV2()], p=1.)\n}","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:21:30.545841Z","iopub.execute_input":"2024-01-01T22:21:30.546193Z","iopub.status.idle":"2024-01-01T22:21:30.556061Z","shell.execute_reply.started":"2024-01-01T22:21:30.546159Z","shell.execute_reply":"2024-01-01T22:21:30.555314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Net_v1(nn.Module):\n    def __init__(self):\n        super(Net_v1, self).__init__()\n        self.encoder = tf_efficientnet_b4_ns(pretrained=False)\n        self.logit = nn.Linear(1792,5) #1280#1792 (5)\n        self.aux = nn.Linear(1792,1)\n        self.aux2 = nn.Linear(1792,1)\n    def forward(self, image):\n        e = self.encoder\n        \n        x = e.forward_features(image)\n        x = F.adaptive_avg_pool2d(x,1)\n        x = torch.flatten(x,1,3)\n       \n        logit = self.logit(x)\n        aux = self.aux(x)\n        aux2 = self.aux2(x)\n        return logit,aux,aux2\n\nclass Net_v2(nn.Module):\n    def __init__(self):\n        super(Net_v2, self).__init__()\n        self.encoder = tf_efficientnetv2_s_in21ft1k(pretrained=False)\n        self.logit = nn.Linear(1280,5) #1280#1792 (5)\n        self.aux = nn.Linear(1280,1)\n        self.aux2 = nn.Linear(1280,1)\n    def forward(self, image):\n        e = self.encoder\n        \n        x = e.forward_features(image)\n        x = F.adaptive_avg_pool2d(x,1)\n        x = torch.flatten(x,1,3)\n       \n        logit = self.logit(x)\n        aux = self.aux(x)\n        aux2 = self.aux2(x)\n        return logit,aux,aux2\n    \nclass Net_v3(nn.Module):\n    def __init__(self):\n        super(Net_v3, self).__init__()\n        self.encoder = maxvit_tiny_tf_512(pretrained=False)\n        self.logit = nn.Linear(512,5) #1280#1792 (5)\n        self.aux = nn.Linear(512,1)\n        self.aux2 = nn.Linear(512,1)\n    def forward(self, image):\n        e = self.encoder\n        \n        x = e.forward_features(image)\n        x = F.adaptive_avg_pool2d(x,1)\n        x = torch.flatten(x,1,3)\n       \n        logit = self.logit(x)\n        aux = self.aux(x)\n        aux2 = self.aux2(x)\n        return logit,aux,aux2","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:21:30.564844Z","iopub.execute_input":"2024-01-01T22:21:30.56554Z","iopub.status.idle":"2024-01-01T22:21:30.578045Z","shell.execute_reply.started":"2024-01-01T22:21:30.565515Z","shell.execute_reply":"2024-01-01T22:21:30.577183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models1=[]\nweights=[\n        \"/kaggle/input/ubc-ocean-weights-bce-v1/tile_0_step3_51_effnetb4_fold_0.bin\",\n        \"/kaggle/input/ubc-ocean-weights-bce-v1/tile_0_step3_51_effnetb4_fold_1.bin\",\n        \"/kaggle/input/ubc-ocean-weights-bce-v1/tile_0_step3_51_effnetb4_fold_2.bin\",\n        \"/kaggle/input/ubc-ocean-weights-bce-v1/tile_0_step3_51_effnetb4_fold_3.bin\",\n        \"/kaggle/input/ubc-ocean-weights-bce-v1/tile_0_step3_51_effnetb4_fold_4.bin\",\n        ]\nfor i in range(len(weights)):\n    model = Net_v1()\n    checkpoint = weights[i]\n    f = torch.load(checkpoint, map_location=lambda storage, loc: storage)\n    model.load_state_dict(f, strict=False) #False\n    model.cuda()\n    model.eval()\n    models1.append(model)\n\n\nmodels2=[]\nweights=[\"/kaggle/input/ubc-ocean-weights-bce-v1/tile_0_step3_52_effnetv2s_fold_0.bin\",\n        \"/kaggle/input/ubc-ocean-weights-bce-v1/tile_0_step3_52_effnetv2s_fold_1.bin\",\n        \"/kaggle/input/ubc-ocean-weights-bce-v1/tile_0_step3_52_effnetv2s_fold_2.bin\",\n        \"/kaggle/input/ubc-ocean-weights-bce-v1/tile_0_step3_52_effnetv2s_fold_3.bin\",\n        \"/kaggle/input/ubc-ocean-weights-bce-v1/tile_0_step3_52_effnetv2s_fold_4.bin\",\n        ]\nfor i in range(len(weights)):\n    model = Net_v2()\n    checkpoint = weights[i]\n    f = torch.load(checkpoint, map_location=lambda storage, loc: storage)\n    model.load_state_dict(f, strict=True) #False\n    model.cuda()\n    model.eval()\n    models2.append(model)\n    \n    \nmodels3=[]\nweights=[\"/kaggle/input/ubc-ocean-weights-bce-v1/tile_1_step3_50_maxvit_fold_0.bin\",\n        \"/kaggle/input/ubc-ocean-weights-bce-v1/tile_1_step3_50_maxvit_fold_1.bin\",\n        \"/kaggle/input/ubc-ocean-weights-bce-v1/tile_1_step3_50_maxvit_fold_2.bin\",\n        \"/kaggle/input/ubc-ocean-weights-bce-v1/tile_1_step3_50_maxvit_fold_3.bin\",\n        \"/kaggle/input/ubc-ocean-weights-bce-v1/tile_1_step3_50_maxvit_fold_4.bin\",\n        ]\nfor i in range(len(weights)):\n    model = Net_v3()\n    checkpoint = weights[i]\n    f = torch.load(checkpoint, map_location=lambda storage, loc: storage)\n    model.load_state_dict(f, strict=True) #False\n    model.cuda()\n    model.eval()\n    models3.append(model)","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:21:30.579287Z","iopub.execute_input":"2024-01-01T22:21:30.579562Z","iopub.status.idle":"2024-01-01T22:21:51.193114Z","shell.execute_reply.started":"2024-01-01T22:21:30.579539Z","shell.execute_reply":"2024-01-01T22:21:51.192079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tile_df=pd.DataFrame({\"path\":glob.glob(\"/kaggle/tmp/*/*.png\")})","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:21:51.194609Z","iopub.execute_input":"2024-01-01T22:21:51.19527Z","iopub.status.idle":"2024-01-01T22:21:51.20037Z","shell.execute_reply.started":"2024-01-01T22:21:51.195236Z","shell.execute_reply":"2024-01-01T22:21:51.199395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tile_df[\"image_id\"]=tile_df[\"path\"].apply(lambda x:int(x.split(\"/\")[-2]))","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:21:51.201621Z","iopub.execute_input":"2024-01-01T22:21:51.201914Z","iopub.status.idle":"2024-01-01T22:21:51.212512Z","shell.execute_reply.started":"2024-01-01T22:21:51.201872Z","shell.execute_reply":"2024-01-01T22:21:51.211603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_dataset = UBCDataset_valid_batch(tile_df,transforms=data_transforms_tile[\"valid\"])\n\nvalid_loader = DataLoader(valid_dataset, batch_size=16, \n                          num_workers=2, shuffle=False, pin_memory=True)","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:21:51.21365Z","iopub.execute_input":"2024-01-01T22:21:51.214021Z","iopub.status.idle":"2024-01-01T22:21:51.221423Z","shell.execute_reply.started":"2024-01-01T22:21:51.213993Z","shell.execute_reply":"2024-01-01T22:21:51.220477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n    \n    \nnum_net1=len(models1)\npred1=[]\ntot_aux1=[]\n\nnum_net2=len(models2)\npred2=[]\ntot_aux2=[]\n\nnum_net3=len(models3)\npred3=[]\ntot_aux3=[]\n# model.cuda()\n# model.eval()\nbar = tqdm(enumerate(valid_loader), total=len(valid_loader))\nfor step, data in bar:\n\n    images = data['image'].to(\"cuda:0\", dtype=torch.float)\n\n    p=0\n    ax=0\n    count=0\n    with torch.no_grad():\n        for i in range(num_net1):\n            output,aux,_ = models1[i](images)\n            tmp_pred=F.sigmoid(output)\n            tmp_aux=F.sigmoid(aux)\n            p+=tmp_pred\n            ax+=tmp_aux\n            count+=1\n            \n            \n    p=p/count\n    ax=ax/count\n    pred1.append(p.cpu().numpy())\n    tot_aux1.append(ax.cpu().numpy())\n    \n#     del images\n#     gc.collect()\n    \n    \n    p=0\n    ax=0\n    count=0\n    with torch.no_grad():\n        for i in range(num_net2):\n            output,aux,_ = models2[i](images)\n            tmp_pred=F.sigmoid(output)\n            tmp_aux=F.sigmoid(aux)\n            p+=tmp_pred\n            ax+=tmp_aux\n            count+=1\n            \n            \n    p=p/count\n    ax=ax/count\n    pred2.append(p.cpu().numpy())\n    tot_aux2.append(ax.cpu().numpy())\n    \n\n    \n    \n    p=0\n    ax=0\n    count=0\n    with torch.no_grad():\n        for i in range(num_net3):\n            output,aux,_ = models3[i](images)\n            tmp_pred=F.sigmoid(output)\n            tmp_aux=F.sigmoid(aux)\n            p+=tmp_pred\n            ax+=tmp_aux\n            count+=1\n            \n            \n    p=p/count\n    ax=ax/count\n    pred3.append(p.cpu().numpy())\n    tot_aux3.append(ax.cpu().numpy())\n    \n    del images\n    gc.collect()\n    ","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:21:51.222524Z","iopub.execute_input":"2024-01-01T22:21:51.222796Z","iopub.status.idle":"2024-01-01T22:22:03.607996Z","shell.execute_reply.started":"2024-01-01T22:21:51.222769Z","shell.execute_reply":"2024-01-01T22:22:03.606958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# del models\ndel valid_loader\ntorch.cuda.empty_cache()\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:03.609359Z","iopub.execute_input":"2024-01-01T22:22:03.609771Z","iopub.status.idle":"2024-01-01T22:22:03.892245Z","shell.execute_reply.started":"2024-01-01T22:22:03.609736Z","shell.execute_reply":"2024-01-01T22:22:03.8912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tot_pred_y1=np.concatenate(pred1)\ntot_pred_y2=np.concatenate(pred2)\ntot_pred_y3=np.concatenate(pred3)\n#########\ntot_aux1=np.concatenate(tot_aux1)\ntot_aux2=np.concatenate(tot_aux2)\ntot_aux3=np.concatenate(tot_aux3)\ntot_aux1=tot_aux1.flatten()\ntot_aux2=tot_aux2.flatten()\ntot_aux3=tot_aux3.flatten()\n#tile_df[\"aux\"]=(tot_aux>0.5).astype(int)\n#########\n# tile_df[\"pred_0\"]=(tot_pred_y1[:,0]+tot_pred_y2[:,0])/2\n# tile_df[\"pred_1\"]=(tot_pred_y1[:,1]+tot_pred_y2[:,1])/2\n# tile_df[\"pred_2\"]=(tot_pred_y1[:,2]+tot_pred_y2[:,2])/2\n# tile_df[\"pred_3\"]=(tot_pred_y1[:,3]+tot_pred_y2[:,3])/2\n# tile_df[\"pred_4\"]=(tot_pred_y1[:,4]+tot_pred_y2[:,4])/2\n\ntile_df1=tile_df.copy()\ntile_df2=tile_df.copy()\ntile_df3=tile_df.copy()\n\ntile_df1[\"pred_0\"]=tot_pred_y1[:,0]\ntile_df1[\"pred_1\"]=tot_pred_y1[:,1]\ntile_df1[\"pred_2\"]=tot_pred_y1[:,2]\ntile_df1[\"pred_3\"]=tot_pred_y1[:,3]\ntile_df1[\"pred_4\"]=tot_pred_y1[:,4]\ntile_df1[\"aux\"]=tot_aux1\n\ntile_df2[\"pred_0\"]=tot_pred_y2[:,0]\ntile_df2[\"pred_1\"]=tot_pred_y2[:,1]\ntile_df2[\"pred_2\"]=tot_pred_y2[:,2]\ntile_df2[\"pred_3\"]=tot_pred_y2[:,3]\ntile_df2[\"pred_4\"]=tot_pred_y2[:,4]\ntile_df2[\"aux\"]=tot_aux2\n\ntile_df3[\"pred_0\"]=tot_pred_y3[:,0]\ntile_df3[\"pred_1\"]=tot_pred_y3[:,1]\ntile_df3[\"pred_2\"]=tot_pred_y3[:,2]\ntile_df3[\"pred_3\"]=tot_pred_y3[:,3]\ntile_df3[\"pred_4\"]=tot_pred_y3[:,4]\ntile_df3[\"aux\"]=tot_aux3\n\n\ntile_df1[\"pred_2\"]=[x*0.95 if x<0.65 else x for x in tile_df1[\"pred_2\"]]\ntile_df2[\"pred_2\"]=[x*0.9 if x<0.85 else x for x in tile_df2[\"pred_2\"]]\n#tile_df3[\"pred_1\"]=[x*0.7 if x<0.45 else x for x in tile_df3[\"pred_1\"]] #0.95 0.4  0.95 0.55 0.55 0.75","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:03.893499Z","iopub.execute_input":"2024-01-01T22:22:03.893814Z","iopub.status.idle":"2024-01-01T22:22:03.913127Z","shell.execute_reply.started":"2024-01-01T22:22:03.893788Z","shell.execute_reply":"2024-01-01T22:22:03.912271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tile_df1[\"prob\"]=np.max(tile_df1[[\"pred_0\",\"pred_1\",\"pred_2\",\"pred_3\",\"pred_4\"]],axis=1)\ntile_df2[\"prob\"]=np.max(tile_df2[[\"pred_0\",\"pred_1\",\"pred_2\",\"pred_3\",\"pred_4\"]],axis=1)\ntile_df3[\"prob\"]=np.max(tile_df3[[\"pred_0\",\"pred_1\",\"pred_2\",\"pred_3\",\"pred_4\"]],axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:03.914082Z","iopub.execute_input":"2024-01-01T22:22:03.914342Z","iopub.status.idle":"2024-01-01T22:22:03.930075Z","shell.execute_reply.started":"2024-01-01T22:22:03.914317Z","shell.execute_reply":"2024-01-01T22:22:03.929241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tile_df1[\"pred\"]=np.argmax(tile_df1[[\"pred_0\",\"pred_1\",\"pred_2\",\"pred_3\",\"pred_4\"]].values,axis=1)\ntile_df2[\"pred\"]=np.argmax(tile_df2[[\"pred_0\",\"pred_1\",\"pred_2\",\"pred_3\",\"pred_4\"]].values,axis=1)\ntile_df3[\"pred\"]=np.argmax(tile_df3[[\"pred_0\",\"pred_1\",\"pred_2\",\"pred_3\",\"pred_4\"]].values,axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:03.931511Z","iopub.execute_input":"2024-01-01T22:22:03.931873Z","iopub.status.idle":"2024-01-01T22:22:03.942515Z","shell.execute_reply.started":"2024-01-01T22:22:03.931842Z","shell.execute_reply":"2024-01-01T22:22:03.941767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#tile_df1","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:03.945735Z","iopub.execute_input":"2024-01-01T22:22:03.946086Z","iopub.status.idle":"2024-01-01T22:22:03.952317Z","shell.execute_reply.started":"2024-01-01T22:22:03.946054Z","shell.execute_reply":"2024-01-01T22:22:03.951517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tile_df1=tile_df1[[\"image_id\",\"pred\",\"prob\",\"aux\"]].groupby([\"image_id\",\"pred\"])[[\"prob\",\"aux\"]].mean().reset_index()\ntile_df2=tile_df2[[\"image_id\",\"pred\",\"prob\",\"aux\"]].groupby([\"image_id\",\"pred\"])[[\"prob\",\"aux\"]].mean().reset_index()\ntile_df3=tile_df3[[\"image_id\",\"pred\",\"prob\",\"aux\"]].groupby([\"image_id\",\"pred\"])[[\"prob\",\"aux\"]].mean().reset_index()\n","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:03.953354Z","iopub.execute_input":"2024-01-01T22:22:03.953637Z","iopub.status.idle":"2024-01-01T22:22:03.978831Z","shell.execute_reply.started":"2024-01-01T22:22:03.953606Z","shell.execute_reply":"2024-01-01T22:22:03.977993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#tile_df3","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:03.980022Z","iopub.execute_input":"2024-01-01T22:22:03.980345Z","iopub.status.idle":"2024-01-01T22:22:03.98463Z","shell.execute_reply.started":"2024-01-01T22:22:03.980313Z","shell.execute_reply":"2024-01-01T22:22:03.983717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# tile_df","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:03.985821Z","iopub.execute_input":"2024-01-01T22:22:03.98626Z","iopub.status.idle":"2024-01-01T22:22:03.991331Z","shell.execute_reply.started":"2024-01-01T22:22:03.986235Z","shell.execute_reply":"2024-01-01T22:22:03.990579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"idx1=tile_df1.groupby([\"image_id\"])[\"prob\"].idxmax()\ntile_df1=tile_df1.loc[idx1].reset_index(drop=True)\n\nidx2=tile_df2.groupby([\"image_id\"])[\"prob\"].idxmax()\ntile_df2=tile_df2.loc[idx2].reset_index(drop=True)\n\nidx3=tile_df3.groupby([\"image_id\"])[\"prob\"].idxmax()\ntile_df3=tile_df3.loc[idx3].reset_index(drop=True)\n","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:03.992402Z","iopub.execute_input":"2024-01-01T22:22:03.992677Z","iopub.status.idle":"2024-01-01T22:22:04.006787Z","shell.execute_reply.started":"2024-01-01T22:22:03.99265Z","shell.execute_reply":"2024-01-01T22:22:04.006033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#tile_df1","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:04.0078Z","iopub.execute_input":"2024-01-01T22:22:04.00808Z","iopub.status.idle":"2024-01-01T22:22:04.015279Z","shell.execute_reply.started":"2024-01-01T22:22:04.008057Z","shell.execute_reply":"2024-01-01T22:22:04.014505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tile_df1.rename(columns={\"pred\":\"pred1\",\"aux\":\"aux1\"},inplace=True)\ntile_df2.rename(columns={\"pred\":\"pred2\",\"aux\":\"aux2\"},inplace=True)\ntile_df3.rename(columns={\"pred\":\"pred3\",\"aux\":\"aux3\"},inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:04.016429Z","iopub.execute_input":"2024-01-01T22:22:04.016722Z","iopub.status.idle":"2024-01-01T22:22:04.024695Z","shell.execute_reply.started":"2024-01-01T22:22:04.016694Z","shell.execute_reply":"2024-01-01T22:22:04.023976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tile_df1[\"aux1\"]=(tile_df1[\"aux1\"]>0.6).astype(int) #0.5\ntile_df2[\"aux2\"]=(tile_df2[\"aux2\"]>0.55).astype(int) #0.5\ntile_df3[\"aux3\"]=(tile_df3[\"aux3\"]>0.5).astype(int) #0.5","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:04.025855Z","iopub.execute_input":"2024-01-01T22:22:04.02616Z","iopub.status.idle":"2024-01-01T22:22:04.035334Z","shell.execute_reply.started":"2024-01-01T22:22:04.026136Z","shell.execute_reply":"2024-01-01T22:22:04.03455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tile_df3","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:04.036393Z","iopub.execute_input":"2024-01-01T22:22:04.036651Z","iopub.status.idle":"2024-01-01T22:22:04.05261Z","shell.execute_reply.started":"2024-01-01T22:22:04.036628Z","shell.execute_reply":"2024-01-01T22:22:04.051818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tile_df=tile_df1[[\"image_id\",\"pred1\",\"aux1\"]].copy()","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:04.053565Z","iopub.execute_input":"2024-01-01T22:22:04.053819Z","iopub.status.idle":"2024-01-01T22:22:04.059412Z","shell.execute_reply.started":"2024-01-01T22:22:04.053797Z","shell.execute_reply":"2024-01-01T22:22:04.05851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tile_df=tile_df.merge(tile_df2[[\"image_id\",\"pred2\",\"aux2\"]],on=\"image_id\")\ntile_df=tile_df.merge(tile_df3[[\"image_id\",\"pred3\",\"aux3\"]],on=\"image_id\")","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:04.060751Z","iopub.execute_input":"2024-01-01T22:22:04.061138Z","iopub.status.idle":"2024-01-01T22:22:04.075242Z","shell.execute_reply.started":"2024-01-01T22:22:04.061103Z","shell.execute_reply":"2024-01-01T22:22:04.074343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def most_common(array1, array2, array3):\n    combined_array = np.array([array1, array2, array3])\n\n    result = np.zeros_like(array1)  \n\n    for i in range(len(array1)):\n        counts = {}  \n\n        for j in range(3):\n            num = combined_array[j, i]\n            counts[num] = counts.get(num, 0) + 1\n\n        max_count = max(counts.values())\n        most_common_num = max([num for num, count in counts.items() if count == max_count])#####\n        result[i] = most_common_num\n\n    return result","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:04.076705Z","iopub.execute_input":"2024-01-01T22:22:04.077101Z","iopub.status.idle":"2024-01-01T22:22:04.083834Z","shell.execute_reply.started":"2024-01-01T22:22:04.077075Z","shell.execute_reply":"2024-01-01T22:22:04.082999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tile_df[\"pred\"]=most_common(tile_df[\"pred1\"].values,tile_df[\"pred2\"].values,tile_df[\"pred3\"].values)\n#tile_df[\"aux\"]=most_common(tile_df[\"aux1\"].values,tile_df[\"aux2\"].values,tile_df[\"aux3\"].values)\ntile_df[\"aux\"]=(tile_df[\"aux1\"].values|tile_df[\"aux2\"].values|tile_df[\"aux3\"].values)","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:04.084891Z","iopub.execute_input":"2024-01-01T22:22:04.085206Z","iopub.status.idle":"2024-01-01T22:22:04.096986Z","shell.execute_reply.started":"2024-01-01T22:22:04.085182Z","shell.execute_reply":"2024-01-01T22:22:04.096164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:24:20.058564Z","iopub.execute_input":"2024-01-01T22:24:20.059227Z","iopub.status.idle":"2024-01-01T22:24:20.065631Z","shell.execute_reply.started":"2024-01-01T22:24:20.059186Z","shell.execute_reply":"2024-01-01T22:24:20.064655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tile_df[\"label\"]=le.inverse_transform(tile_df[\"pred\"].values)","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:04.732514Z","iopub.status.idle":"2024-01-01T22:22:04.732875Z","shell.execute_reply.started":"2024-01-01T22:22:04.732704Z","shell.execute_reply":"2024-01-01T22:22:04.732721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tile_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:04.734053Z","iopub.status.idle":"2024-01-01T22:22:04.734379Z","shell.execute_reply.started":"2024-01-01T22:22:04.734221Z","shell.execute_reply":"2024-01-01T22:22:04.734236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#other_id=[x for x in np.unique(tile_df0[\"image_id\"]) if x not in tile_df[\"image_id\"].values]\n#other_id=tile_df[tile_df[\"prob\"]<0.4][\"image_id\"].values\nother_id=tile_df[tile_df[\"aux\"]==0][\"image_id\"].values","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:04.736071Z","iopub.status.idle":"2024-01-01T22:22:04.736416Z","shell.execute_reply.started":"2024-01-01T22:22:04.73625Z","shell.execute_reply":"2024-01-01T22:22:04.736267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"other_id","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:04.737998Z","iopub.status.idle":"2024-01-01T22:22:04.738346Z","shell.execute_reply.started":"2024-01-01T22:22:04.73818Z","shell.execute_reply":"2024-01-01T22:22:04.738196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# WSI_thumbnail model","metadata":{}},{"cell_type":"code","source":"df_wsi_thumbnail=df_wsi.copy()","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:04.739534Z","iopub.status.idle":"2024-01-01T22:22:04.739867Z","shell.execute_reply.started":"2024-01-01T22:22:04.739703Z","shell.execute_reply":"2024-01-01T22:22:04.739719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_test_file_path(image_id):\n    if os.path.exists(f\"{root_dir}/{inference}_thumbnails/{image_id}_thumbnail.png\"):\n        return f\"{root_dir}/{inference}_thumbnails/{image_id}_thumbnail.png\"\n    else:\n        return f\"{root_dir}/{inference}_images/{image_id}.png\"","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:04.741528Z","iopub.status.idle":"2024-01-01T22:22:04.741869Z","shell.execute_reply.started":"2024-01-01T22:22:04.741704Z","shell.execute_reply":"2024-01-01T22:22:04.741721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_wsi_thumbnail['file_path'] = df_wsi_thumbnail['image_id'].apply(get_test_file_path)","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:04.74332Z","iopub.status.idle":"2024-01-01T22:22:04.743692Z","shell.execute_reply.started":"2024-01-01T22:22:04.743517Z","shell.execute_reply":"2024-01-01T22:22:04.743534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_wsi_thumbnail.head()","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:04.745096Z","iopub.status.idle":"2024-01-01T22:22:04.745428Z","shell.execute_reply.started":"2024-01-01T22:22:04.745266Z","shell.execute_reply":"2024-01-01T22:22:04.745282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Net(nn.Module):\n    def __init__(self):\n        super(Net, self).__init__()\n        self.encoder = tf_efficientnetv2_s_in21ft1k(pretrained=False)\n        self.logit = nn.Linear(1280,5) #1792\n \n    def forward(self, image):\n        e = self.encoder\n        \n        x = e.forward_features(image)\n        x = F.adaptive_avg_pool2d(x,1)\n        x = torch.flatten(x,1,3)\n       \n        logit = self.logit(x)\n     \n        return logit\n    \ndef crop_image(img1,show=False):\n    # Binarize the image\n    img2 = np.mean(img1,axis=2)\n    bin_pixels = (cv2.threshold(img2, 0, 255, cv2.THRESH_BINARY)[1]).astype(\"uint8\")\n   \n    # Make contours around the binarized image, keep only the largest contour\n    contours, _ = cv2.findContours(bin_pixels, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_NONE)\n    contour = max(contours, key=cv2.contourArea)\n    \n    \n    # get bounding box of contour\n    y1, y2 = np.min(contour[:, :, 1]), np.max(contour[:, :, 1])\n    x1, x2 = np.min(contour[:, :, 0]), np.max(contour[:, :, 0])\n    \n\n    if show:\n        plt.imshow(img1[y1:y2, x1:x2,:]) ; \n  \n\n    return img1[y1:y2, x1:x2,:]\n\nclass UBCDataset(Dataset):\n    def __init__(self, df, transforms=None):\n        self.df = df\n        self.file_names = df['file_path'].values\n#         self.labels = df['label'].values\n        self.transforms = transforms\n        \n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, index):\n        img_path = self.file_names[index]\n        img = cv2.imread(img_path)\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        img = crop_image(img,show=False)\n#         label = self.labels[index]\n        \n        if self.transforms is not None:\n            img = self.transforms(image=img)[\"image\"]\n            \n        return {\n            'image': img,\n#             'label': torch.tensor(label, dtype=torch.long)\n        }\n    \ndata_transforms = {\n\n    \n    \"valid\": A.Compose([\n        A.Resize(512, 512),\n\n        #A.Normalize(),\n        ToTensorV2()], p=1.)\n}","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:04.747069Z","iopub.status.idle":"2024-01-01T22:22:04.747382Z","shell.execute_reply.started":"2024-01-01T22:22:04.747226Z","shell.execute_reply":"2024-01-01T22:22:04.747241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset = UBCDataset(df_wsi_thumbnail,transforms=data_transforms[\"valid\"])\n\ntest_loader = DataLoader(test_dataset, batch_size=16, \n                          num_workers=2, shuffle=False, pin_memory=True)","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:04.749053Z","iopub.status.idle":"2024-01-01T22:22:04.749389Z","shell.execute_reply.started":"2024-01-01T22:22:04.749226Z","shell.execute_reply":"2024-01-01T22:22:04.749242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models=[]\nweights=[\"/kaggle/input/ubc-ocean-weights-fate/job_3_effnetv2s_fold_0.bin\",\n        \"/kaggle/input/ubc-ocean-weights-fate/job_3_effnetv2s_fold_1.bin\",\n        \"/kaggle/input/ubc-ocean-weights-fate/job_3_effnetv2s_fold_2.bin\",\n        \"/kaggle/input/ubc-ocean-weights-fate/job_3_effnetv2s_fold_3.bin\",\n        \"/kaggle/input/ubc-ocean-weights-fate/job_3_effnetv2s_fold_4.bin\",\n        ]\nfor i in range(len(weights)):\n    model = Net()\n    checkpoint = weights[i]\n    f = torch.load(checkpoint, map_location=lambda storage, loc: storage)\n    model.load_state_dict(f, strict=False) #False\n    model.cuda()\n    model.eval()\n    models.append(model)","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:04.750553Z","iopub.status.idle":"2024-01-01T22:22:04.750859Z","shell.execute_reply.started":"2024-01-01T22:22:04.750705Z","shell.execute_reply":"2024-01-01T22:22:04.75072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_net=len(models)\npred=[]\n# model.cuda()\n# model.eval()\nbar = tqdm(enumerate(test_loader), total=len(test_loader))\nfor step, data in bar:\n\n    images = data['image'].to(\"cuda:0\", dtype=torch.float)\n#     images1 = torch.flip(images, dims=[3, ])  #2,3 \n#     images2 = torch.flip(images, dims=[2, ])  #2,3 \n\n    p=0\n    count=0\n    with torch.no_grad():\n        for i in range(num_net):\n            output = models[i](images)\n            tmp_pred=torch.nn.Softmax(dim=1)(output)\n            p+=tmp_pred\n            count+=1\n            \n            \n#             output = models[i](images1)\n#             tmp_pred=torch.nn.Softmax(dim=1)(output)\n#             p+=tmp_pred\n#             count+=1\n            \n#             output = models[i](images2)\n#             tmp_pred=torch.nn.Softmax(dim=1)(output)\n#             p+=tmp_pred\n#             count+=1\n            \n    p=p/count\n    pred.append(p.cpu().numpy())\n    \n    del images\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:04.753176Z","iopub.status.idle":"2024-01-01T22:22:04.753496Z","shell.execute_reply.started":"2024-01-01T22:22:04.753337Z","shell.execute_reply":"2024-01-01T22:22:04.753352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del models\ndel test_loader\ntorch.cuda.empty_cache()\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:04.754466Z","iopub.status.idle":"2024-01-01T22:22:04.754773Z","shell.execute_reply.started":"2024-01-01T22:22:04.754619Z","shell.execute_reply":"2024-01-01T22:22:04.754633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = np.argmax(np.concatenate(pred),axis=1)\npred = le.inverse_transform(pred)","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:04.756212Z","iopub.status.idle":"2024-01-01T22:22:04.756549Z","shell.execute_reply.started":"2024-01-01T22:22:04.756383Z","shell.execute_reply":"2024-01-01T22:22:04.756399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_wsi_thumbnail[\"label\"]=pred","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:04.758201Z","iopub.status.idle":"2024-01-01T22:22:04.758516Z","shell.execute_reply.started":"2024-01-01T22:22:04.758357Z","shell.execute_reply":"2024-01-01T22:22:04.758372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_wsi_thumbnail_res=df_wsi_thumbnail[~df_wsi_thumbnail[\"image_id\"].isin(tile_df[\"image_id\"].values)]","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:04.759666Z","iopub.status.idle":"2024-01-01T22:22:04.760006Z","shell.execute_reply.started":"2024-01-01T22:22:04.759819Z","shell.execute_reply":"2024-01-01T22:22:04.759833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# WSI merge","metadata":{}},{"cell_type":"code","source":"df_wsi=pd.concat([df_wsi_thumbnail_res[[\"image_id\",\"label\"]],tile_df[[\"image_id\",\"label\"]]]).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:04.761462Z","iopub.status.idle":"2024-01-01T22:22:04.761909Z","shell.execute_reply.started":"2024-01-01T22:22:04.761678Z","shell.execute_reply":"2024-01-01T22:22:04.7617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"########\ndf_wsi.loc[df_wsi[\"image_id\"].isin(other_id),\"label\"]=\"Other\"","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:04.762933Z","iopub.status.idle":"2024-01-01T22:22:04.763398Z","shell.execute_reply.started":"2024-01-01T22:22:04.763168Z","shell.execute_reply":"2024-01-01T22:22:04.763189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:04.765064Z","iopub.status.idle":"2024-01-01T22:22:04.765396Z","shell.execute_reply.started":"2024-01-01T22:22:04.765234Z","shell.execute_reply":"2024-01-01T22:22:04.76525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_wsi.head()","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:04.766774Z","iopub.status.idle":"2024-01-01T22:22:04.767144Z","shell.execute_reply.started":"2024-01-01T22:22:04.766946Z","shell.execute_reply":"2024-01-01T22:22:04.766991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# TMA","metadata":{}},{"cell_type":"code","source":"def crop_tma(img):\n    ks=min(min(img.shape[0],img.shape[1])//150,20)\n    \n    mask=(img.max(axis=2)-img.min(axis=2))>20\n    kernel = np.ones((ks, ks),np.uint8)\n    mask=cv2.erode(mask.astype(np.uint8),kernel)\n    nonzero_pixels = np.column_stack(np.where(mask > 0))\n    \n    if (nonzero_pixels.size)<(img.size//60):\n        return img\n    else:\n    \n        min_y, min_x = np.min(nonzero_pixels, axis=0)\n        max_y, max_x = np.max(nonzero_pixels, axis=0)\n\n        return img[max(0,min_y-ks):max_y+ks+1,max(0,min_x-ks):max_x+ks+1,:]","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:04.768874Z","iopub.status.idle":"2024-01-01T22:22:04.769237Z","shell.execute_reply.started":"2024-01-01T22:22:04.769073Z","shell.execute_reply":"2024-01-01T22:22:04.76909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_tma.head()","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:04.770427Z","iopub.status.idle":"2024-01-01T22:22:04.770736Z","shell.execute_reply.started":"2024-01-01T22:22:04.770582Z","shell.execute_reply":"2024-01-01T22:22:04.770596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_tma['file_path'] = df_tma['image_id'].apply(get_test_file_path)","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:04.772247Z","iopub.status.idle":"2024-01-01T22:22:04.772582Z","shell.execute_reply.started":"2024-01-01T22:22:04.772419Z","shell.execute_reply":"2024-01-01T22:22:04.772435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_tma.head()","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:04.773714Z","iopub.status.idle":"2024-01-01T22:22:04.774079Z","shell.execute_reply.started":"2024-01-01T22:22:04.773879Z","shell.execute_reply":"2024-01-01T22:22:04.773895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Net_v1(nn.Module):\n    def __init__(self):\n        super(Net_v1, self).__init__()\n        self.encoder = tf_efficientnet_b4_ns(pretrained=False)\n        self.logit = nn.Linear(1792,5) #1280#1792 (5)\n        self.aux = nn.Linear(1792,1)\n        self.aux2 = nn.Linear(1792,1)\n    def forward(self, image):\n        e = self.encoder\n        \n        x = e.forward_features(image)\n        x = F.adaptive_avg_pool2d(x,1)\n        x = torch.flatten(x,1,3)\n       \n        logit = self.logit(x)\n        aux = self.aux(x)\n        aux2 = self.aux2(x)\n        return logit,aux,aux2\n\nclass Net_v2(nn.Module):\n    def __init__(self):\n        super(Net_v2, self).__init__()\n        self.encoder = tf_efficientnetv2_s_in21ft1k(pretrained=False)\n        self.logit = nn.Linear(1280,5) #1280#1792 (5)\n        self.aux = nn.Linear(1280,1)\n        self.aux2 = nn.Linear(1280,1)\n    def forward(self, image):\n        e = self.encoder\n        \n        x = e.forward_features(image)\n        x = F.adaptive_avg_pool2d(x,1)\n        x = torch.flatten(x,1,3)\n       \n        logit = self.logit(x)\n        aux = self.aux(x)\n        aux2 = self.aux2(x)\n        return logit,aux,aux2\n    \n    \nclass Net_v3(nn.Module):\n    def __init__(self):\n        super(Net_v3, self).__init__()\n        self.encoder = maxvit_tiny_tf_512(pretrained=False)\n        self.logit = nn.Linear(512,5) #1280#1792 (5)\n        self.aux = nn.Linear(512,1)\n        self.aux2 = nn.Linear(512,1)\n    def forward(self, image):\n        e = self.encoder\n        \n        x = e.forward_features(image)\n        x = F.adaptive_avg_pool2d(x,1)\n        x = torch.flatten(x,1,3)\n       \n        logit = self.logit(x)\n        aux = self.aux(x)\n        aux2 = self.aux2(x)\n        return logit,aux,aux2","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:04.775275Z","iopub.status.idle":"2024-01-01T22:22:04.775583Z","shell.execute_reply.started":"2024-01-01T22:22:04.775428Z","shell.execute_reply":"2024-01-01T22:22:04.775443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class UBCDataset_tma(Dataset):\n    def __init__(self, df, transforms=None):\n        self.df = df\n        self.file_names = df['file_path'].values\n#         self.labels = df['label'].values\n        self.transforms = transforms\n        \n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, index): \n        \n        img_path = self.file_names[index]\n\n        img = cv2.imread(img_path)\n        \n#         #######################################\n#         img=cv2.resize(img,(0,0),fx=0.165,fy=0.165,interpolation=cv2.INTER_AREA)\n        \n#         h = img.shape[0]\n#         w = img.shape[1]\n#         h_c = (h//2) \n#         w_c = (w//2)\n#         if (h>512) &(w>512):\n#             img=img[h_c-256:h_c+256,w_c-256:w_c+256,:]\n#         ########################################\n        \n        \n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        ####################\n        try:\n            img = crop_tma(img)\n        except:\n            pass\n        ###################\n        img = img.astype(np.float32)/255\n        \n#         label = self.labels[index]\n        \n        if self.transforms is not None:\n            img = self.transforms(image=img)[\"image\"]\n                \n        return {\n            'image': img,\n#             'label': torch.tensor(label, dtype=torch.long)\n        }","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:04.777202Z","iopub.status.idle":"2024-01-01T22:22:04.777522Z","shell.execute_reply.started":"2024-01-01T22:22:04.777359Z","shell.execute_reply":"2024-01-01T22:22:04.777374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_transforms_tma = {\n\n    \n    \"valid\": A.Compose([\n        A.Resize(512, 512),\n\n        #A.Normalize(),\n        ToTensorV2()], p=1.)\n}","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:04.77883Z","iopub.status.idle":"2024-01-01T22:22:04.779181Z","shell.execute_reply.started":"2024-01-01T22:22:04.779014Z","shell.execute_reply":"2024-01-01T22:22:04.77903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_dataset = UBCDataset_tma(df_tma,transforms=data_transforms_tma[\"valid\"])\n\ntma_loader = DataLoader(valid_dataset, batch_size=16, \n                          num_workers=2, shuffle=False, pin_memory=True)","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:04.780604Z","iopub.status.idle":"2024-01-01T22:22:04.780937Z","shell.execute_reply.started":"2024-01-01T22:22:04.780774Z","shell.execute_reply":"2024-01-01T22:22:04.78079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models1=[]\nweights=[\n        \"/kaggle/input/ubc-ocean-weights-bce-v1/tile_0_step3_51_effnetb4_fold_0.bin\",\n        \"/kaggle/input/ubc-ocean-weights-bce-v1/tile_0_step3_51_effnetb4_fold_1.bin\",\n        \"/kaggle/input/ubc-ocean-weights-bce-v1/tile_0_step3_51_effnetb4_fold_2.bin\",\n        \"/kaggle/input/ubc-ocean-weights-bce-v1/tile_0_step3_51_effnetb4_fold_3.bin\",\n        \"/kaggle/input/ubc-ocean-weights-bce-v1/tile_0_step3_51_effnetb4_fold_4.bin\",\n        ]\nfor i in range(len(weights)):\n    model = Net_v1()\n    checkpoint = weights[i]\n    f = torch.load(checkpoint, map_location=lambda storage, loc: storage)\n    model.load_state_dict(f, strict=False) #False\n    model.cuda()\n    model.eval()\n    models1.append(model)\n\n\n\nmodels2=[]\nweights=[\n\n        \"/kaggle/input/ubc-ocean-weights-bce-v1/tile_0_step3_52_effnetv2s_fold_0.bin\",\n        \"/kaggle/input/ubc-ocean-weights-bce-v1/tile_0_step3_52_effnetv2s_fold_1.bin\",\n        \"/kaggle/input/ubc-ocean-weights-bce-v1/tile_0_step3_52_effnetv2s_fold_2.bin\",\n        \"/kaggle/input/ubc-ocean-weights-bce-v1/tile_0_step3_52_effnetv2s_fold_3.bin\",\n        \"/kaggle/input/ubc-ocean-weights-bce-v1/tile_0_step3_52_effnetv2s_fold_4.bin\",\n        ]\nfor i in range(len(weights)):\n    model = Net_v2()\n    checkpoint = weights[i]\n    f = torch.load(checkpoint, map_location=lambda storage, loc: storage)\n    model.load_state_dict(f, strict=False) #False\n    model.cuda()\n    model.eval()\n    models2.append(model)\n    \n    \n\n    \nmodels3=[]\nweights=[\n        \"/kaggle/input/ubc-ocean-weights-bce-v1/tile_1_step3_50_1_maxvit_fold_0_tma.bin\",\n        \"/kaggle/input/ubc-ocean-weights-bce-v1/tile_1_step3_50_1_maxvit_fold_1_tma.bin\",\n        \"/kaggle/input/ubc-ocean-weights-bce-v1/tile_1_step3_50_1_maxvit_fold_2_tma.bin\",\n        \"/kaggle/input/ubc-ocean-weights-bce-v1/tile_1_step3_50_1_maxvit_fold_3_tma.bin\",\n        \"/kaggle/input/ubc-ocean-weights-bce-v1/tile_1_step3_50_1_maxvit_fold_4_tma.bin\",\n        ]\nfor i in range(len(weights)):\n    model = Net_v3()\n    checkpoint = weights[i]\n    f = torch.load(checkpoint, map_location=lambda storage, loc: storage)\n    model.load_state_dict(f, strict=True) #False\n    model.cuda()\n    model.eval()\n    models3.append(model)","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:04.782123Z","iopub.status.idle":"2024-01-01T22:22:04.782465Z","shell.execute_reply.started":"2024-01-01T22:22:04.782298Z","shell.execute_reply":"2024-01-01T22:22:04.782315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_net1=len(models1)\npred1=[]\ntot_aux1=[]\n\nnum_net2=len(models2)\npred2=[]\ntot_aux2=[]\n\nnum_net3=len(models3)\npred3=[]\ntot_aux3=[]\n\nmodel.cuda()\nmodel.eval()\nbar = tqdm(enumerate(tma_loader), total=len(tma_loader))\nfor step, data in bar:\n\n    images = data['image'].to(\"cuda:0\", dtype=torch.float)\n#     images1 = torch.flip(images, dims=[3, ])  #2,3\n#     images2 = torch.flip(images, dims=[2, ]) \n#     images3 = torch.flip(images1, dims=[2, ])\n    \n    p=0\n    ax=0\n    count=0\n    with torch.no_grad():\n        for i in range(num_net1):\n            output,aux,_ = models1[i](images)\n            tmp_pred=F.sigmoid(output)\n            tmp_aux=F.sigmoid(aux)\n            p+=tmp_pred\n            ax+=tmp_aux\n            count+=1\n            \n    p=p/count\n    ax=ax/count\n    pred1.append(p.cpu().numpy())\n    tot_aux1.append(ax.cpu().numpy())\n    \n    p=0\n    ax=0\n    count=0\n    with torch.no_grad():\n        for i in range(num_net2):\n            output,aux,_ = models2[i](images)\n            tmp_pred=F.sigmoid(output)\n            tmp_aux=F.sigmoid(aux)\n            p+=tmp_pred\n            ax+=tmp_aux\n            count+=1\n            \n\n            \n    p=p/count\n    ax=ax/count\n    pred2.append(p.cpu().numpy())\n    tot_aux2.append(ax.cpu().numpy())\n    \n    p=0\n    ax=0\n    count=0\n    with torch.no_grad():\n        for i in range(num_net3):\n            output,aux,_ = models3[i](images)\n            tmp_pred=F.sigmoid(output)\n            tmp_aux=F.sigmoid(aux)\n            p+=tmp_pred\n            ax+=tmp_aux\n            count+=1\n            \n    p=p/count\n    ax=ax/count\n    pred3.append(p.cpu().numpy())\n    tot_aux3.append(ax.cpu().numpy())","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:04.784513Z","iopub.status.idle":"2024-01-01T22:22:04.784847Z","shell.execute_reply.started":"2024-01-01T22:22:04.78467Z","shell.execute_reply":"2024-01-01T22:22:04.784685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# del models\ndel tma_loader\ntorch.cuda.empty_cache()\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:04.786334Z","iopub.status.idle":"2024-01-01T22:22:04.786671Z","shell.execute_reply.started":"2024-01-01T22:22:04.786507Z","shell.execute_reply":"2024-01-01T22:22:04.786523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def most_common(array1, array2, array3):\n    combined_array = np.array([array1, array2, array3])\n\n    result = np.zeros_like(array1)  \n\n    for i in range(len(array1)):\n        counts = {}  \n\n        for j in range(3):\n            num = combined_array[j, i]\n            counts[num] = counts.get(num, 0) + 1\n\n        max_count = max(counts.values())\n        most_common_num = max([num for num, count in counts.items() if count == max_count])#####\n        result[i] = most_common_num\n\n    return result","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:04.787627Z","iopub.status.idle":"2024-01-01T22:22:04.787933Z","shell.execute_reply.started":"2024-01-01T22:22:04.787778Z","shell.execute_reply":"2024-01-01T22:22:04.787793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred=[]\ntry:\n    \n    pred1=np.concatenate(pred1)\n    pred2=np.concatenate(pred2)\n    pred3=np.concatenate(pred3)\n    \n    \n\n#     pred1[:,2]=[x*0.6 if x<0.55 else x for x in pred1[:,2]]#0.6 0.55\n#     pred2[:,2]=[x*0.7 if x<0.5 else x for x in pred2[:,2]] #0.7 0.5\n#     pred3[:,1]=[x*0.7 if x<0.45 else x for x in pred3[:,1]] #0.7 0.5\n    \n#     prob1=np.max(pred1,axis=1)\n#     prob2=np.max(pred2,axis=1)\n    \n    tot_aux1=np.concatenate(tot_aux1)\n    tot_aux2=np.concatenate(tot_aux2)\n    tot_aux3=np.concatenate(tot_aux3)\n    \n    tot_aux1=tot_aux1.flatten()\n    tot_aux2=tot_aux2.flatten()\n    tot_aux3=tot_aux3.flatten()\n    \n    ###\n    pred1 = np.argmax(pred1,axis=1)\n    pred2 = np.argmax(pred2,axis=1)\n    pred3 = np.argmax(pred3,axis=1)\n    \n    pred = most_common(np.array(pred1),np.array(pred2),np.array(pred3))\n    \n    \n   \n    pred = le.inverse_transform(pred)\n    \n    \n    aux=most_common((tot_aux1>0.55).astype(int),(tot_aux2>0.55).astype(int),(tot_aux3>0.55).astype(int)) #0.5 0.5 0.5\n    pred[aux==0]=\"Other\"\n    \n#     pred[tot_aux1<0.5]=\"Other\"\n#     pred[tot_aux2<0.5]=\"Other\"\n#     pred[prob<0.15]=\"Other\"\n\n    \n    \nexcept:\n    pass\n","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:04.789816Z","iopub.status.idle":"2024-01-01T22:22:04.790173Z","shell.execute_reply.started":"2024-01-01T22:22:04.790006Z","shell.execute_reply":"2024-01-01T22:22:04.790022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_tma[\"label\"]=pred","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:04.791669Z","iopub.status.idle":"2024-01-01T22:22:04.792034Z","shell.execute_reply.started":"2024-01-01T22:22:04.791839Z","shell.execute_reply":"2024-01-01T22:22:04.791855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_tma.head(50)","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:04.793227Z","iopub.status.idle":"2024-01-01T22:22:04.793562Z","shell.execute_reply.started":"2024-01-01T22:22:04.793398Z","shell.execute_reply":"2024-01-01T22:22:04.793414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Combine","metadata":{}},{"cell_type":"code","source":"df_pred=pd.concat([df_wsi[[\"image_id\",\"label\"]],df_tma[[\"image_id\",\"label\"]]]).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:04.794588Z","iopub.status.idle":"2024-01-01T22:22:04.794915Z","shell.execute_reply.started":"2024-01-01T22:22:04.794752Z","shell.execute_reply":"2024-01-01T22:22:04.794768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_pred=df_pred.sort_values(by=[\"image_id\"]).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:04.795847Z","iopub.status.idle":"2024-01-01T22:22:04.796187Z","shell.execute_reply.started":"2024-01-01T22:22:04.796025Z","shell.execute_reply":"2024-01-01T22:22:04.796041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_pred.head()","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:04.797706Z","iopub.status.idle":"2024-01-01T22:22:04.79805Z","shell.execute_reply.started":"2024-01-01T22:22:04.797859Z","shell.execute_reply":"2024-01-01T22:22:04.797873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_pred.to_csv(\"submission.csv\",index=False)","metadata":{"execution":{"iopub.status.busy":"2024-01-01T22:22:04.799207Z","iopub.status.idle":"2024-01-01T22:22:04.799537Z","shell.execute_reply.started":"2024-01-01T22:22:04.799376Z","shell.execute_reply":"2024-01-01T22:22:04.799392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}