{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"},{"sourceId":7133840,"sourceType":"datasetVersion","datasetId":4116040},{"sourceId":7303861,"sourceType":"datasetVersion","datasetId":4077749},{"sourceId":3729,"sourceType":"modelInstanceVersion","modelInstanceId":2656}],"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import sys; \nsys.path.append('/kaggle/input/ubc-ocean-pkg/pretrainedmodels-0.7.4/pretrainedmodels-0.7.4')\nsys.path.append('/kaggle/input/ubc-ocean-pkg/EfficientNet-PyTorch-master/EfficientNet-PyTorch-master')\nsys.path.append('/kaggle/input/ubc-ocean-pkg/pytorch-image-models-main/pytorch-image-models-main')\nsys.path.append('/kaggle/input/ubc-ocean-pkg/segmentation_models.pytorch-master/segmentation_models.pytorch-master')","metadata":{"execution":{"iopub.status.busy":"2023-12-28T00:47:21.051233Z","iopub.execute_input":"2023-12-28T00:47:21.051568Z","iopub.status.idle":"2023-12-28T00:47:21.062378Z","shell.execute_reply.started":"2023-12-28T00:47:21.051531Z","shell.execute_reply":"2023-12-28T00:47:21.061483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nos.environ[\"OPENCV_IO_MAX_IMAGE_PIXELS\"] = pow(2, 40).__str__()\n\nimport gc\nimport cv2\nimport math\nimport copy\nimport time\nimport random\nimport glob\nimport os, shutil\nfrom joblib import Parallel, delayed\n\nfrom matplotlib import pyplot as plt\n\n# For data manipulation\nimport numpy as np\nimport pandas as pd\n\n# Pytorch Imports\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nimport torch.nn.functional as F\nfrom torch.optim import lr_scheduler\nfrom torch.utils.data import Dataset, DataLoader\nfrom torch.cuda import amp\nimport torchvision\n\n# Utils\nimport joblib\nfrom tqdm import tqdm\nfrom collections import defaultdict\n\n# For Image Models\nimport timm\nfrom glob import glob\n\n# Albumentations for augmentations\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\n\nfrom pathlib import Path\n\nimport segmentation_models_pytorch as smp\n\n\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")","metadata":{"execution":{"iopub.status.busy":"2023-12-28T00:47:21.063863Z","iopub.execute_input":"2023-12-28T00:47:21.064147Z","iopub.status.idle":"2023-12-28T00:47:29.969471Z","shell.execute_reply.started":"2023-12-28T00:47:21.064123Z","shell.execute_reply":"2023-12-28T00:47:29.96863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:\n    model_name = \"tf_efficientnetv2_s_in21ft1k\"\n    selected_folds = [0, 1, 2, 3, 4]\n    img_size = 512\n    # batch_size and epochs\n    batch_size = 8\n    num_workers = 4\n    label_dict = {0: \"CC\", 1: \"EC\", 2: \"HGSC\", 3: \"LGSC\", 4: \"MC\", 5:\"Other\"}\n    num_classes = 5\n    \n    seg_model_name = \"efficientnet-b0\"\n    seg_model_weight = \"/kaggle/input/ubc-ocean/seg-03-fold-0.pth\"\n    \n    clc_model_name = \"tf_efficientnetv2_s_in21ft1k\"\n    clc_model_weight = \"/kaggle/input/ubc-ocean/exp-13-fold-0.pth\"","metadata":{"execution":{"iopub.status.busy":"2023-12-28T00:47:29.970581Z","iopub.execute_input":"2023-12-28T00:47:29.971031Z","iopub.status.idle":"2023-12-28T00:47:29.97687Z","shell.execute_reply.started":"2023-12-28T00:47:29.971004Z","shell.execute_reply":"2023-12-28T00:47:29.975821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def seed_everything(seed=42):\n    random.seed(seed)\n    os.environ[\"PYTHONHASHSEED\"] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False\n    \nseed_everything()","metadata":{"execution":{"iopub.status.busy":"2023-12-28T00:47:29.979547Z","iopub.execute_input":"2023-12-28T00:47:29.980337Z","iopub.status.idle":"2023-12-28T00:47:30.015358Z","shell.execute_reply.started":"2023-12-28T00:47:29.980286Z","shell.execute_reply":"2023-12-28T00:47:30.014462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Debug= False\ndata_dir = Path(\"/kaggle/input/UBC-OCEAN\")\ntest_thumbnails_path = data_dir / \"test_thumbnails\"\ntest_df = pd.read_csv(data_dir / \"test.csv\",dtype={\"image_id\": str})\ntest_df[\"is_tma\"] =test_df[\"image_width\"] < 6000\n\nimg_path = \"test_images\"\n\nif Debug:\n    train_df = pd.read_csv(data_dir / \"train.csv\" ,dtype={\"image_id\": str})\n    test_df = train_df[:10]\n    img_path = \"train_images\"","metadata":{"execution":{"iopub.status.busy":"2023-12-28T00:47:30.016493Z","iopub.execute_input":"2023-12-28T00:47:30.017143Z","iopub.status.idle":"2023-12-28T00:47:30.046215Z","shell.execute_reply.started":"2023-12-28T00:47:30.01711Z","shell.execute_reply":"2023-12-28T00:47:30.045364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df","metadata":{"execution":{"iopub.status.busy":"2023-12-28T00:47:30.047382Z","iopub.execute_input":"2023-12-28T00:47:30.047945Z","iopub.status.idle":"2023-12-28T00:47:30.062137Z","shell.execute_reply.started":"2023-12-28T00:47:30.047918Z","shell.execute_reply":"2023-12-28T00:47:30.06124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['image_pth'] = f\"/kaggle/input/UBC-OCEAN/{img_path}/\" + test_df[\"image_id\"] + \".png\"\ntest_df['image_pth'] = test_df['image_pth']","metadata":{"execution":{"iopub.status.busy":"2023-12-28T00:47:30.063291Z","iopub.execute_input":"2023-12-28T00:47:30.064135Z","iopub.status.idle":"2023-12-28T00:47:30.069639Z","shell.execute_reply.started":"2023-12-28T00:47:30.06411Z","shell.execute_reply":"2023-12-28T00:47:30.068735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tma_df = test_df[test_df[\"is_tma\"] == True].reset_index(drop=True)\nnot_tma_df = test_df[test_df[\"is_tma\"] == False].reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2023-12-28T00:47:30.07099Z","iopub.execute_input":"2023-12-28T00:47:30.071271Z","iopub.status.idle":"2023-12-28T00:47:30.08322Z","shell.execute_reply.started":"2023-12-28T00:47:30.071247Z","shell.execute_reply":"2023-12-28T00:47:30.082332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class UBCDataset(Dataset):\n    def __init__(self, df, transforms=None):\n        self.img_paths = df[\"tile_path\"].values\n        self.transforms = transforms\n        \n    def __len__(self):\n        return len(self.img_paths)\n    \n    def __getitem__(self, index):\n        \n        img_path = self.img_paths[index]\n        img = cv2.imread(str(img_path))\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    \n        if self.transforms:\n            img = self.transforms(image=img)[\"image\"]\n            \n        return {\n            \"image\": img,\n        }\n    \n\n\n\nclass GeM(nn.Module):\n    def __init__(self, p=3, eps=1e-6):\n        super(GeM, self).__init__()\n        self.p = nn.Parameter(torch.ones(1) * p)\n        self.eps = eps\n\n    def forward(self, x):\n        return self.gem(x, p=self.p, eps=self.eps)\n\n    def gem(self, x, p=3, eps=1e-6):\n        return F.avg_pool2d(x.clamp(min=eps).pow(p), (x.size(-2), x.size(-1))).pow(\n            1.0 / p\n        )\n\n    def __repr__(self):\n        return (\n                self.__class__.__name__\n                + \"(\"\n                + \"p=\"\n                + \"{:.4f}\".format(self.p.data.tolist()[0])\n                + \", \"\n                + \"eps=\"\n                + str(self.eps)\n                + \")\"\n        )\n\n\nclass UBCModel(nn.Module):\n    def __init__(self, model_name, num_classes, pretrained=False, checkpoint_path=None):\n        super(UBCModel, self).__init__()\n        self.model = timm.create_model(model_name, pretrained=pretrained)\n\n        in_features = self.model.classifier.in_features\n        self.model.classifier = nn.Identity()\n        self.model.global_pool = nn.Identity()\n        self.pooling = GeM()\n        self.linear = nn.Linear(in_features, num_classes)\n\n    def forward(self, images):\n        features = self.model(images)\n        pooled_features = self.pooling(features).flatten(1)\n        output = self.linear(pooled_features)\n        return output\n    \nclass UBCClcDataset(Dataset):\n    def __init__(self, df, transforms=None):\n        self.df = df\n        self.file_names = df[\"tile_path\"].values\n        self.transforms = transforms\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, index):\n        img_path = self.file_names[index]\n        img = cv2.imread(img_path)\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n\n        if self.transforms:\n            img = self.transforms(image=img)[\"image\"]\n\n        return {\"image\": img}\n\n    \ndef seg_infer(test_loader, model):\n    \n    mask_ratio = []\n    for index, data in enumerate(test_loader):\n        images = data[\"image\"].to(device)\n        batch_size = images.size(0) \n        with torch.no_grad():\n            outputs = model(images)\n\n        probs = torch.sigmoid(outputs)\n        probs = probs.detach().cpu().numpy()\n\n        threshold = 0.33\n        binary_masks = (probs > threshold).astype(int)\n        \n         \n        for _index in range(batch_size):\n            true_pixel_ratio = np.count_nonzero(binary_masks[_index]) /(512*512)\n            mask_ratio.append(true_pixel_ratio)\n\n    torch.cuda.empty_cache()\n    \n    return mask_ratio\n\ndef infer_clc_tma(test_loader, model):\n\n    probs = []\n    for index, data in enumerate(test_loader):\n        data = data[\"image\"].to(device)\n        outputs = model(data)\n        p = torch.sigmoid(outputs)\n        p = p.detach().cpu().numpy()\n        probs.append(p)\n        \n        del data\n        del outputs\n        \n    torch.cuda.empty_cache()\n    probs = np.concatenate(probs)\n\n    return probs\n\ndef infer_clc_not_tma(test_loader, model, tile_df):\n\n    probs = []\n    for index, data in enumerate(test_loader):\n        data = data[\"image\"].to(device)\n        outputs = model(data)\n        p = torch.sigmoid(outputs)\n        p = p.detach().cpu().numpy()\n        probs.append(p)\n        \n        del data\n        del outputs\n        \n    torch.cuda.empty_cache()\n    probs = np.concatenate(probs)\n    \n    tile_df[\"CC\"] = probs[:,0]\n    tile_df[\"EC\"] = probs[:,1]\n    tile_df[\"HGSC\"] = probs[:,2]\n    tile_df[\"LGSC\"] = probs[:,3]\n    tile_df[\"MC\"] = probs[:,4]\n\n    return tile_df","metadata":{"execution":{"iopub.status.busy":"2023-12-28T00:49:02.49281Z","iopub.execute_input":"2023-12-28T00:49:02.493176Z","iopub.status.idle":"2023-12-28T00:49:02.519263Z","shell.execute_reply.started":"2023-12-28T00:49:02.49314Z","shell.execute_reply":"2023-12-28T00:49:02.518367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_transforms():\n    return A.Compose(\n        [\n            A.Resize(CFG.img_size, CFG.img_size),\n            A.Normalize(\n                mean=[0.485, 0.456, 0.406],\n                std=[0.229, 0.224, 0.225],\n                max_pixel_value=255.0,\n                p=1.0,\n            ),\n            ToTensorV2(),\n        ],\n        p=1.0,)","metadata":{"execution":{"iopub.status.busy":"2023-12-28T00:49:02.757735Z","iopub.execute_input":"2023-12-28T00:49:02.758084Z","iopub.status.idle":"2023-12-28T00:49:02.763635Z","shell.execute_reply.started":"2023-12-28T00:49:02.758056Z","shell.execute_reply":"2023-12-28T00:49:02.762737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class TmaDataset(Dataset):\n    def __init__(self, df, transforms=None):\n        self.img_paths = df[\"image_pth\"].values\n        self.transforms = transforms\n        self.crop_size = 2000\n        \n    def __len__(self):\n        return len(self.img_paths)\n    \n    def __getitem__(self, index):\n        \n        img_path = self.img_paths[index]\n        img = cv2.imread(str(img_path))\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        \n        w, h, _ = img.shape\n        left = (w - self.crop_size) // 2\n        upper = (h - self.crop_size) // 2\n        crop_img = img[left:left+self.crop_size, upper:upper+self.crop_size]\n        \n        if self.transforms:\n            img = self.transforms(image=img)[\"image\"]\n            \n        return {\n            \"image\": img,\n        }\n    \ndef infer_tma(df):\n    tma_probs = []\n    for fold in range(5):\n        test_data = TmaDataset(df, transforms=get_transforms())\n        test_loader = DataLoader(test_data, batch_size=8, \n                                    num_workers=4, shuffle=False, pin_memory=True)\n        weight = f\"/kaggle/input/ubc-ocean/exp-15-tma-center-crop/exp-15-tma-center-crop-fold-{fold}.pth\"\n        tma_model = UBCModel(CFG.clc_model_name, CFG.num_classes, pretrained=False)\n        tma_model.eval()\n        tma_model.to(device)\n        tma_model.load_state_dict(torch.load(weight))\n        print(\"tam weights loaded\")\n        probs = infer_clc_tma(test_loader, tma_model)\n        \n        tma_probs.append(probs)\n\n\n    tma_probs = np.array(tma_probs)\n    tma_probs = np.mean(tma_probs, axis =2)\n\n    \n    df[\"CC\"] = probs[:,0]\n    df[\"EC\"] = probs[:,1]\n    df[\"HGSC\"] = probs[:,2]\n    df[\"LGSC\"] = probs[:,3]\n    df[\"MC\"] = probs[:,4]\n    \n    selected_columns = ['CC', 'EC', 'HGSC', 'LGSC', 'MC']\n    df['label'] = df[selected_columns].idxmax(axis=1)\n    df['max_value'] = df[selected_columns].max(axis=1)\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-12-28T00:49:02.989009Z","iopub.execute_input":"2023-12-28T00:49:02.989879Z","iopub.status.idle":"2023-12-28T00:49:03.003057Z","shell.execute_reply.started":"2023-12-28T00:49:02.989846Z","shell.execute_reply":"2023-12-28T00:49:03.002078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if len(tma_df) > 0:\n    tma_result = infer_tma(tma_df)","metadata":{"execution":{"iopub.status.busy":"2023-12-28T00:49:03.372969Z","iopub.execute_input":"2023-12-28T00:49:03.373881Z","iopub.status.idle":"2023-12-28T00:49:18.093081Z","shell.execute_reply.started":"2023-12-28T00:49:03.373847Z","shell.execute_reply":"2023-12-28T00:49:18.091833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"seg_model = smp.Unet(\n    encoder_name=CFG.seg_model_name,\n    encoder_weights=None,\n    in_channels=3,\n    classes=1,\n    activation=None,\n)\n\nseg_model.to(device)\nseg_model.eval()\nseg_model.load_state_dict(torch.load(CFG.seg_model_weight))\nprint(\"seg_model_weight loaded\")\n\n\nclc_model = UBCModel(CFG.clc_model_name, CFG.num_classes, pretrained=False)\nclc_model.eval()\nclc_model.to(device)\nclc_model.load_state_dict(torch.load(\"/kaggle/input/ubc-ocean/exp-15-fold-0.pth\"))\nprint(\"clc weights loaded\")\n\n\nclc_model2 = UBCModel(CFG.clc_model_name, CFG.num_classes, pretrained=False)\nclc_model2.eval()\nclc_model2.to(device)\nclc_model2.load_state_dict(torch.load(\"/kaggle/input/ubc-ocean/exp-15-fold-1.pth\"))\nprint(\"clc weights loaded\")","metadata":{"execution":{"iopub.status.busy":"2023-12-28T00:49:24.47399Z","iopub.execute_input":"2023-12-28T00:49:24.474718Z","iopub.status.idle":"2023-12-28T00:49:27.582254Z","shell.execute_reply.started":"2023-12-28T00:49:24.474682Z","shell.execute_reply":"2023-12-28T00:49:27.581364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_image(row, data_dir, target_path):\n    tma = row[\"is_tma\"]\n    image_id = row[\"image_id\"]\n\n    if Debug:\n        img_path = str(data_dir / \"train_images\" / f\"{row.image_id}.png\")\n    else:\n        img_path = str(data_dir / \"test_images\" / f\"{row.image_id}.png\")\n\n    new_size = (512, 512)\n    tile_size = 2048\n    \n    if tma:\n        pass\n    else:\n        img = cv2.imread(img_path)  # Load image using OpenCV\n        height, width = img.shape[:2]  # Get height and width using OpenCV\n\n        for y in range(0, height, tile_size):\n            for x in range(0, width, tile_size):\n                # Extract a tile from the image using OpenCV\n                img_tile = img[y:y+min(tile_size, height-y), x:x+min(tile_size, width-x)]\n\n                gray = cv2.cvtColor(img_tile, cv2.COLOR_BGR2GRAY)\n                _, binary_image = cv2.threshold(gray, 1, 255, cv2.THRESH_BINARY)\n\n                # Black area ratio threshold\n                black_pixels = np.count_nonzero(binary_image == 0)\n                total_pixels = np.prod(binary_image.shape[:2])\n                black_area_ratio = black_pixels / total_pixels\n\n                if black_area_ratio > 0.3:\n                    continue\n\n                tile_save_path = str(target_path / f\"{image_id}_tile_{x}_{y}.png\")\n                cv2.imwrite(tile_save_path, cv2.resize(img_tile, new_size, interpolation=cv2.INTER_LANCZOS4))","metadata":{"execution":{"iopub.status.busy":"2023-12-28T00:49:27.584147Z","iopub.execute_input":"2023-12-28T00:49:27.584769Z","iopub.status.idle":"2023-12-28T00:49:27.594842Z","shell.execute_reply.started":"2023-12-28T00:49:27.584734Z","shell.execute_reply":"2023-12-28T00:49:27.593864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def infer_not_tma(df):\n    \n    tile_result = []\n    for _, row in tqdm(df.iterrows(), total=len(df)):\n        target_path = Path(\"/tmp/output\")\n        target_path.mkdir(exist_ok=True)\n        tma = row[\"is_tma\"]\n\n        if tma:\n            continue\n        process_image(row, data_dir, target_path)\n\n\n        tile_files = glob(str(target_path / \"*.png\"))\n        tile_df = pd.DataFrame(tile_files, columns=[\"tile_path\"])\n        tile_df[\"image_id\"] = tile_df[\"tile_path\"].apply(lambda x: x.split(\"/\")[-1].split(\"_\")[0])\n\n        if tma:\n            pass\n        else:\n            seg_dataset = UBCDataset(tile_df, transforms=get_transforms())\n            test_loader = DataLoader(seg_dataset, batch_size=16, \n                                        num_workers=4, shuffle=False, pin_memory=True)\n\n            mask_ratio = seg_infer(test_loader=test_loader, model=seg_model)\n            tile_df[\"mask_ratio\"] = mask_ratio\n            tile_df = tile_df.groupby('image_id').apply(lambda x: x.sort_values(by='mask_ratio', ascending=False).head(10))\n            tile_df = tile_df.reset_index(drop=True)\n\n            clc_dataset = UBCDataset(tile_df, transforms=get_transforms())\n            test_loader = DataLoader(clc_dataset, batch_size=8, \n                                        num_workers=4, shuffle=False, pin_memory=True)\n\n            tile_df_pred1 = infer_clc_not_tma(test_loader, clc_model, tile_df)\n            tile_result.append(tile_df_pred1)\n            tile_df_pred2 = infer_clc_not_tma(test_loader, clc_model2, tile_df)\n            tile_result.append(tile_df_pred2)\n            shutil.rmtree(\"/tmp/output\")\n\n    tile_result = pd.concat(tile_result)\n    cc_df = tile_result.groupby('image_id')['CC'].mean().reset_index()\n    ec_df = tile_result.groupby('image_id')['EC'].mean().reset_index()\n    hgsc_df = tile_result.groupby('image_id')['HGSC'].mean().reset_index()\n    lgsc_df = tile_result.groupby('image_id')['LGSC'].mean().reset_index()\n    mc_df = tile_result.groupby('image_id')['MC'].mean().reset_index()\n\n\n    tile_result = pd.merge(cc_df, ec_df, on='image_id', how='inner')\n    tile_result = pd.merge(tile_result, hgsc_df, on='image_id', how='inner')\n    tile_result = pd.merge(tile_result, lgsc_df, on='image_id', how='inner')\n    tile_result = pd.merge(tile_result, mc_df, on='image_id', how='inner')\n\n    selected_columns = ['CC', 'EC', 'HGSC', 'LGSC', 'MC']\n    # Find the column with the maximum value for each row\n    tile_result['label'] = tile_result[selected_columns].idxmax(axis=1)\n\n    # Find the maximum value for each row\n    tile_result['max_value'] = tile_result[selected_columns].max(axis=1)\n    \n    return tile_result","metadata":{"execution":{"iopub.status.busy":"2023-12-28T00:49:27.596197Z","iopub.execute_input":"2023-12-28T00:49:27.59647Z","iopub.status.idle":"2023-12-28T00:49:27.612201Z","shell.execute_reply.started":"2023-12-28T00:49:27.596447Z","shell.execute_reply":"2023-12-28T00:49:27.611391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"not_tma_result = infer_not_tma(not_tma_df)","metadata":{"execution":{"iopub.status.busy":"2023-12-28T00:49:28.162825Z","iopub.execute_input":"2023-12-28T00:49:28.163941Z","iopub.status.idle":"2023-12-28T00:56:27.462698Z","shell.execute_reply.started":"2023-12-28T00:49:28.163901Z","shell.execute_reply":"2023-12-28T00:56:27.461508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if len(tma_df)>0:\n    sub_df = pd.concat([tma_result, not_tma_result])\n    sub_df\nelse:\n    sub_df = not_tma_result","metadata":{"execution":{"iopub.status.busy":"2023-12-28T00:56:27.464878Z","iopub.execute_input":"2023-12-28T00:56:27.465219Z","iopub.status.idle":"2023-12-28T00:56:27.474273Z","shell.execute_reply.started":"2023-12-28T00:56:27.465187Z","shell.execute_reply":"2023-12-28T00:56:27.473348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df","metadata":{"execution":{"iopub.status.busy":"2023-12-28T00:56:27.475501Z","iopub.execute_input":"2023-12-28T00:56:27.475786Z","iopub.status.idle":"2023-12-28T00:56:27.650625Z","shell.execute_reply.started":"2023-12-28T00:56:27.475762Z","shell.execute_reply":"2023-12-28T00:56:27.649733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"condition = sub_df['max_value'] < 0.1\nsub_df.loc[condition, 'label'] = \"Other\" ","metadata":{"execution":{"iopub.status.busy":"2023-12-28T00:56:27.652656Z","iopub.execute_input":"2023-12-28T00:56:27.652961Z","iopub.status.idle":"2023-12-28T00:56:27.659198Z","shell.execute_reply.started":"2023-12-28T00:56:27.652937Z","shell.execute_reply":"2023-12-28T00:56:27.658192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df[[\"image_id\", \"label\"]].to_csv(\"submission.csv\", index=False)\nsub_df[[\"image_id\", \"label\"]]","metadata":{"execution":{"iopub.status.busy":"2023-12-28T00:56:27.660225Z","iopub.execute_input":"2023-12-28T00:56:27.660514Z","iopub.status.idle":"2023-12-28T00:56:27.678123Z","shell.execute_reply.started":"2023-12-28T00:56:27.660482Z","shell.execute_reply":"2023-12-28T00:56:27.677358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}