{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"},{"sourceId":6774400,"sourceType":"datasetVersion","datasetId":3895136},{"sourceId":6774553,"sourceType":"datasetVersion","datasetId":3898019},{"sourceId":6947868,"sourceType":"datasetVersion","datasetId":3958714},{"sourceId":6957918,"sourceType":"datasetVersion","datasetId":3886036}],"dockerImageVersionId":30587,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"#### import torch\n\nfrom IPython.display import clear_output\n\n\n                                                   \n\n\n\nfrom PIL import Image\nImage.MAX_IMAGE_PIXELS = None\nimport pandas as pd \nimport numpy as np\nimport gc\nimport math\n\nfrom collections import Counter   ","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-12-02T06:11:07.278889Z","iopub.execute_input":"2023-12-02T06:11:07.279278Z","iopub.status.idle":"2023-12-02T06:11:12.086404Z","shell.execute_reply.started":"2023-12-02T06:11:07.279243Z","shell.execute_reply":"2023-12-02T06:11:12.084952Z"}}},{"cell_type":"code","source":"import torch\nfrom IPython.display import clear_output\n\nfrom PIL import Image \nImage.MAX_IMAGE_PIXELS = None \nimport pandas as pd\nimport numpy as np\nimport gc\nimport math\n\nfrom collections import Counter","metadata":{"execution":{"iopub.status.busy":"2023-12-04T09:34:02.43745Z","iopub.execute_input":"2023-12-04T09:34:02.43789Z","iopub.status.idle":"2023-12-04T09:34:06.040167Z","shell.execute_reply.started":"2023-12-04T09:34:02.437843Z","shell.execute_reply":"2023-12-04T09:34:06.038873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls /kaggle/input/pyvips-python-and-deb-package\n# intall the deb packages\n!yes | dpkg -i --force-depends /kaggle/input/pyvips-python-and-deb-package/linux_packages/archives/*.deb\n# install the python wrapper\n!pip install pyvips -f /kaggle/input/pyvips-python-and-deb-package/python_packages/ --no-index\n!pip list | grep pyvips\n\n\nfrom IPython import display\ndisplay.clear_output()","metadata":{"execution":{"iopub.status.busy":"2023-12-04T09:34:06.042272Z","iopub.execute_input":"2023-12-04T09:34:06.042779Z","iopub.status.idle":"2023-12-04T09:35:48.67492Z","shell.execute_reply.started":"2023-12-04T09:34:06.042746Z","shell.execute_reply":"2023-12-04T09:35:48.673422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport pyvips\nimport numpy as np\nimport random\nfrom PIL import Image\nfrom IPython import display\nfrom tqdm import tqdm\ndef extract_image_tiles(\n    p_img,label, folder, size: int = 2048, scale: float = 0.5,\n    drop_thr: float = 0.6, white_thr: int = 240, max_samples: int = 50\n) -> list:\n    name, _ = os.path.splitext(os.path.basename(p_img))\n    im = pyvips.Image.new_from_file(p_img)\n    w = h = size\n    print(f\"processing: {p_img}\")\n    # https://stackoverflow.com/a/47581978/4521646\n    idxs = [(y, y + h, x, x + w) for y in range(0, im.height, h) for x in range(0, im.width, w)]\n    # random subsample\n    max_samples = max_samples if isinstance(max_samples, int) else int(len(idxs) * max_samples)\n    random.shuffle(idxs)\n    files = []\n    imageeslist=[]\n    for y, y_, x, x_ in idxs:        # https://libvips.github.io/pyvips/vimage.html#pyvips.Image.crop\n\n    #for y, y_, x, x_ in tqdm(idxs, total=len(idxs)):        # https://libvips.github.io/pyvips/vimage.html#pyvips.Image.crop\n        tile = im.crop(x, y, min(w, im.width - x), min(h, im.height - y)).numpy()[..., :3]\n        if tile.shape[:2] != (h, w):\n            tile_ = tile\n            tile_size = (h, w) if tile.ndim == 2 else (h, w, tile.shape[2])\n            tile = np.zeros(tile_size, dtype=tile.dtype)\n            tile[:tile_.shape[0], :tile_.shape[1], ...] = tile_\n        black_bg = np.sum(tile, axis=2) == 0\n        tile[black_bg, :] = 255\n        img = np.dot(tile[..., :3], [0.2989, 0.5870, 0.1140]).astype(np.uint8)\n        white_pixels = np.sum(img>220)\n#         mask_bg = np.mean(tile, axis=2) > white_thr\n        if np.sum(white_pixels) >= (np.prod(img.shape) * drop_thr):\n#             display.clear_output()\n#             plt.imshow(tile)\n#             plt.show()\n            continue\n        imageeslist.append(tile)\n#         p_img = os.path.join(folder, f\"label-{label}-{int(x_ / w)}-{int(y_ / h)}.png\")\n#         # print(tile.shape, tile.dtype, tile.min(), tile.max())\n# #         new_size = int(size * scale), int(size * scale)\n#          #Image.fromarray(tile).resize(new_size, Image.LANCZOS).save(p_img)\n#         Image.fromarray(tile).save(p_img)\n\n#         files.append(p_img)\n#         # need to set counter check as some empty tiles could be skipped earlier\n        if len(imageeslist) >= max_samples:\n            break\n    return imageeslist","metadata":{"execution":{"iopub.status.busy":"2023-12-04T09:46:52.23754Z","iopub.execute_input":"2023-12-04T09:46:52.238022Z","iopub.status.idle":"2023-12-04T09:46:52.257138Z","shell.execute_reply.started":"2023-12-04T09:46:52.237987Z","shell.execute_reply":"2023-12-04T09:46:52.255571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# imagees_list = extract_prune_tiles(\"/kaggle/input/UBC-OCEAN/test_images/41.png\",0, IMAGES_FOLDER, size=256, scale=1,drop_thr=0.4,white_thr=222,max_samples=10)","metadata":{"execution":{"iopub.status.busy":"2023-12-04T09:46:52.370412Z","iopub.execute_input":"2023-12-04T09:46:52.370886Z","iopub.status.idle":"2023-12-04T09:46:52.375842Z","shell.execute_reply.started":"2023-12-04T09:46:52.370851Z","shell.execute_reply":"2023-12-04T09:46:52.374943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os, glob\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport re\n\nDATASET_FOLDER = \"/kaggle/input/UBC-OCEAN/\"\nIMAGES_FOLDER = \"./test_tiles\"\n\nos.environ['VIPS_CONCURRENCY'] = '4'\nos.environ['VIPS_DISC_THRESHOLD'] = '15gb'\nimport torch\nfrom PIL import Image\nfrom torch.utils.data import Dataset\n\nclass TilesFolderDataset(Dataset):\n\n    def __init__(\n        self,\n        folder: str,\n        image_ext: str =  '.png',\n        transforms = None\n    ):\n        assert os.path.isdir(folder)\n        self.transforms = transforms\n        self.imgs = glob.glob(os.path.join(folder, \"*\" + image_ext))\n\n    def __getitem__(self, idx: int) -> tuple:\n        img_path = self.imgs[idx]\n        assert os.path.isfile(img_path), f\"missing: {img_path}\"\n        img = np.array(Image.open(img_path))[..., :3]\n        # filter background\n        mask = np.sum(img, axis=2) == 0\n        img[mask, :] = 255\n        if np.max(img) < 1.5:\n            img = np.clip(img * 255, 0, 255).astype(np.uint8)\n        # augmentation\n        if self.transforms:\n            img = self.transforms(Image.fromarray(img))\n        #print(f\"img dim: {img.shape}\")\n        return img\n\n    def __len__(self) -> int:\n        return len(self.imgs)","metadata":{"execution":{"iopub.status.busy":"2023-12-04T09:46:52.542078Z","iopub.execute_input":"2023-12-04T09:46:52.543297Z","iopub.status.idle":"2023-12-04T09:46:52.555334Z","shell.execute_reply.started":"2023-12-04T09:46:52.54326Z","shell.execute_reply":"2023-12-04T09:46:52.553957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def extract_prune_tiles(\n    path_img: str,label:str, folder: str, size: int = 2048, scale: float = 0.25,\n    drop_thr: float = 0.6,white_thr: int=0.5, max_samples: int = 1000\n) -> str:\n    print(f\"processing: {path_img}\")\n    name, _ = os.path.splitext(os.path.basename(path_img))\n    folder = os.path.join(folder, name)\n    os.makedirs(folder, exist_ok=True)\n    tiles = extract_image_tiles(\n        path_img,label, folder, size=size, scale=scale,\n        drop_thr=drop_thr,white_thr=225 ,max_samples=max_samples)\n    return tiles","metadata":{"execution":{"iopub.status.busy":"2023-12-04T09:46:52.676242Z","iopub.execute_input":"2023-12-04T09:46:52.677488Z","iopub.status.idle":"2023-12-04T09:46:52.684582Z","shell.execute_reply.started":"2023-12-04T09:46:52.677448Z","shell.execute_reply":"2023-12-04T09:46:52.68376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# submission_df = pd.read_csv(\"/kaggle/input/UBC-OCEAN/sample_submission.csv\")\n# test_df=pd.read_csv('/kaggle/input/UBC-OCEAN/test.csv')\n","metadata":{"execution":{"iopub.status.busy":"2023-12-04T09:46:53.019667Z","iopub.execute_input":"2023-12-04T09:46:53.020086Z","iopub.status.idle":"2023-12-04T09:46:53.025337Z","shell.execute_reply.started":"2023-12-04T09:46:53.020055Z","shell.execute_reply":"2023-12-04T09:46:53.023859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-04T09:46:53.213119Z","iopub.execute_input":"2023-12-04T09:46:53.213538Z","iopub.status.idle":"2023-12-04T09:46:53.2188Z","shell.execute_reply.started":"2023-12-04T09:46:53.213508Z","shell.execute_reply":"2023-12-04T09:46:53.217786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os, glob\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport re\n\nDATASET_FOLDER = \"/kaggle/input/UBC-OCEAN/\"\nIMAGES_FOLDER = \"./test_tiles\"\n\nos.environ['VIPS_CONCURRENCY'] = '4'\nos.environ['VIPS_DISC_THRESHOLD'] = '15gb'","metadata":{"execution":{"iopub.status.busy":"2023-12-04T09:46:53.360626Z","iopub.execute_input":"2023-12-04T09:46:53.361055Z","iopub.status.idle":"2023-12-04T09:46:53.367649Z","shell.execute_reply.started":"2023-12-04T09:46:53.361024Z","shell.execute_reply":"2023-12-04T09:46:53.366624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df=pd.read_csv('/kaggle/input/UBC-OCEAN/train.csv')\ntest_df=pd.read_csv('/kaggle/input/UBC-OCEAN/test.csv')\n# submission_df = pd.read_csv(\"/kaggle/input/UBC-OCEAN/sample_submission.csv\")\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-04T09:46:53.888919Z","iopub.execute_input":"2023-12-04T09:46:53.889324Z","iopub.status.idle":"2023-12-04T09:46:53.907542Z","shell.execute_reply.started":"2023-12-04T09:46:53.889293Z","shell.execute_reply":"2023-12-04T09:46:53.906227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\n\n\n# for idx,row in test_df.iterrows():\n#     path=f\"/kaggle/input/UBC-OCEAN/test_images/{row['image_id']}.png\"\n#     label=row['image_id']\n#     folder_tiles = extract_prune_tiles(path,label, IMAGES_FOLDER, size=256, scale=1,drop_thr=0.4,white_thr=222)\n    ","metadata":{"execution":{"iopub.status.busy":"2023-12-04T09:46:54.101497Z","iopub.execute_input":"2023-12-04T09:46:54.101964Z","iopub.status.idle":"2023-12-04T09:46:54.106696Z","shell.execute_reply.started":"2023-12-04T09:46:54.101906Z","shell.execute_reply":"2023-12-04T09:46:54.10557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Defining model","metadata":{}},{"cell_type":"code","source":"from functools import partial\nfrom typing import Any, Callable, List, Optional, Type, Union\n\nimport torch\nimport torch.nn as nn\nfrom torch import Tensor\n\n\n\n\n\n\ndef conv3x3(in_planes: int, out_planes: int, stride: int = 1, groups: int = 1, dilation: int = 1) -> nn.Conv2d:\n    \"\"\"3x3 convolution with padding\"\"\"\n    return nn.Conv2d(\n        in_planes,\n        out_planes,\n        kernel_size=3,\n        stride=stride,\n        padding=dilation,\n        groups=groups,\n        bias=False,\n        dilation=dilation,\n    )\n\n\ndef conv1x1(in_planes: int, out_planes: int, stride: int = 1) -> nn.Conv2d:\n    \"\"\"1x1 convolution\"\"\"\n    return nn.Conv2d(in_planes, out_planes, kernel_size=1, stride=stride, bias=False)\n\n\nclass BasicBlock(nn.Module):\n    expansion: int = 1\n\n    def __init__(\n        self,\n        inplanes: int,\n        planes: int,\n        stride: int = 1,\n        downsample: Optional[nn.Module] = None,\n        groups: int = 1,\n        base_width: int = 64,\n        dilation: int = 1,\n        norm_layer: Optional[Callable[..., nn.Module]] = None,\n    ) -> None:\n        super().__init__()\n        if norm_layer is None:\n            norm_layer = nn.BatchNorm2d\n        if groups != 1 or base_width != 64:\n            raise ValueError(\"BasicBlock only supports groups=1 and base_width=64\")\n        if dilation > 1:\n            raise NotImplementedError(\"Dilation > 1 not supported in BasicBlock\")\n        # Both self.conv1 and self.downsample layers downsample the input when stride != 1\n        self.conv1 = conv3x3(inplanes, planes, stride)\n        self.bn1 = norm_layer(planes)\n        self.relu = nn.ReLU(inplace=True)\n        self.conv2 = conv3x3(planes, planes)\n        self.bn2 = norm_layer(planes)\n        self.downsample = downsample\n        self.stride = stride\n\n    def forward(self, x: Tensor) -> Tensor:\n        identity = x\n\n        out = self.conv1(x)\n        out = self.bn1(out)\n        out = self.relu(out)\n\n        out = self.conv2(out)\n        out = self.bn2(out)\n\n        if self.downsample is not None:\n            identity = self.downsample(x)\n\n        out += identity\n        out = self.relu(out)\n\n        return out\n\n\nclass Bottleneck(nn.Module):\n    # Bottleneck in torchvision places the stride for downsampling at 3x3 convolution(self.conv2)\n    # while original implementation places the stride at the first 1x1 convolution(self.conv1)\n    # according to \"Deep residual learning for image recognition\" https://arxiv.org/abs/1512.03385.\n    # This variant is also known as ResNet V1.5 and improves accuracy according to\n    # https://ngc.nvidia.com/catalog/model-scripts/nvidia:resnet_50_v1_5_for_pytorch.\n\n    expansion: int = 4\n\n    def __init__(\n        self,\n        inplanes: int,\n        planes: int,\n        stride: int = 1,\n        downsample: Optional[nn.Module] = None,\n        groups: int = 1,\n        base_width: int = 64,\n        dilation: int = 1,\n        norm_layer: Optional[Callable[..., nn.Module]] = None,\n    ) -> None:\n        super().__init__()\n        if norm_layer is None:\n            norm_layer = nn.BatchNorm2d\n        width = int(planes * (base_width / 64.0)) * groups\n        # Both self.conv2 and self.downsample layers downsample the input when stride != 1\n        self.conv1 = conv1x1(inplanes, width)\n        self.bn1 = norm_layer(width)\n        self.conv2 = conv3x3(width, width, stride, groups, dilation)\n        self.bn2 = norm_layer(width)\n        self.conv3 = conv1x1(width, planes * self.expansion)\n        self.bn3 = norm_layer(planes * self.expansion)\n        self.relu = nn.ReLU(inplace=True)\n        self.downsample = downsample\n        self.stride = stride\n\n    def forward(self, x: Tensor) -> Tensor:\n        identity = x\n\n        out = self.conv1(x)\n        out = self.bn1(out)\n        out = self.relu(out)\n\n        out = self.conv2(out)\n        out = self.bn2(out)\n        out = self.relu(out)\n\n        out = self.conv3(out)\n        out = self.bn3(out)\n\n        if self.downsample is not None:\n            identity = self.downsample(x)\n\n        out += identity\n        out = self.relu(out)\n\n        return out\n\n\nclass ResNet(nn.Module):\n    def __init__(\n        self,\n        block: Type[Union[BasicBlock, Bottleneck]],\n        layers: List[int],\n        num_classes: int = 1000,\n        zero_init_residual: bool = False,\n        groups: int = 1,\n        width_per_group: int = 64,\n        replace_stride_with_dilation: Optional[List[bool]] = None,\n        norm_layer: Optional[Callable[..., nn.Module]] = None,\n    ) -> None:\n        super().__init__()\n        if norm_layer is None:\n            norm_layer = nn.BatchNorm2d\n        self._norm_layer = norm_layer\n\n        self.inplanes = 64\n        self.dilation = 1\n        if replace_stride_with_dilation is None:\n            # each element in the tuple indicates if we should replace\n            # the 2x2 stride with a dilated convolution instead\n            replace_stride_with_dilation = [False, False, False]\n        if len(replace_stride_with_dilation) != 3:\n            raise ValueError(\n                \"replace_stride_with_dilation should be None \"\n                f\"or a 3-element tuple, got {replace_stride_with_dilation}\"\n            )\n        self.groups = groups\n        self.base_width = width_per_group\n        self.conv1 = nn.Conv2d(3, self.inplanes, kernel_size=7, stride=2, padding=3, bias=False)\n        self.bn1 = norm_layer(self.inplanes)\n        self.relu = nn.ReLU(inplace=True)\n        self.maxpool = nn.MaxPool2d(kernel_size=3, stride=2, padding=1)\n        self.layer1 = self._make_layer(block, 64, layers[0])\n        self.layer2 = self._make_layer(block, 128, layers[1], stride=2, dilate=replace_stride_with_dilation[0])\n        self.layer3 = self._make_layer(block, 256, layers[2], stride=2, dilate=replace_stride_with_dilation[1])\n        self.layer4 = self._make_layer(block, 512, layers[3], stride=2, dilate=replace_stride_with_dilation[2])\n        self.avgpool = nn.AdaptiveAvgPool2d((1, 1))\n        self.fc = nn.Linear(512 * block.expansion, num_classes)\n\n        for m in self.modules():\n            if isinstance(m, nn.Conv2d):\n                nn.init.kaiming_normal_(m.weight, mode=\"fan_out\", nonlinearity=\"relu\")\n            elif isinstance(m, (nn.BatchNorm2d, nn.GroupNorm)):\n                nn.init.constant_(m.weight, 1)\n                nn.init.constant_(m.bias, 0)\n\n        # Zero-initialize the last BN in each residual branch,\n        # so that the residual branch starts with zeros, and each residual block behaves like an identity.\n        # This improves the model by 0.2~0.3% according to https://arxiv.org/abs/1706.02677\n        if zero_init_residual:\n            for m in self.modules():\n                if isinstance(m, Bottleneck) and m.bn3.weight is not None:\n                    nn.init.constant_(m.bn3.weight, 0)  # type: ignore[arg-type]\n                elif isinstance(m, BasicBlock) and m.bn2.weight is not None:\n                    nn.init.constant_(m.bn2.weight, 0)  # type: ignore[arg-type]\n\n    def _make_layer(\n        self,\n        block: Type[Union[BasicBlock, Bottleneck]],\n        planes: int,\n        blocks: int,\n        stride: int = 1,\n        dilate: bool = False,\n    ) -> nn.Sequential:\n        norm_layer = self._norm_layer\n        downsample = None\n        previous_dilation = self.dilation\n        if dilate:\n            self.dilation *= stride\n            stride = 1\n        if stride != 1 or self.inplanes != planes * block.expansion:\n            downsample = nn.Sequential(\n                conv1x1(self.inplanes, planes * block.expansion, stride),\n                norm_layer(planes * block.expansion),\n            )\n\n        layers = []\n        layers.append(\n            block(\n                self.inplanes, planes, stride, downsample, self.groups, self.base_width, previous_dilation, norm_layer\n            )\n        )\n        self.inplanes = planes * block.expansion\n        for _ in range(1, blocks):\n            layers.append(\n                block(\n                    self.inplanes,\n                    planes,\n                    groups=self.groups,\n                    base_width=self.base_width,\n                    dilation=self.dilation,\n                    norm_layer=norm_layer,\n                )\n            )\n\n        return nn.Sequential(*layers)\n\n    def _forward_impl(self, x: Tensor) -> Tensor:\n        # See note [TorchScript super()]\n        x = self.conv1(x)\n        x = self.bn1(x)\n        x = self.relu(x)\n        x = self.maxpool(x)\n\n        x = self.layer1(x)\n        x = self.layer2(x)\n        x = self.layer3(x)\n        x = self.layer4(x)\n\n        x = self.avgpool(x)\n        x = torch.flatten(x, 1)\n        x = self.fc(x)\n\n        return x\n\n    def forward(self, x: Tensor) -> Tensor:\n        return self._forward_impl(x)","metadata":{"execution":{"iopub.status.busy":"2023-12-04T09:46:54.423055Z","iopub.execute_input":"2023-12-04T09:46:54.423496Z","iopub.status.idle":"2023-12-04T09:46:54.47144Z","shell.execute_reply.started":"2023-12-04T09:46:54.42346Z","shell.execute_reply":"2023-12-04T09:46:54.470197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"infer code","metadata":{}},{"cell_type":"code","source":"device='cuda' if torch.cuda.is_available() else 'cpu'\nmodel=ResNet(Bottleneck, [3, 4, 6, 3], 6).to(device)\nnum_features = model.fc.in_features\nmodel.fc = nn.Linear(num_features, 6) \nstate_dict=torch.load(\"/kaggle/input/first-resnet-model/resnetv4.pth\",map_location=torch.device('cpu'))\nmodel.load_state_dict(state_dict[\"model_state_dict\"])","metadata":{"execution":{"iopub.status.busy":"2023-12-04T09:46:55.736102Z","iopub.execute_input":"2023-12-04T09:46:55.737543Z","iopub.status.idle":"2023-12-04T09:46:56.344694Z","shell.execute_reply.started":"2023-12-04T09:46:55.737486Z","shell.execute_reply":"2023-12-04T09:46:56.343692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from torchvision import transforms\n","metadata":{"execution":{"iopub.status.busy":"2023-12-04T09:46:56.346658Z","iopub.execute_input":"2023-12-04T09:46:56.347046Z","iopub.status.idle":"2023-12-04T09:46:56.352268Z","shell.execute_reply.started":"2023-12-04T09:46:56.347014Z","shell.execute_reply":"2023-12-04T09:46:56.350993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_df=pd.read_csv(\"/kaggle/input/UBC-OCEAN/train.csv\").sample(3)\n# test_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-04T10:13:40.3135Z","iopub.execute_input":"2023-12-04T10:13:40.31398Z","iopub.status.idle":"2023-12-04T10:13:40.333832Z","shell.execute_reply.started":"2023-12-04T10:13:40.313932Z","shell.execute_reply":"2023-12-04T10:13:40.332484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import shutil\ntransform = transforms.Compose([transforms.ToTensor(),\n                               transforms.Normalize(mean = [0.485, 0.456, 0.406], \n                                                    std = [0.229, 0.224, 0.225])])\nimagelist=[]\nnewlist=[]\nanswer=[]\ndef extract_features(image_path):\n    image=Image.fromarray(image_path)\n    #image = image.convert('RGB')\n    image_tensor = transform(image).unsqueeze(0).to(device)\n#     print(image_tensor)# Add batch dimension\n    with torch.no_grad():\n        output = model(image_tensor)\n        \n#         probabilities = torch.nn.functional.softmax(output, dim=1)  # Apply softmax\n        \n    return output.squeeze().cpu().numpy()\n\n\nfor idx,row in test_df.iterrows():\n    \n    desired_path=f\"/kaggle/input/UBC-OCEAN/test_images/{row['image_id']}.png\"\n    if os.path.exists(desired_path):\n        path = desired_path\n    else:\n        path=f\"/kaggle/input/UBC-OCEAN/train_images/{row['image_id']}.png\"\n        #path=f\"/kaggle/input/UBC-OCEAN/test_thumbnails/{row['image_id']}.png\"\n    label=row['image_id']\n    #imagees_list = extract_prune_tiles(path,label, IMAGES_FOLDER, size=256, scale=1,drop_thr=0.4,white_thr=222,max_samples=10)\n    imagees_list=extract_image_tiles( path,label, IMAGES_FOLDER, size=256, scale=1,drop_thr=0.4,white_thr=222,max_samples=10)\n    \n#     imagefolderpath=f\"/kaggle/working/test_tiles/{row['image_id']}\"\n#     image_list=[]\n#     image_list= glob.glob(os.path.join(imagefolderpath,\"*.png\"))\n    print(\"images finished processing\")\n    results=[] \n    #for idx,imagepath in tqdm(enumerate(image_list),total=1000):\n    for idx1,imagepath in enumerate(imagees_list):\n \n\n            probabilities = extract_features(imagepath)\n            \n            result=np.argmax(probabilities)\n            \n#             if (idx>300):\n           \n            \n\n            \n            results.append(result.item())\n            \n    print(results)\n    element_counts = Counter(results)\n    most_common_element = element_counts.most_common(1)[0][0]\n    answer.append(most_common_element)\n    imagelist.append(row['image_id'])\n\n#     newlist.append(new_data)\n#     if idx > 3:\n#         break\n\n \nnew_data = {\n             'image_id': imagelist,\n             'label': answer\n                    }  \n  \nsubmission_df= pd.DataFrame(new_data)\nlabel_to_be_replace=['CC','EC','HGSC','LGSC','MC','Others']\nlabel_to_be_replaced=[0,1,2,3,4,5]\nsubmission_df['label'].replace(label_to_be_replaced,label_to_be_replace,inplace=True)\n        # Save the DataFrame to a CSV file named 'submission.csv'\nsubmission_df.to_csv('submission.csv', index=False)\n        \n        # Print the DataFrame\nprint(submission_df)\n    ","metadata":{"execution":{"iopub.status.busy":"2023-12-04T10:13:47.283363Z","iopub.execute_input":"2023-12-04T10:13:47.283834Z","iopub.status.idle":"2023-12-04T10:16:31.00558Z","shell.execute_reply.started":"2023-12-04T10:13:47.283798Z","shell.execute_reply":"2023-12-04T10:16:31.004382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n#     print(len(imagees_list))","metadata":{"execution":{"iopub.status.busy":"2023-12-04T09:47:12.665535Z","iopub.status.idle":"2023-12-04T09:47:12.666616Z","shell.execute_reply.started":"2023-12-04T09:47:12.666369Z","shell.execute_reply":"2023-12-04T09:47:12.666393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from matplotlib import pyplot as plt\n# for i in range(15):\n#     plt.imshow(imagees_list[i])\n#     plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-04T09:52:38.971331Z","iopub.execute_input":"2023-12-04T09:52:38.971755Z","iopub.status.idle":"2023-12-04T09:52:42.628738Z","shell.execute_reply.started":"2023-12-04T09:52:38.971723Z","shell.execute_reply":"2023-12-04T09:52:42.627494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(imagees_list)","metadata":{"execution":{"iopub.status.busy":"2023-12-04T09:40:19.914661Z","iopub.status.idle":"2023-12-04T09:40:19.915613Z","shell.execute_reply.started":"2023-12-04T09:40:19.915305Z","shell.execute_reply":"2023-12-04T09:40:19.915334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import shutil\n\n# folder_path = '/kaggle/working/test_tiles/41'\n\n# # Check if the folder exists before removing it\n# if shutil.os.path.exists(folder_path) and shutil.os.path.isdir(folder_path):\n#     shutil.rmtree(folder_path)\n#     print(f\"Folder '{folder_path}' and its contents have been successfully removed.\")\n# else:\n#     print(f\"Folder '{folder_path}' does not exist or is not a directory.\")","metadata":{"execution":{"iopub.status.busy":"2023-12-04T09:40:19.917424Z","iopub.status.idle":"2023-12-04T09:40:19.918616Z","shell.execute_reply.started":"2023-12-04T09:40:19.918314Z","shell.execute_reply":"2023-12-04T09:40:19.918343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"code for mixing code with tsne \n","metadata":{}},{"cell_type":"code","source":"\n\n# # Load your DataFrame containing file paths and labels of images into 'newpatched_df'\n# # Example: newpatched_df = pd.read_csv('your_dataframe.csv')\n# device = 'cuda' if torch.cuda.is_available() else 'cpu'\n\n# # Define transformation for the images (resize and normalize)\n# transform = transforms.Compose([\n#     transforms.Resize((224, 224)),\n#     transforms.ToTensor(),\n#     transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225]),\n# ])\n\n# # Load pre-trained ResNet model\n\n# model = model.eval().to(device)\n\n# # Function to extract features from an image file\n# def extract_features(image_path):\n#     image = Image.open(image_path).convert('RGB')\n#     image_tensor = transform(image).unsqueeze(0).to(device)  # Add batch dimension\n#     with torch.no_grad():\n#         output = model(image_tensor)\n# #         probabilities = torch.nn.functional.softmax(output, dim=1)  # Apply softmax\n        \n#     return output.squeeze().cpu().numpy()\n\n# # Extract features and labels from image paths in 'newpatched_df'\n# feature_vectors = []\n# labels = []  # Assuming 'labels' is a column in newpatched_df containing label information\n# for _, row in newpatched_df.iterrows():\n#     image_path = row['path']  # Assuming 'path' is the column name containing file paths\n#     label = row['label']  # Assuming 'label' is the column name containing label information\n#     probabilities = extract_features(image_path)\n#     print(np.argmax(probabilities),label)\n#     feature_vectors.append(probabilities)\n#     labels.append(label)\n\n# feature_vectors = np.array(feature_vectors)\n# labels = np.array(labels)\n\n# # Apply t-SNE on the feature vectors (3D visualization)\n# tsne = TSNE(n_components=3, random_state=42)\n# tsne_results = tsne.fit_transform(feature_vectors)\n\n# # Visualize 3D t-SNE results with a color map according to labels\n# fig = plt.figure(figsize=(10, 8))\n# ax = fig.add_subplot(111, projection='3d')\n# scatter = ax.scatter(tsne_results[:, 0], tsne_results[:, 1], tsne_results[:, 2], c=labels, cmap='viridis', marker='o', s=30)\n# ax.set_xlabel('t-SNE Component 1')\n# ax.set_ylabel('t-SNE Component 2')\n# ax.set_zlabel('t-SNE Component 3')\n# plt.title('3D t-SNE Visualization of Images with Softmax Applied to Model Output')\n# plt.colorbar(scatter)\n# plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-04T09:40:19.920408Z","iopub.status.idle":"2023-12-04T09:40:19.920927Z","shell.execute_reply.started":"2023-12-04T09:40:19.920668Z","shell.execute_reply":"2023-12-04T09:40:19.920692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !rm -r /kaggle/working/test_tiles/66","metadata":{"execution":{"iopub.status.busy":"2023-12-04T09:40:19.922698Z","iopub.status.idle":"2023-12-04T09:40:19.923647Z","shell.execute_reply.started":"2023-12-04T09:40:19.923377Z","shell.execute_reply":"2023-12-04T09:40:19.923403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}