{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"},{"sourceId":146934283,"sourceType":"kernelVersion"}],"dockerImageVersionId":30580,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Smart and Fast tiles","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"!yes | sudo dpkg -i /kaggle/input/libvips-pyvips-installation-and-getting-started/libvips/*.deb\n!pip install /kaggle/input/libvips-pyvips-installation-and-getting-started/pyvips/pyvips-2.2.1-py2.py3-none-any.whl --no-index --find-links /kaggle/input/libvips-pyvips-installation-and-getting-started/pyvips","metadata":{"execution":{"iopub.status.busy":"2023-11-17T01:36:05.732059Z","iopub.execute_input":"2023-11-17T01:36:05.732384Z","iopub.status.idle":"2023-11-17T01:36:40.527736Z","shell.execute_reply.started":"2023-11-17T01:36:05.732354Z","shell.execute_reply":"2023-11-17T01:36:40.526725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from __future__ import print_function, division\nimport os, gc, copy\nfrom time import time\nfrom typing import List, Tuple\nimport h5py\nimport json\nimport subprocess\nos.environ[\"OPENCV_IO_MAX_IMAGE_PIXELS\"] = pow(2,40).__str__()\nfrom pathlib import Path\nfrom tqdm.auto import tqdm\ntqdm.pandas()\nfrom collections import defaultdict, Counter\nimport pickle\nimport ssl\nssl._create_default_https_context = ssl._create_unverified_context\n\nimport math\nimport random\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\n\nimport tensorflow as tf\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data import DataLoader, Dataset\nimport torchvision\n\nimport timm\nfrom timm.data import resolve_data_config\nfrom timm.data.transforms_factory import create_transform\n\nimport IPython.display as display\n\nfrom PIL import Image\nimport cv2 as cv\nimport pyvips\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2023-11-17T01:46:09.690169Z","iopub.execute_input":"2023-11-17T01:46:09.690955Z","iopub.status.idle":"2023-11-17T01:46:09.701191Z","shell.execute_reply.started":"2023-11-17T01:46:09.69092Z","shell.execute_reply":"2023-11-17T01:46:09.700114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"config = dict(\n    seed = 42,\n    folds = 8,\n    img_size = [512, 512],\n    learning_rate = 3e-4, # 2e-5, 3e-4\n    eta_min = 1e-5,\n    epochs = 8,\n    batch_size = 32,\n)\n\ndef seeding(SEED):\n    np.random.seed(SEED)\n    random.seed(SEED)\n    os.environ['PYTHONHASHSEED'] = str(SEED)\n    torch.manual_seed(SEED)\n    if torch.cuda.is_available(): \n        torch.cuda.manual_seed(SEED)\n        torch.cuda.manual_seed_all(SEED)\n        torch.backends.cudnn.deterministic = True\n        torch.backends.cudnn.benchmark = True\n#     os.environ['TF_CUDNN_DETERMINISTIC'] = str(SEED)\n#     tf.random.set_seed(SEED)\n#     keras.utils.set_random_seed(seed=SEED)\n    print('seeding done!!!')\n\ndef flush():\n    gc.collect()\n    if torch.cuda.is_available():\n        torch.cuda.empty_cache()\n        torch.cuda.reset_peak_memory_stats()\n    \nseeding(config['seed'])","metadata":{"execution":{"iopub.status.busy":"2023-11-17T01:36:57.499067Z","iopub.execute_input":"2023-11-17T01:36:57.499631Z","iopub.status.idle":"2023-11-17T01:36:57.541882Z","shell.execute_reply.started":"2023-11-17T01:36:57.499604Z","shell.execute_reply":"2023-11-17T01:36:57.540881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATA_PATH = Path(\"../input/UBC-OCEAN/\")\nos.listdir(DATA_PATH)","metadata":{"execution":{"iopub.status.busy":"2023-11-17T01:36:57.544333Z","iopub.execute_input":"2023-11-17T01:36:57.544953Z","iopub.status.idle":"2023-11-17T01:36:57.559395Z","shell.execute_reply.started":"2023-11-17T01:36:57.544917Z","shell.execute_reply":"2023-11-17T01:36:57.558535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(DATA_PATH/'train.csv')\ntest_df = pd.read_csv(DATA_PATH/'test.csv')\nsample_df = pd.read_csv(DATA_PATH/'sample_submission.csv')\n\nget_train_images = lambda x: \"/kaggle/input/UBC-OCEAN/train_images/\" + str(x) + \".png\"\nget_test_images = lambda x: \"/kaggle/input/UBC-OCEAN/test_images/\" + str(x) + \".png\"\n\ncheck_path = lambda path: tf.io.gfile.exists(path)\n\ntrain_df['image_path'] = train_df.loc[:, 'image_id'].progress_apply(get_train_images)\ntrain_df['exists'] = train_df.loc[:, 'image_path'].map(check_path)\n\nprint(\"Checking training data ...\")\ndisplay.display(train_df['exists'].value_counts())\ntrain_df = train_df[train_df['exists'] == True]\ntrain_df.reset_index(drop=True, inplace=True)\n\ntest_df['image_path'] = test_df.loc[:, 'image_id'].progress_apply(get_test_images)\ntest_df['exists'] = test_df.loc[:, 'image_path'].map(check_path)\n\nprint(\"Checking test data ...\")\ndisplay.display(test_df['exists'].value_counts())\ntest_df = test_df[test_df['exists'] == True]\ntest_df.reset_index(drop=True, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-11-17T01:36:57.560443Z","iopub.execute_input":"2023-11-17T01:36:57.560725Z","iopub.status.idle":"2023-11-17T01:36:58.590318Z","shell.execute_reply.started":"2023-11-17T01:36:57.560701Z","shell.execute_reply":"2023-11-17T01:36:58.589418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_thumbnails = os.listdir(DATA_PATH/'train_thumbnails')\n# get_thumbnail = lambda thumbnail: int(thumbnail.split(\"_\")[0])\n# train_thumbnails = [get_thumbnail(t) for t in train_thumbnails]\n# train_df.loc[train_df['image_id'].isin(train_thumbnails), 'is_tma'] = False","metadata":{"execution":{"iopub.status.busy":"2023-11-17T01:36:58.591487Z","iopub.execute_input":"2023-11-17T01:36:58.591795Z","iopub.status.idle":"2023-11-17T01:36:58.595928Z","shell.execute_reply.started":"2023-11-17T01:36:58.591768Z","shell.execute_reply":"2023-11-17T01:36:58.594984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = train_df['label'].unique().tolist()\nid2label = {l:i for i, l in enumerate(labels)}\nlabel2id = {i:l for i, l in enumerate(labels)}\n\ntrain_df['target'] = train_df['label'].map(id2label)\ntrain_df['target'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-11-17T01:36:58.597548Z","iopub.execute_input":"2023-11-17T01:36:58.597895Z","iopub.status.idle":"2023-11-17T01:36:58.614962Z","shell.execute_reply.started":"2023-11-17T01:36:58.597868Z","shell.execute_reply":"2023-11-17T01:36:58.614007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_image(img_path, scale_factor=2):\n    img = pyvips.Image.new_from_file(img_path, access='sequential').resize(1/scale_factor).numpy()\n    return img","metadata":{"execution":{"iopub.status.busy":"2023-11-17T01:36:58.616156Z","iopub.execute_input":"2023-11-17T01:36:58.616408Z","iopub.status.idle":"2023-11-17T01:36:58.622562Z","shell.execute_reply.started":"2023-11-17T01:36:58.616385Z","shell.execute_reply":"2023-11-17T01:36:58.621563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Class abstraction of smart tiles","metadata":{"execution":{"iopub.status.busy":"2023-11-15T12:08:42.939227Z","iopub.execute_input":"2023-11-15T12:08:42.939975Z","iopub.status.idle":"2023-11-15T12:08:42.959024Z","shell.execute_reply.started":"2023-11-15T12:08:42.939942Z","shell.execute_reply":"2023-11-15T12:08:42.958142Z"}}},{"cell_type":"code","source":"BLOCK_SIZE = 28\nBLOCKS_PER_CROP = 8\nCROP_SIZE = BLOCK_SIZE * BLOCKS_PER_CROP\nBLOCK_THR = 50\nCROP_THR = 0.6\nMAX_CROPS_PER_IMAGE = 20\nIMAGES_PER_SAMPLE = 4\nEPOCHS_NUM = 10\nSCALE_FACTOR = 4","metadata":{"execution":{"iopub.status.busy":"2023-11-17T01:36:58.623662Z","iopub.execute_input":"2023-11-17T01:36:58.623959Z","iopub.status.idle":"2023-11-17T01:36:58.631115Z","shell.execute_reply.started":"2023-11-17T01:36:58.623934Z","shell.execute_reply":"2023-11-17T01:36:58.630268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from time import time\nclass DataPreparation:\n    def __init__(self, data, visualize: bool = False, seed: int = 42):\n        self.visualize = visualize\n        self.seed = seed\n\n        self.data = data\n        \n        \n    @staticmethod\n    def _filter_bad_images(data: List[Tuple]) -> List[Tuple]:\n        return [\n            (image_id, label, center_id)\n            for image_id, label, center_id in data\n            if image_id not in BAD_IMAGE_IDS\n        ]\n\n    @staticmethod\n    def _add_rect_to_numpy(image: np.ndarray, x: int, y: int, size: int, thickness: int) -> None:\n        image[x:x + size, y:y + thickness] = (0, 0, 0)\n        image[x:x + thickness, y:y + size] = (0, 0, 0)\n        image[x:x + size, y + size:y + size + thickness] = (0, 0, 0)\n        image[x + size:x + size + thickness, y:y + size] = (0, 0, 0)\n\n    \n    @staticmethod\n    def _get_blocks_map(image: np.ndarray) -> np.ndarray:\n        pixels_diff = np.sum((image[:-1, :, :] - image[1:, :, :]) ** 2, axis=2)\n        pixels_diff = np.cumsum(np.cumsum(pixels_diff, axis=0), axis=1)\n        blocks_map = np.zeros((\n            (image.shape[0] + BLOCK_SIZE - 1) // BLOCK_SIZE,\n            (image.shape[1] + BLOCK_SIZE - 1) // BLOCK_SIZE,\n        ))\n        for x in range(0, pixels_diff.shape[0], BLOCK_SIZE):\n            for y in range(0, pixels_diff.shape[1], BLOCK_SIZE):\n                nx = min(x + BLOCK_SIZE, pixels_diff.shape[0])\n                ny = min(y + BLOCK_SIZE, pixels_diff.shape[1])\n                block_sum = int(pixels_diff[nx - 1, ny - 1])\n                if x:\n                    block_sum -= int(pixels_diff[x - 1, ny - 1])\n                if y:\n                    block_sum -= int(pixels_diff[nx - 1, y - 1])\n                if x and y:\n                    block_sum += int(pixels_diff[x - 1, y - 1])\n                blocks_map[x // BLOCK_SIZE][y // BLOCK_SIZE] = \\\n                    (block_sum / BLOCK_SIZE / BLOCK_SIZE) > BLOCK_THR\n        return blocks_map\n    \n    def _generate_crops_positions(\n            self,\n            image: np.ndarray,\n            crop_thr: float,\n    ) -> Tuple[List[Tuple[int, int]], np.ndarray, np.ndarray, np.ndarray, np.ndarray]:\n        blocks_map = self._get_blocks_map(image)\n\n        if self.visualize:\n            for i in range(blocks_map.shape[0]):\n                for j in range(blocks_map.shape[1]):\n                    if blocks_map[i][j]:\n                        self._add_rect_to_numpy(\n                            image,\n                            i * BLOCK_SIZE,\n                            j * BLOCK_SIZE,\n                            BLOCK_SIZE,\n                            1,\n                        )\n\n        good_crops_starts = []\n        for x in range(0, image.shape[0] - CROP_SIZE + 1, BLOCK_SIZE):\n            for y in range(0, image.shape[1] - CROP_SIZE + 1, BLOCK_SIZE):\n                _x, _y = x // BLOCK_SIZE, y // BLOCK_SIZE\n                crop_sum = blocks_map[_x:_x + BLOCKS_PER_CROP, _y:_y + BLOCKS_PER_CROP].sum()\n                if crop_sum > BLOCKS_PER_CROP * BLOCKS_PER_CROP * crop_thr:\n                    good_crops_starts.append((x, y))\n\n        if self.visualize:\n            for x, y in good_crops_starts:\n                self._add_rect_to_numpy(image, x, y, CROP_SIZE, 1)\n\n        return good_crops_starts\n\n    @staticmethod\n    def _process_crop(crop: np.ndarray) -> np.ndarray:\n        return crop\n\n    def _create_crops(\n        self,\n        image: np.ndarray,\n        crops_starts: List[Tuple[int]],\n    ) -> List[np.ndarray]:\n        return [\n            Image.fromarray(\n                self._process_crop(\n                    image[x:x + CROP_SIZE, y:y + CROP_SIZE],\n                )\n            )\n            for x, y in crops_starts\n        ]\n\n    @staticmethod\n    def _get_unique_crops(crop_starts: List[Tuple[int, int]], order) -> List[Tuple[int, int]]:\n        def inter_size_1d(a: int, b: int, c: int, d: int) -> int:\n            return max(0, min(b, d) - max(a, c))\n\n        def inter_size_2d(crop_start_1: Tuple[int, int], crop_start_2: Tuple[int, int]) -> int:\n            return inter_size_1d(\n                crop_start_1[0], crop_start_1[0] + CROP_SIZE,\n                crop_start_2[0], crop_start_2[0] + CROP_SIZE,\n            ) * inter_size_1d(\n                crop_start_1[1], crop_start_1[1] + CROP_SIZE,\n                crop_start_2[1], crop_start_2[1] + CROP_SIZE,\n            )\n\n        crop_starts_sorted = sorted(crop_starts, key=order)\n        final_crop_starts = []\n        for crop_start in crop_starts_sorted:\n            if any(\n                    inter_size_2d(crop_start, crop_start_prev) > CROP_SIZE * CROP_SIZE // 2\n                    for crop_start_prev in final_crop_starts\n            ):\n                continue\n            final_crop_starts.append(crop_start)\n        return final_crop_starts\n    \n    @staticmethod\n    def _read_and_resize_image(image_id: str, base_image_path: str, scale_factor: int) -> np.ndarray:       \n        image_path = os.path.join(base_image_path, f'{image_id}.png')\n        image = pyvips.Image.new_from_file(image_path, access='sequential')\n        return image.resize(1.0 / scale_factor).numpy()\n    \n    def prepare_crops(\n            self,\n            image_ids: List[int],\n            base_image_path: str,\n    ) -> Tuple[List[List[np.ndarray]], List[List[Tuple[int]]], List[Tuple[np.ndarray, np.ndarray]]]:\n        np.random.seed(self.seed)\n        image_crops = []\n        image_crops_indices = []\n        idx = 0\n        for image_id in tqdm(image_ids):\n            start_time = time()\n#             print(f\"image id: {image_id}\")\n            is_tma = self.data.loc[idx, 'is_tma']\n            sf = SCALE_FACTOR if is_tma else SCALE_FACTOR*5\n            image = self._read_and_resize_image(image_id, base_image_path, sf)\n            gc.collect()\n            print(f'Is tma: {is_tma}, scale factor: {sf}, Rescaling done in {time() - start_time} seconds. Image shape is {image.shape}')\n            idx += 1\n            found_flag = False\n            for crop_thr in np.arange(CROP_THR, -0.1, -0.1):\n                good_crops_starts = self._generate_crops_positions(image, crop_thr)\n                if len(good_crops_starts) < IMAGES_PER_SAMPLE:\n                    print('Bad image', image_id, 'crop_thr', crop_thr, 'only', len(good_crops_starts))\n                    continue\n\n                good_crops_starts_unique = []\n                for order in [\n                    lambda x: (x[0], x[1]),\n                    lambda x: (-x[0], -x[1]),\n                ]:\n                    good_crops_starts_unique.extend(self._get_unique_crops(good_crops_starts, order))\n                good_crops_starts_unique = list(set(good_crops_starts_unique))\n\n                if len(good_crops_starts_unique) < IMAGES_PER_SAMPLE:\n                    print('Bad image', image_id, 'crop_thr', crop_thr, 'only', len(good_crops_starts_unique))\n                    continue\n\n                good_crops_starts_sample_ids = np.random.choice(\n                    list(range(len(good_crops_starts_unique))),\n                    min(len(good_crops_starts_unique), MAX_CROPS_PER_IMAGE),\n                    replace=False,\n                )\n                good_crops_starts_sample = np.array(good_crops_starts_unique)[good_crops_starts_sample_ids]\n                image_crops_indices.append(good_crops_starts_sample)\n                image_crops.append(self._create_crops(image, good_crops_starts_sample))\n                found_flag = True\n                break\n            if not found_flag:\n                image_crops_indices.append([])\n                image_crops.append([])\n                print('No crops was found')\n            print(f'Done {image_id} in {time() - start_time} seconds')\n        gc.collect()\n        return image_crops, image_crops_indices\n    \n    def process_train(\n            self\n    ) -> Tuple[List[List[np.ndarray]], List[List[Tuple[int]]]]:\n        return self.prepare_crops(\n            self.data['image_id'].tolist(),\n            '/kaggle/input/UBC-OCEAN/train_images/',\n        )","metadata":{"execution":{"iopub.status.busy":"2023-11-17T01:36:58.633917Z","iopub.execute_input":"2023-11-17T01:36:58.63419Z","iopub.status.idle":"2023-11-17T01:36:58.968102Z","shell.execute_reply.started":"2023-11-17T01:36:58.634166Z","shell.execute_reply":"2023-11-17T01:36:58.967157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_df.iloc[35:45, :]","metadata":{"execution":{"iopub.status.busy":"2023-11-17T01:48:48.268969Z","iopub.execute_input":"2023-11-17T01:48:48.269921Z","iopub.status.idle":"2023-11-17T01:48:48.273699Z","shell.execute_reply.started":"2023-11-17T01:48:48.269876Z","shell.execute_reply":"2023-11-17T01:48:48.272744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# data_prep = DataPreparation(data=train_df.iloc[:5, :])\n# image_crops, image_crops_indices = data_prep.process_train()","metadata":{"execution":{"iopub.status.busy":"2023-11-17T01:48:04.650071Z","iopub.execute_input":"2023-11-17T01:48:04.65091Z","iopub.status.idle":"2023-11-17T01:48:04.655905Z","shell.execute_reply.started":"2023-11-17T01:48:04.650828Z","shell.execute_reply":"2023-11-17T01:48:04.654787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# b_size = len(image_crops[2])\n# row = 4\n# col = b_size // row\n\n# plt.figure(figsize=(16, 6))\n\n# for i in tqdm(range(b_size)):\n#     image = image_crops[0][i]\n# #     image = image.permute(1,2,0).cpu().numpy()\n    \n#     plt.subplot(row, col, i + 1)\n#     plt.xticks([])\n#     plt.yticks([])\n# #     plt.title(target.cpu().numpy(), fontsize=8)\n#     plt.imshow(image);","metadata":{"execution":{"iopub.status.busy":"2023-11-17T01:48:13.528799Z","iopub.execute_input":"2023-11-17T01:48:13.529196Z","iopub.status.idle":"2023-11-17T01:48:13.533988Z","shell.execute_reply.started":"2023-11-17T01:48:13.529166Z","shell.execute_reply":"2023-11-17T01:48:13.532992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DUMPED_DATALOADER_PATH = '/kaggle/working/data_loaders.pkl'\ndef get_sub_data(data, image_crops, image_crops_indices, sample_ids):\n    return [data[i][0] for i in sample_ids], \\\n        [data[i][1] for i in sample_ids], \\\n        [data[i][2] for i in sample_ids], \\\n        [image_crops[i] for i in sample_ids], \\\n        [image_crops_indices[i] for i in sample_ids]\n\ndata_prep = DataPreparation(data=train_df)\n\nimage_crops, image_crops_indices = data_prep.process_train()\nwith open(DUMPED_DATALOADER_PATH, 'wb') as file:\n    pickle.dump([image_crops, image_crops_indices], file)","metadata":{"execution":{"iopub.status.busy":"2023-11-17T01:49:19.377623Z","iopub.execute_input":"2023-11-17T01:49:19.378024Z","iopub.status.idle":"2023-11-17T09:39:44.01313Z","shell.execute_reply.started":"2023-11-17T01:49:19.377992Z","shell.execute_reply":"2023-11-17T09:39:44.012119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# help(Image)","metadata":{"execution":{"iopub.status.busy":"2023-11-17T01:39:38.19767Z","iopub.execute_input":"2023-11-17T01:39:38.198002Z","iopub.status.idle":"2023-11-17T01:39:38.201857Z","shell.execute_reply.started":"2023-11-17T01:39:38.197973Z","shell.execute_reply":"2023-11-17T01:39:38.200892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}