{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"},{"sourceId":6774400,"sourceType":"datasetVersion","datasetId":3895136},{"sourceId":6984590,"sourceType":"datasetVersion","datasetId":4014175},{"sourceId":147716186,"sourceType":"kernelVersion"}],"dockerImageVersionId":30587,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### this notebook is based on Jirka's great notebook, [Cancer🔬Subtype: decompose 🖽 large image -> tiles](https://www.kaggle.com/code/jirkaborovec/cancer-subtype-decompose-large-image-tiles/notebook)\n### visualization in this notebook will give us the way to think about sampling tiles with labels","metadata":{}},{"cell_type":"code","source":"!ls /kaggle/input/pyvips-python-and-deb-package\n# intall the deb packages\n!dpkg -i --force-depends /kaggle/input/pyvips-python-and-deb-package/linux_packages/archives/*.deb\n# install the python wrapper+-\n!pip install pyvips -f /kaggle/input/pyvips-python-and-deb-package/python_packages/ --no-index","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":false,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-11-17T17:34:53.919434Z","iopub.execute_input":"2023-11-17T17:34:53.920016Z","iopub.status.idle":"2023-11-17T17:35:28.934432Z","shell.execute_reply.started":"2023-11-17T17:34:53.919974Z","shell.execute_reply":"2023-11-17T17:35:28.932997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n\nimport pyvips\nimport numpy as np\nfrom PIL import Image\nimport matplotlib.pyplot as plt\nfrom glob import glob\n\nos.environ['OPENCV_IO_MAX_IMAGE_PIXELS'] = str(pow(2, 40))\nimport cv2\n\nImage.MAX_IMAGE_PIXELS = 10000000000\nos.environ['VIPS_DISC_THRESHOLD'] = '9gb'","metadata":{"execution":{"iopub.status.busy":"2023-11-17T16:46:24.276388Z","iopub.execute_input":"2023-11-17T16:46:24.276955Z","iopub.status.idle":"2023-11-17T16:46:24.285219Z","shell.execute_reply.started":"2023-11-17T16:46:24.276912Z","shell.execute_reply":"2023-11-17T16:46:24.283945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_image_paths = glob(\"/kaggle/input/UBC-OCEAN/train_images/*.png\")\nmask_image_paths = glob(\"/kaggle/input/ubc-ovarian-cancer-competition-supplemental-masks/*.png\")\n\n# find the id that exists in both train image folder and mask folder\ntrain_ids = [os.path.splitext(os.path.basename(p))[0] for p in train_image_paths]\nmask_ids = [os.path.splitext(os.path.basename(p))[0] for p in mask_image_paths]\nids = list(set(train_ids).intersection(mask_ids))\nlen(ids)","metadata":{"execution":{"iopub.status.busy":"2023-11-17T16:48:49.671617Z","iopub.execute_input":"2023-11-17T16:48:49.672073Z","iopub.status.idle":"2023-11-17T16:48:49.692506Z","shell.execute_reply.started":"2023-11-17T16:48:49.672038Z","shell.execute_reply":"2023-11-17T16:48:49.69113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# a little modification from Jirka's original\n\ndef extract_image_tiles(p_img, folder, size:int = 768, scale:float = 1.0, drop_thr:float = 0.8, save:bool = True) -> list:\n    \"\"\"\n    get tiles from image\n\n    params\n    ------\n    p_img: str\n        path to image\n    folder: str\n        folder to save tiles\n    size: int\n        size of square tiles. default: 768.\n    scale: float\n        scale of tiles to resize. default: 1.0.\n    drop_thr: float\n        threshold to drop almost empty tiles(almost black or white). default: 0.8.\n\n    returns\n    -------\n    files: list\n        list of saved tile paths\n    idxs: list\n        list of coordinates of tiles\n    \"\"\"\n\n    # create outout folder\n    img_name = os.path.splitext(os.path.basename(p_img))[0]\n    folder = os.path.join(folder, img_name)\n    os.makedirs(folder, exist_ok=True)\n\n    im = pyvips.Image.new_from_file(p_img)\n    w = h = size\n    # https://stackoverflow.com/a/47581978/4521646\n    idxs = [(y, y + h, x, x + w) for y in range(0, im.height, h) for x in range(0, im.width, w)]\n    files = []\n    for k, (y, y_, x, x_) in enumerate(idxs):\n        # https://libvips.github.io/pyvips/vimage.html#pyvips.Image.crop\n        tile = im.crop(x, y, min(w, im.width - x), min(h, im.height - y)).numpy()[..., :3]\n\n        # padding edges\n        if tile.shape[:2] != (h, w):\n            tile_ = tile\n            tile_size = (h, w) if tile.ndim == 2 else (h, w, tile.shape[2])\n            tile = np.zeros(tile_size, dtype=tile.dtype)\n            tile[:tile_.shape[0], :tile_.shape[1], ...] = tile_\n        \n        # filter out almost black tiles\n        mask_bg = np.sum(tile, axis=2) == 0 # black_flag, shape: (size, size)\n        # if percentage of black pixels is more than drop_thr, skip this tile\n        if np.sum(mask_bg) >= (np.prod(mask_bg.shape) * drop_thr):\n            #print(f\"skip almost empty tile: {k:06}_{int(x_ / w)}-{int(y_ / h)}\")\n            continue\n\n        # filter out almost white tiles\n        tile[mask_bg, :] = 255\n        mask_bg = np.mean(tile, axis=2) > 240\n        if np.sum(mask_bg) >= (np.prod(mask_bg.shape) * drop_thr):\n            #print(f\"skip almost empty tile: {k:06}_{int(x_ / w)}-{int(y_ / h)}\")\n            continue\n\n        p_img = os.path.join(folder, f\"{k:06}_{int(x_ / w)}-{int(y_ / h)}.png\")\n        # print(tile.shape, tile.dtype, tile.min(), tile.max())\n        new_size = int(size * scale), int(size * scale)\n        tile_resize = Image.fromarray(tile).resize(new_size, Image.LANCZOS)\n        if save: tile_resize.save(p_img)\n        files.append(p_img)\n    return files, idxs\n#-------------------------------\ndef extract_mask_tiles(p_img, folder, size:int = 768, scale:float = 1.0, drop_thr:float = 0.95, save:bool = True) -> list:\n    \"\"\"\n    get tiles from image\n\n    params\n    ------\n    p_img: str\n        path to image\n    folder: str\n        folder to save tiles\n    size: int\n        size of square tiles. default: 768.\n    scale: float\n        scale of tiles to resize. default: 1.0.\n    drop_thr: float\n        threshold to drop almost empty tiles(almost black or white). default: 0.8.\n\n    returns\n    -------\n    files: list\n        list of saved tile paths\n    idxs: list\n        list of coordinates of tiles\n    \"\"\"\n\n    # create outout folder\n    img_name = os.path.splitext(os.path.basename(p_img))[0]\n    folder = os.path.join(folder, img_name + \"_mask\")\n    os.makedirs(folder, exist_ok=True)\n\n    im = pyvips.Image.new_from_file(p_img)\n    w = h = size\n    # https://stackoverflow.com/a/47581978/4521646\n    idxs = [(y, y + h, x, x + w) for y in range(0, im.height, h) for x in range(0, im.width, w)]\n    files = []\n    for k, (y, y_, x, x_) in enumerate(idxs):\n        # https://libvips.github.io/pyvips/vimage.html#pyvips.Image.crop\n        tile = im.crop(x, y, min(w, im.width - x), min(h, im.height - y)).numpy()[..., :3]\n\n        # padding edges\n        if tile.shape[:2] != (h, w):\n            tile_ = tile\n            tile_size = (h, w) if tile.ndim == 2 else (h, w, tile.shape[2])\n            tile = np.zeros(tile_size, dtype=tile.dtype)\n            tile[:tile_.shape[0], :tile_.shape[1], ...] = tile_\n\n        # filter out almost black tiles\n        mask_bg = np.sum(tile, axis=2) == 0 # black_flag, shape: (size, size)\n        # if percentage of black pixels is more than drop_thr, skip this tile\n        if np.sum(mask_bg) >= (np.prod(mask_bg.shape) * drop_thr):\n            #print(f\"skip almost empty tile: {k:06}_{int(x_ / w)}-{int(y_ / h)}\")\n            continue\n\n        p_img = os.path.join(folder, f\"{k:06}_{int(x_ / w)}-{int(y_ / h)}.png\")\n        # print(tile.shape, tile.dtype, tile.min(), tile.max())\n        new_size = int(size * scale), int(size * scale)\n        tile_resize = Image.fromarray(tile).resize(new_size, Image.LANCZOS)\n        if save: tile_resize.save(p_img)\n        files.append(p_img)\n    return files, idxs","metadata":{"execution":{"iopub.status.busy":"2023-11-17T16:49:21.034282Z","iopub.execute_input":"2023-11-17T16:49:21.034745Z","iopub.status.idle":"2023-11-17T16:49:21.065037Z","shell.execute_reply.started":"2023-11-17T16:49:21.03471Z","shell.execute_reply":"2023-11-17T16:49:21.063199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# example \nid = 1101\nstr(id) in ids","metadata":{"execution":{"iopub.status.busy":"2023-11-17T16:57:32.167494Z","iopub.execute_input":"2023-11-17T16:57:32.16865Z","iopub.status.idle":"2023-11-17T16:57:32.176996Z","shell.execute_reply.started":"2023-11-17T16:57:32.16859Z","shell.execute_reply":"2023-11-17T16:57:32.175537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntile_img_paths__, ___ = extract_image_tiles(\"/kaggle/input/UBC-OCEAN/train_images/1101.png\", \n                                            \"/kaggle/temp/images\",\n                                            size=1024, scale=0.5)","metadata":{"execution":{"iopub.status.busy":"2023-11-17T16:58:55.802314Z","iopub.execute_input":"2023-11-17T16:58:55.802784Z","iopub.status.idle":"2023-11-17T16:59:52.857862Z","shell.execute_reply.started":"2023-11-17T16:58:55.802753Z","shell.execute_reply":"2023-11-17T16:59:52.856446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls -l  /kaggle/temp/images/1101","metadata":{"execution":{"iopub.status.busy":"2023-11-17T16:59:52.860666Z","iopub.execute_input":"2023-11-17T16:59:52.8611Z","iopub.status.idle":"2023-11-17T16:59:54.059237Z","shell.execute_reply.started":"2023-11-17T16:59:52.861064Z","shell.execute_reply":"2023-11-17T16:59:54.057248Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nmask_img_paths__, ___ = extract_mask_tiles(\"/kaggle/input/ubc-ovarian-cancer-competition-supplemental-masks/1101.png\", \n                                           \"/kaggle/temp/images\",\n                                            size=1024, scale=0.5)","metadata":{"execution":{"iopub.status.busy":"2023-11-17T16:59:54.06185Z","iopub.execute_input":"2023-11-17T16:59:54.06238Z","iopub.status.idle":"2023-11-17T17:00:25.835811Z","shell.execute_reply.started":"2023-11-17T16:59:54.062333Z","shell.execute_reply":"2023-11-17T17:00:25.834295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls -l  /kaggle/temp/images/1101_mask","metadata":{"execution":{"iopub.status.busy":"2023-11-17T17:00:25.838162Z","iopub.execute_input":"2023-11-17T17:00:25.838545Z","iopub.status.idle":"2023-11-17T17:00:26.992947Z","shell.execute_reply.started":"2023-11-17T17:00:25.838515Z","shell.execute_reply":"2023-11-17T17:00:26.991434Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nimg_tmp_ = cv2.imread('/kaggle/input/UBC-OCEAN/train_images/1101.png')\nimg_tmp_.shape","metadata":{"execution":{"iopub.status.busy":"2023-11-17T17:04:06.622382Z","iopub.execute_input":"2023-11-17T17:04:06.623048Z","iopub.status.idle":"2023-11-17T17:04:24.09834Z","shell.execute_reply.started":"2023-11-17T17:04:06.623005Z","shell.execute_reply":"2023-11-17T17:04:24.097457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(cv2.cvtColor(img_tmp_, cv2.COLOR_BGR2RGB))","metadata":{"execution":{"iopub.status.busy":"2023-11-17T17:04:24.099911Z","iopub.execute_input":"2023-11-17T17:04:24.100697Z","iopub.status.idle":"2023-11-17T17:05:50.817319Z","shell.execute_reply.started":"2023-11-17T17:04:24.100666Z","shell.execute_reply":"2023-11-17T17:05:50.815591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# a little modification from Jirka's original\n\n# names = [os.path.splitext(os.path.basename(p_img))[0] for p_img in tile_img_paths__]\n# pos = [name.split(\"_\")[-1] for name in names]\n# idx_x, idx_y = zip(*[list(map(int, p.split(\"-\"))) for p in pos])\n# nb_rows = len(set(idx_y))\n# nb_cols = len(set(idx_x))\nnb_rows = img_tmp_.shape[0] // 1024 + 1\nnb_cols = img_tmp_.shape[1] // 1024 + 1\n\nprint(f\"{nb_rows=}\\n{nb_cols=}\")\n\nfig, axes = plt.subplots(nrows=nb_rows, ncols=nb_cols,\n    figsize=(nb_cols * 0.5, nb_rows * 0.5)\n)\nfor i in range(nb_rows):\n    for j in range(nb_cols):\n        axes[i, j].set_facecolor(\"aqua\")\n        axes[i, j].set_xticklabels([])\n        axes[i, j].set_yticklabels([])\n\n        tile_flag = (len(glob(f\"/kaggle/temp/images/1101/*_{j+1}-{i+1}.png\")) > 0)\n        mask_flag = (len(glob(f\"/kaggle/temp/images/1101_mask/*_{j+1}-{i+1}.png\")) > 0)\n\n        # if len(glob(f\"../output/test_tiles/1101/*_{j+1}-{i+1}.png\")):\n        #     axes[i, j].set_facecolor(\"red\")\n        if tile_flag & mask_flag:\n            img_path = glob(f\"/kaggle/temp/images/1101/*_{j+1}-{i+1}.png\")[0]\n            tile_path = glob(f\"/kaggle/temp/images/1101_mask/*_{j+1}-{i+1}.png\")[0]\n            img_tile = cv2.imread(img_path)\n            img_mask = cv2.imread(tile_path)\n            assert img_tile.shape == img_mask.shape\n            img_add = cv2.addWeighted(img_tile, 1, img_mask, 0.4, 0)\n            axes[i, j].imshow(cv2.cvtColor(img_add, cv2.COLOR_BGR2RGB))\n        elif tile_flag:\n            tile_path = glob(f\"/kaggle/temp/images/1101/*_{j+1}-{i+1}.png\")[0]\n            img_tile = cv2.imread(tile_path)\n            axes[i, j].imshow(cv2.cvtColor(img_tile, cv2.COLOR_BGR2RGB))\n        elif mask_flag:\n            img_path = glob(f\"/kaggle/temp/images/1101_mask/*_{j+1}-{i+1}.png\")[0]\n            img_mask = cv2.imread(img_path)\n            axes[i, j].imshow(cv2.cvtColor(img_mask, cv2.COLOR_BGR2RGB))\n        else:\n            pass\n\n\n# for p_img, x, y in zip(tile_img_paths__, idx_x, idx_y):\n#     img = plt.imread(p_img)\n#     ax = axes[y - 1, x - 1]\n#     ax.imshow(img)\nprint(f\"image size: {img_tile.shape}\")\n\n\n\nplt.subplots_adjust(wspace=0, hspace=0)\n# fig.tight_layout()\n","metadata":{"execution":{"iopub.status.busy":"2023-11-17T17:06:13.374363Z","iopub.execute_input":"2023-11-17T17:06:13.374895Z","iopub.status.idle":"2023-11-17T17:07:20.02118Z","shell.execute_reply.started":"2023-11-17T17:06:13.374858Z","shell.execute_reply":"2023-11-17T17:07:20.019518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}