{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"},{"sourceId":6774400,"sourceType":"datasetVersion","datasetId":3895136}],"dockerImageVersionId":30558,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!ls /kaggle/input/pyvips-python-and-deb-package\n# intall the deb packages\n!dpkg -i --force-depends /kaggle/input/pyvips-python-and-deb-package/linux_packages/archives/*.deb\n# install the python wrapper\n!pip install pyvips -f /kaggle/input/pyvips-python-and-deb-package/python_packages/ --no-index","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-11-18T20:48:58.369049Z","iopub.execute_input":"2023-11-18T20:48:58.369421Z","iopub.status.idle":"2023-11-18T20:50:08.468327Z","shell.execute_reply.started":"2023-11-18T20:48:58.369393Z","shell.execute_reply":"2023-11-18T20:50:08.466835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Decompose image to tiles/grid 🖽\n\nThis notebook is meant to run mutiple times just on different chunk of training data...\n\n1. set range and run notebook\n2. set following range and run\n3. downaload all particular results\n4. merge to single dataset - https://www.kaggle.com/datasets/jirkaborovec/tiles-of-ovarian-cancer-1024px-scale-0-25","metadata":{}},{"cell_type":"code","source":"import os\nimport pyvips\nimport numpy as np\nfrom PIL import Image\n\nos.environ['VIPS_DISC_THRESHOLD'] = '9gb'\n\nDATASET_IMAGES = \"/kaggle/input/UBC-OCEAN/train_images\"\n\n!mkdir -p /kaggle/temp/images","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-11-18T20:50:08.472781Z","iopub.execute_input":"2023-11-18T20:50:08.473149Z","iopub.status.idle":"2023-11-18T20:50:08.807823Z","shell.execute_reply.started":"2023-11-18T20:50:08.473114Z","shell.execute_reply":"2023-11-18T20:50:08.806804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def extract_image_tiles(\n    p_img, folder, size: int = 2048, scale: float = 0.5, drop_thr: float = 0.85\n) -> list:\n    name, _ = os.path.splitext(os.path.basename(p_img))\n    im = pyvips.Image.new_from_file(p_img)\n    w = h = size\n    # https://stackoverflow.com/a/47581978/4521646\n    inds = [(y, y + h, x, x + w)\n            for y in range(0, im.height, h) for x in range(0, im.width, w)]\n    files, idxs = [], []\n    for k, idx in enumerate(inds):\n        y, y_, x, x_ = idx\n        # https://libvips.github.io/pyvips/vimage.html#pyvips.Image.crop\n        tile = im.crop(x, y, min(w, im.width - x), min(h, im.height - y)).numpy()[..., :3]\n        mask_bg = np.sum(tile, axis=2) == 0\n        if np.sum(mask_bg) >= (np.prod(mask_bg.shape) * drop_thr):\n            #print(f\"skip almost empty tile: {k:06}_{int(x_ / w)}-{int(y_ / h)}\")\n            continue\n        if tile.shape[:2] != (h, w):\n            tile_ = tile\n            tile_size = (h, w) if tile.ndim == 2 else (h, w, tile.shape[2])\n            tile = np.zeros(tile_size, dtype=tile.dtype)\n            tile[:tile_.shape[0], :tile_.shape[1], ...] = tile_\n        p_img = os.path.join(folder, f\"{k:06}_{int(x_ / w)}-{int(y_ / h)}.png\")\n        # print(tile.shape, tile.dtype, tile.min(), tile.max())\n        new_size = int(size * scale), int(size * scale)\n        Image.fromarray(tile).resize(new_size, Image.LANCZOS).save(p_img)\n        files.append(p_img)\n        idxs.append(idx)\n    return files, idxs","metadata":{"execution":{"iopub.status.busy":"2023-11-18T20:55:35.486251Z","iopub.execute_input":"2023-11-18T20:55:35.486628Z","iopub.status.idle":"2023-11-18T20:55:35.497479Z","shell.execute_reply.started":"2023-11-18T20:55:35.486598Z","shell.execute_reply":"2023-11-18T20:55:35.496428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tiles_img, _ = extract_image_tiles(\n    os.path.join(DATASET_IMAGES, \"12244.png\"),\n    \"/kaggle/temp/images\", size=1024, scale=0.5\n)\n\n!ls -lh /kaggle/temp/images","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-11-18T20:55:37.443446Z","iopub.execute_input":"2023-11-18T20:55:37.444014Z","iopub.status.idle":"2023-11-18T20:56:40.29754Z","shell.execute_reply.started":"2023-11-18T20:55:37.443981Z","shell.execute_reply":"2023-11-18T20:56:40.296367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Show the image tiles with segmentations","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nnames = [os.path.splitext(os.path.basename(p_img))[0] for p_img in tiles_img]\npos = [name.split(\"_\")[-1] for name in names]\nidx_x, idx_y = zip(*[list(map(int, p.split(\"-\"))) for p in pos])\nnb_rows = len(set(idx_y))\nnb_cols = len(set(idx_x))\nprint(f\"{nb_rows=}\\n{nb_cols=}\")\n\nfig, axes = plt.subplots(nrows=nb_rows, ncols=nb_cols,\n    figsize=(nb_cols * 0.5, nb_rows * 0.5)\n)\nfor p_img, x, y in zip(tiles_img, idx_x, idx_y):\n    img = plt.imread(p_img)\n    ax = axes[y - 1, x - 1]\n    ax.imshow(img)\nprint(f\"image size: {img.shape}\")\n\nfor i in range(nb_rows):\n    for j in range(nb_cols):\n        axes[i, j].set_xticklabels([])\n        axes[i, j].set_yticklabels([])\n\nplt.subplots_adjust(wspace=0, hspace=0)\n# # fig.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2023-11-18T20:59:26.936951Z","iopub.execute_input":"2023-11-18T20:59:26.937358Z","iopub.status.idle":"2023-11-18T21:00:27.57339Z","shell.execute_reply.started":"2023-11-18T20:59:26.937323Z","shell.execute_reply":"2023-11-18T21:00:27.572614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Export all image tiles¶","metadata":{}},{"cell_type":"code","source":"import random\n\ndef subsample_rand_prune(files: list, max_samples: float = 1.0):\n    max_samples = max_samples if isinstance(max_samples, int) else int(len(files) * max_samples)\n    random.shuffle(files)\n    for p_img in files[max_samples:]:\n        os.remove(p_img)","metadata":{"execution":{"iopub.status.busy":"2023-11-18T21:06:45.716961Z","iopub.execute_input":"2023-11-18T21:06:45.717396Z","iopub.status.idle":"2023-11-18T21:06:45.723104Z","shell.execute_reply.started":"2023-11-18T21:06:45.717364Z","shell.execute_reply":"2023-11-18T21:06:45.722307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def extract_prune_tiles(\n    idx_path_img, folder, size: int = 2048, scale: float = 0.25,\n    drop_thr: float = 0.9, max_samples: float = 1.0\n) -> None:\n    idx, p_img = idx_path_img\n    print(f\"processing #{idx}: {p_img}\")\n    name, _ = os.path.splitext(os.path.basename(p_img))\n    folder = os.path.join(folder, name)\n    os.makedirs(folder, exist_ok=True)\n    tiles, _ = extract_image_tiles(p_img, folder, size, scale, drop_thr)\n    subsample_rand_prune(tiles, max_samples)","metadata":{"execution":{"iopub.status.busy":"2023-11-18T21:06:45.7245Z","iopub.execute_input":"2023-11-18T21:06:45.725366Z","iopub.status.idle":"2023-11-18T21:06:45.737862Z","shell.execute_reply.started":"2023-11-18T21:06:45.725336Z","shell.execute_reply":"2023-11-18T21:06:45.736818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"take chank of the whole dataset...","metadata":{}},{"cell_type":"code","source":"import glob\n\nls = sorted(glob.glob(os.path.join(DATASET_IMAGES, '*.png')))\nprint(f\"found images: {len(ls)}\")\nimg_name = lambda p_img: os.path.splitext(os.path.basename(p_img))[0]\nls_subset = ls[400:600]\n# print(ls_subset)","metadata":{"execution":{"iopub.status.busy":"2023-11-18T21:06:45.73938Z","iopub.execute_input":"2023-11-18T21:06:45.739883Z","iopub.status.idle":"2023-11-18T21:06:45.857787Z","shell.execute_reply.started":"2023-11-18T21:06:45.739854Z","shell.execute_reply":"2023-11-18T21:06:45.856567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"process only images from the fix - https://www.kaggle.com/competitions/UBC-OCEAN/discussion/451892","metadata":{}},{"cell_type":"code","source":"import json\n\nwith open(\"/kaggle/input/UBC-OCEAN/updated_image_ids.json\") as fo:\n    updated_image_ids = json.load(fo)\n\nls_subset = [os.path.join(DATASET_IMAGES, f'{idx}.png') for idx in updated_image_ids]\nprint(f\"updated images: {len(ls_subset)}\")","metadata":{"execution":{"iopub.status.busy":"2023-11-18T21:06:45.859604Z","iopub.execute_input":"2023-11-18T21:06:45.860036Z","iopub.status.idle":"2023-11-18T21:06:45.868971Z","shell.execute_reply.started":"2023-11-18T21:06:45.859996Z","shell.execute_reply":"2023-11-18T21:06:45.868271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Run the cutting in parallel with multiple workers","metadata":{}},{"cell_type":"code","source":"from tqdm.auto import tqdm\nfrom joblib import Parallel, delayed\n\n!mkdir -p train_tiles\n    \n_= Parallel(n_jobs=3)(\n    delayed(extract_prune_tiles)\n    (id_pimg, \"train_tiles\", size=2048, drop_thr=0.85, scale=0.25)\n    for id_pimg in tqdm(enumerate(ls_subset), total=len(ls_subset))\n)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-11-18T21:06:45.870768Z","iopub.execute_input":"2023-11-18T21:06:45.871315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Show some samples","metadata":{}},{"cell_type":"code","source":"ls = [p for p in glob.glob('train_tiles/*') if os.path.isdir(p)]\nprint(f\"found folders: {len(ls)}\")\nls = glob.glob('train_tiles/*/*.png')\nprint(f\"found images: {len(ls)}\")\n\nfig, axes = plt.subplots(nrows=5, ncols=5, figsize=(10, 10))\nfor i, p_img in enumerate(ls[:25]):\n    img = plt.imread(p_img)\n    ax = axes[i // 5, i % 5]\n    ax.imshow(img)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}