{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"},{"sourceId":6774400,"sourceType":"datasetVersion","datasetId":3895136},{"sourceId":6984590,"sourceType":"datasetVersion","datasetId":4014175}],"dockerImageVersionId":30558,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!ls /kaggle/input/pyvips-python-and-deb-package\n# intall the deb packages\n!dpkg -i --force-depends /kaggle/input/pyvips-python-and-deb-package/linux_packages/archives/*.deb\n# install the python wrapper\n!pip install pyvips -f /kaggle/input/pyvips-python-and-deb-package/python_packages/ --no-index","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-11-20T08:26:41.661777Z","iopub.execute_input":"2023-11-20T08:26:41.662334Z","iopub.status.idle":"2023-11-20T08:28:10.930722Z","shell.execute_reply.started":"2023-11-20T08:26:41.662287Z","shell.execute_reply":"2023-11-20T08:28:10.9292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os, glob\nimport pyvips\nimport numpy as np\nfrom PIL import Image\n\nos.environ['VIPS_DISC_THRESHOLD'] = '9gb'\n\nDATASET_IMAGES = \"/kaggle/input/UBC-OCEAN/train_images\"\nDATASET_MASKS = \"/kaggle/input/ubc-ovarian-cancer-competition-supplemental-masks\"\n\n!mkdir -p /kaggle/temp/images\n!mkdir -p /kaggle/temp/annotations\n!mkdir -p /kaggle/temp/masks\n\n!cp /kaggle/input/UBC-OCEAN/train.csv .","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-11-20T08:28:10.934497Z","iopub.execute_input":"2023-11-20T08:28:10.935177Z","iopub.status.idle":"2023-11-20T08:28:15.869004Z","shell.execute_reply.started":"2023-11-20T08:28:10.935127Z","shell.execute_reply":"2023-11-20T08:28:15.867559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Supplement with segm masks\n\nadjusted version for segmentations, see:\n- https://www.kaggle.com/competitions/UBC-OCEAN/discussion/455890\n- https://www.kaggle.com/datasets/sohier/ubc-ovarian-cancer-competition-supplemental-masks\n\nMask info - The masks use the following color codes:\n- Red: Tumor\n- Green: Stroma (healthy tissue)\n- Blue: Necrosis (dead or dying non-cancerous tissue)\n\nVisual\n```py\nblended_image = Image.blend(thumbnail_img, mask_image, alpha=0.3)\n```","metadata":{}},{"cell_type":"code","source":"ls_masks = glob.glob(os.path.join(DATASET_MASKS, \"*.png\"))\nprint(f\"found masks: {len(ls_masks)}\")","metadata":{"execution":{"iopub.status.busy":"2023-11-19T12:53:51.031747Z","iopub.execute_input":"2023-11-19T12:53:51.032226Z","iopub.status.idle":"2023-11-19T12:53:51.083725Z","shell.execute_reply.started":"2023-11-19T12:53:51.032183Z","shell.execute_reply":"2023-11-19T12:53:51.082824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Decompose image to tiles/grid 🖽","metadata":{}},{"cell_type":"code","source":"def extract_image_tiles(\n    p_img, folder, size: int = 2048, scale: float = 0.5,\n    drop_thr: float = 0.85, inds = None\n) -> list:\n    name, _ = os.path.splitext(os.path.basename(p_img))\n    im = pyvips.Image.new_from_file(p_img)\n    w = h = size\n    if not inds:\n        # https://stackoverflow.com/a/47581978/4521646\n        inds = [(y, y + h, x, x + w)\n                for y in range(0, im.height, h)\n                for x in range(0, im.width, w)]\n    files, idxs, k = [], [], 0\n    for idx in inds:\n        y, y_, x, x_ = idx\n        # https://libvips.github.io/pyvips/vimage.html#pyvips.Image.crop\n        tile = im.crop(x, y, min(w, im.width - x), min(h, im.height - y)).numpy()[..., :3]\n        if drop_thr is not None:\n            mask_bg = np.sum(tile, axis=2) == 0\n            if np.sum(mask_bg) >= (np.prod(mask_bg.shape) * drop_thr):\n                #print(f\"skip almost empty tile: {k:06}_{int(x_ / w)}-{int(y_ / h)}\")\n                continue\n        if tile.shape[:2] != (h, w):\n            tile_ = tile\n            tile_size = (h, w) if tile.ndim == 2 else (h, w, tile.shape[2])\n            tile = np.zeros(tile_size, dtype=tile.dtype)\n            tile[:tile_.shape[0], :tile_.shape[1], ...] = tile_\n        p_img = os.path.join(folder, f\"{k:05}_{int(x_ / w)}-{int(y_ / h)}.png\")\n        # print(tile.shape, tile.dtype, tile.min(), tile.max())\n        new_size = int(size * scale), int(size * scale)\n        Image.fromarray(tile).resize(new_size, Image.LANCZOS).save(p_img)\n        files.append(p_img)\n        idxs.append(idx)\n        k += 1\n    return files, idxs","metadata":{"execution":{"iopub.status.busy":"2023-11-19T12:53:51.085625Z","iopub.execute_input":"2023-11-19T12:53:51.086201Z","iopub.status.idle":"2023-11-19T12:53:51.149759Z","shell.execute_reply.started":"2023-11-19T12:53:51.086168Z","shell.execute_reply":"2023-11-19T12:53:51.148995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tiles_img, idxs = extract_image_tiles(\n    os.path.join(DATASET_IMAGES, \"1020.png\"),\n    \"/kaggle/temp/images\", size=1024, scale=0.5,\n)\ntiles_seg, _ = extract_image_tiles(\n    os.path.join(DATASET_MASKS, \"1020.png\"),\n    \"/kaggle/temp/annotations\", size=1024, scale=0.5,\n    drop_thr=None, inds=idxs\n)\nprint(f\"tiles_img={len(tiles_img)}\")\nprint(f\"tiles_seg={len(tiles_seg)}\")\n\n# !ls -lh /kaggle/temp/images","metadata":{"_kg_hide-output":false,"execution":{"iopub.status.busy":"2023-11-19T12:53:51.152313Z","iopub.execute_input":"2023-11-19T12:53:51.152871Z","iopub.status.idle":"2023-11-19T12:56:46.264621Z","shell.execute_reply.started":"2023-11-19T12:53:51.152843Z","shell.execute_reply":"2023-11-19T12:56:46.263088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Show the image tiles with segmentations","metadata":{}},{"cell_type":"code","source":"from PIL import Image\nimport matplotlib.pyplot as plt\n\nnames = [os.path.splitext(os.path.basename(p_img))[0] for p_img in tiles_img]\npos = [name.split(\"_\")[-1] for name in names]\nidx_x, idx_y = zip(*[list(map(int, p.split(\"-\"))) for p in pos])\nnb_rows = len(set(idx_y))\nnb_cols = len(set(idx_x))\nprint(f\"{nb_rows=}\\n{nb_cols=}\")\n\nfig, axes = plt.subplots(nrows=nb_rows, ncols=nb_cols,\n    figsize=(nb_cols * 0.5, nb_rows * 0.5)\n)\nfor p_img, p_seg, x, y in zip(tiles_img, tiles_seg, idx_x, idx_y):\n    blended_tile = Image.blend(Image.open(p_img), Image.open(p_seg), alpha=0.3)\n    ax = axes[y - 1, x - 1]\n    ax.imshow(blended_tile)\n#print(f\"image size: {img.shape}\")\n\nfor i in range(nb_rows):\n    for j in range(nb_cols):\n        axes[i, j].set_xticklabels([])\n        axes[i, j].set_yticklabels([])\n\nplt.subplots_adjust(wspace=0, hspace=0)\n# # fig.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2023-11-19T12:56:46.26603Z","iopub.execute_input":"2023-11-19T12:56:46.266396Z","iopub.status.idle":"2023-11-19T12:58:51.445295Z","shell.execute_reply.started":"2023-11-19T12:56:46.266367Z","shell.execute_reply":"2023-11-19T12:58:51.4442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Conver RGB annotation to labels","metadata":{}},{"cell_type":"code","source":"def convert_rgb_to_labels(img_path: str, folder: str):\n    name = os.path.basename(img_path)\n    img = np.array(Image.open(img_path))\n    #plt.imshow(img)\n    bg = np.ones((img.shape[0], img.shape[1], 1)) * 128\n    stack = np.concatenate((bg, img), axis=2)\n    mask = np.argmax(stack, axis=2).astype(np.uint8)\n    #print(np.unique(mask))\n    #plt.imshow(mask)\n    img_path = os.path.join(folder, name)\n    Image.fromarray(mask).save(img_path) \n    return img_path\n\ntiles_mask = [\n    convert_rgb_to_labels(p, \"/kaggle/temp/masks\")\n    for p in tiles_seg] ","metadata":{"execution":{"iopub.status.busy":"2023-11-19T12:58:51.447062Z","iopub.execute_input":"2023-11-19T12:58:51.448104Z","iopub.status.idle":"2023-11-19T12:59:00.484318Z","shell.execute_reply.started":"2023-11-19T12:58:51.44807Z","shell.execute_reply":"2023-11-19T12:59:00.482518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Show the image tiles with","metadata":{}},{"cell_type":"code","source":"p_triple = list(zip(tiles_img, tiles_seg, tiles_mask)) \nfor p_img, p_seg, p_mask in p_triple[::150]:\n    fig, axarr = plt.subplots(ncols=3, figsize=(12, 5))\n    axarr[0].set_title(\"tissue image\")\n    axarr[0].imshow(plt.imread(p_img))\n    axarr[1].set_title(\"belnded image & annotation\")\n    axarr[1].imshow(Image.blend(Image.open(p_img), Image.open(p_seg), alpha=0.3))\n    axarr[2].set_title(\"tissue labels\")\n    axarr[2].imshow(np.array(Image.open(p_mask)), vmin=0, vmax=3)","metadata":{"execution":{"iopub.status.busy":"2023-11-19T13:02:01.535706Z","iopub.execute_input":"2023-11-19T13:02:01.537967Z","iopub.status.idle":"2023-11-19T13:02:04.859558Z","shell.execute_reply.started":"2023-11-19T13:02:01.537829Z","shell.execute_reply":"2023-11-19T13:02:04.855147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Export all image tiles¶","metadata":{}},{"cell_type":"code","source":"!mkdir -p train_images\n!mkdir -p train_annotations\n!mkdir -p train_masks","metadata":{"execution":{"iopub.status.busy":"2023-11-19T13:02:24.243056Z","iopub.execute_input":"2023-11-19T13:02:24.243409Z","iopub.status.idle":"2023-11-19T13:02:25.288128Z","shell.execute_reply.started":"2023-11-19T13:02:24.243383Z","shell.execute_reply":"2023-11-19T13:02:25.286684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def extract_tiles_masks(\n    idx_name,\n    folder_img: str = \"train_images\",\n    folder_seg: str = \"train_annotations\",\n    folder_mask: str = \"train_masks\",\n    size: int = 2048, scale: float = 0.25, drop_thr: float = 0.9\n) -> None:\n    idx, name = idx_name\n    print(f\"processing #{idx}: {name}\")\n    \n    folder_img = os.path.join(folder_img, name)\n    os.makedirs(folder_img, exist_ok=True)\n    folder_seg = os.path.join(folder_seg, name)\n    os.makedirs(folder_seg, exist_ok=True)\n    folder_mask = os.path.join(folder_mask, name)\n    os.makedirs(folder_mask, exist_ok=True)\n    \n    _, idxs = extract_image_tiles(\n        os.path.join(DATASET_IMAGES, f\"{name}.png\"),\n        folder_img, size=size, scale=scale,\n        drop_thr=drop_thr,\n    )\n    tiles_seg, _ = extract_image_tiles(\n        os.path.join(DATASET_MASKS, f\"{name}.png\"),\n        folder_seg, size=size, scale=scale,\n        drop_thr=None, inds=idxs,\n    )\n    tiles_mask = [\n        convert_rgb_to_labels(p, folder_mask) for p in tiles_seg\n    ]","metadata":{"execution":{"iopub.status.busy":"2023-11-19T13:02:36.995822Z","iopub.execute_input":"2023-11-19T13:02:36.996211Z","iopub.status.idle":"2023-11-19T13:02:37.007688Z","shell.execute_reply.started":"2023-11-19T13:02:36.996181Z","shell.execute_reply":"2023-11-19T13:02:37.005886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Run the cutting in parallel with multiple workers","metadata":{}},{"cell_type":"code","source":"from tqdm.auto import tqdm\nfrom joblib import Parallel, delayed\n\nnames = [os.path.splitext(os.path.basename(p))[0] for p in ls_masks]\n    \n_= Parallel(n_jobs=3)(\n    delayed(extract_tiles_masks)\n    (id_name, size=2048, drop_thr=0.85, scale=0.25)\n    for id_name in tqdm(enumerate(names), total=len(names))\n)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-11-19T13:10:53.657221Z","iopub.execute_input":"2023-11-19T13:10:53.657656Z","iopub.status.idle":"2023-11-19T13:22:20.878351Z","shell.execute_reply.started":"2023-11-19T13:10:53.657623Z","shell.execute_reply":"2023-11-19T13:22:20.875075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Show some samples","metadata":{}},{"cell_type":"code","source":"import random\n\nls = [p for p in glob.glob('train_images/*') if os.path.isdir(p)]\nprint(f\"found folders: {len(ls)}\")\nls_imgs = sorted(glob.glob('train_images/*/*.png'))\nls_segs = sorted(glob.glob('train_annotations/*/*.png'))\nls_masks = sorted(glob.glob('train_masks/*/*.png'))\nls_samples = list(zip(ls_imgs, ls_segs, ls_masks))\nrandom.shuffle(ls_samples)\nprint(f\"found images: {len(ls)}\")\n\nfor i, (p_img, p_ant, p_mask) in enumerate(ls_samples[:10]):\n    fig, axarr = plt.subplots(ncols=3, figsize=(12, 4))\n    axarr[0].set_title(\"tissue image\")\n    axarr[0].imshow(plt.imread(p_img))\n    axarr[1].set_title(\"RGB annotation\")\n    axarr[1].imshow(plt.imread(p_ant))\n    axarr[2].set_title(\"tissue labels\")\n    axarr[2].imshow(np.array(Image.open(p_mask)), vmin=0, vmax=3)","metadata":{"execution":{"iopub.status.busy":"2023-11-19T13:29:05.649494Z","iopub.execute_input":"2023-11-19T13:29:05.649964Z","iopub.status.idle":"2023-11-19T13:29:11.763211Z","shell.execute_reply.started":"2023-11-19T13:29:05.649897Z","shell.execute_reply":"2023-11-19T13:29:11.761775Z"},"trusted":true},"execution_count":null,"outputs":[]}]}