{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"},{"sourceId":6984590,"sourceType":"datasetVersion","datasetId":4014175}],"dockerImageVersionId":30558,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**This notebook is for transforming images to tiles.**\n\n**It needs to run twice to process and zip all images due to default kaggle output storage limitation.**\n\n**Final tile size is 256 × 256, reduced from original 2048 × 2048.**","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-11-06T05:57:57.483377Z","iopub.execute_input":"2023-11-06T05:57:57.483781Z","iopub.status.idle":"2023-11-06T05:57:58.131388Z","shell.execute_reply.started":"2023-11-06T05:57:57.48375Z","shell.execute_reply":"2023-11-06T05:57:58.130019Z"}}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-11-17T14:41:29.878463Z","iopub.execute_input":"2023-11-17T14:41:29.878874Z","iopub.status.idle":"2023-11-17T14:41:30.578328Z","shell.execute_reply.started":"2023-11-17T14:41:29.878841Z","shell.execute_reply":"2023-11-17T14:41:30.576782Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# setup pyvips\n\n!sudo apt-get update\n!sudo apt-get install libvips-dev -y --no-install-recommends --download-only -o dir::cache='./'\n\n!mkdir ./libvips\n!mv ./archives/* ./libvips\n!rm -rf ./archives\n!ls ./libvips\n\n!yes | sudo dpkg -i ./libvips/*.deb\n\n!pip install pyvips\n!pip wheel pyvips\n!mkdir pyvips\n!mv *.whl ./pyvips","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-11-17T14:41:42.379852Z","iopub.execute_input":"2023-11-17T14:41:42.380509Z","iopub.status.idle":"2023-11-17T14:43:26.377835Z","shell.execute_reply.started":"2023-11-17T14:41:42.380472Z","shell.execute_reply":"2023-11-17T14:43:26.376406Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATASET_IMAGES = \"/kaggle/input/UBC-OCEAN/train_images\"","metadata":{"execution":{"iopub.status.busy":"2023-11-09T06:42:20.799615Z","iopub.execute_input":"2023-11-09T06:42:20.800486Z","iopub.status.idle":"2023-11-09T06:42:20.806179Z","shell.execute_reply.started":"2023-11-09T06:42:20.80044Z","shell.execute_reply":"2023-11-09T06:42:20.804917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport pyvips\nimport numpy as np\nfrom PIL import Image\n\nos.environ['VIPS_DISC_THRESHOLD'] = '9gb' #use disk caching instead of memory when the image exceeds 9GB\n\ndef extract_tiles(raw_dir, dest_dir, size: int = 2048, scale: float = 0.5, drop_thr: float = 0.7) -> list:\n    name, _ = os.path.splitext(os.path.basename(raw_dir)) #get image name\n    im = pyvips.Image.new_from_file(raw_dir) #load image\n    w = h = size\n    # https://stackoverflow.com/a/47581978/4521646\n    idxs = [(y, y + h, x, x + w) for y in range(0, im.height, h) for x in range(0, im.width, w)]\n    files = []\n    for k, (y, y_, x, x_) in enumerate(idxs):\n        tile = im.crop(x, y, min(w, im.width - x), min(h, im.height - y)).numpy()[..., :3]\n        \n        # increase tile size to (h,w) for edge tiles\n        if tile.shape[:2] != (h, w):\n            tile_ = tile\n            tile_size = (h, w) if tile.ndim == 2 else (h, w, tile.shape[2])\n            tile = np.zeros(tile_size, dtype=tile.dtype)\n            tile[:tile_.shape[0], :tile_.shape[1], ...] = tile_\n        \n        # skip the tile of which empty ratio exceeds drop_thr\n        mask_bg = np.sum(tile, axis=2) == 0\n        tile[mask_bg, :] = 255\n        mask_bg = np.mean(tile, axis=2) > 250\n        if np.sum(mask_bg) >= (np.prod(mask_bg.shape) * drop_thr):\n            #print(f\"skip almost empty tile: {k:06}_{int(x_ / w)}-{int(y_ / h)}\")\n            continue\n        \n        p_img = os.path.join(dest_dir, f\"{k:06}_{int(x_ / w)}-{int(y_ / h)}.png\")\n        # print(tile.shape, tile.dtype, tile.min(), tile.max())\n        new_size = int(size * scale), int(size * scale)\n        Image.fromarray(tile).resize(new_size, Image.LANCZOS).save(p_img)\n        files.append(p_img)\n    return files, idxs","metadata":{"_kg_hide-input":false,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-11-09T06:42:24.107229Z","iopub.execute_input":"2023-11-09T06:42:24.107978Z","iopub.status.idle":"2023-11-09T06:42:24.675222Z","shell.execute_reply.started":"2023-11-09T06:42:24.107931Z","shell.execute_reply":"2023-11-09T06:42:24.674326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# segment one image as an example\n\n!mkdir -p /kaggle/temp/images\n\ntiles_img, _ = extract_tiles(\n    os.path.join(DATASET_IMAGES, \"12244.png\"),\n    \"/kaggle/temp/images\", size=2048, scale=0.125\n)\n\n!ls -lh /kaggle/temp/images","metadata":{"_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-11-09T06:43:03.855742Z","iopub.execute_input":"2023-11-09T06:43:03.85617Z","iopub.status.idle":"2023-11-09T06:44:14.457712Z","shell.execute_reply.started":"2023-11-09T06:43:03.856137Z","shell.execute_reply":"2023-11-09T06:44:14.456253Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Show the image tiles with segmentations\n\nimport matplotlib.pyplot as plt\n\nnames = [os.path.splitext(os.path.basename(p_img))[0] for p_img in tiles_img]\npos = [name.split(\"_\")[-1] for name in names]\nidx_x, idx_y = zip(*[list(map(int, p.split(\"-\"))) for p in pos])\nnb_rows = len(set(idx_y))\nnb_cols = len(set(idx_x))\nprint(f\"{nb_rows=}\\n{nb_cols=}\")\n\nfig, axes = plt.subplots(nrows=nb_rows, ncols=nb_cols,\n    figsize=(nb_cols * 0.5, nb_rows * 0.5)\n)\nfor i in range(nb_rows):\n    for j in range(nb_cols):\n        axes[i, j].set_facecolor(\"blue\")\n        axes[i, j].set_xticklabels([])\n        axes[i, j].set_yticklabels([])\n\nfor p_img, x, y in zip(tiles_img, idx_x, idx_y):\n    img = plt.imread(p_img)\n    ax = axes[y - 1, x - 1]\n    ax.imshow(img)\nprint(f\"image size: {img.shape}\")\n\nplt.subplots_adjust(wspace=0, hspace=0)","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-11-09T06:44:25.808444Z","iopub.execute_input":"2023-11-09T06:44:25.808889Z","iopub.status.idle":"2023-11-09T06:44:38.150907Z","shell.execute_reply.started":"2023-11-09T06:44:25.808853Z","shell.execute_reply":"2023-11-09T06:44:38.14976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Randomly reduce num of tiles to the specified limit.\n\nimport random\n\ndef prune_tiles(files: list, max_samples: float = 1.0):\n    max_samples = max_samples if isinstance(max_samples, int) else int(len(files) * max_samples)\n    random.shuffle(files)\n    for file_path in files[max_samples:]:\n        os.remove(file_path)","metadata":{"execution":{"iopub.status.busy":"2023-11-07T05:42:52.94606Z","iopub.execute_input":"2023-11-07T05:42:52.946557Z","iopub.status.idle":"2023-11-07T05:42:52.955842Z","shell.execute_reply.started":"2023-11-07T05:42:52.946523Z","shell.execute_reply":"2023-11-07T05:42:52.954176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# extract tiles from an image and reduce num of tiles\n\nimport gc, time\n\ndef extract_prune_tiles(\n    idx_path_img, dest_dir: str = \"./train_tiles\", \n    size: int = 2048, scale: float = 0.125, drop_thr: float = 0.75, max_samples: float = None\n):\n    idx, raw_dir = idx_path_img\n    print(f\"processing #{idx}: {raw_dir}\")\n    name, _ = os.path.splitext(os.path.basename(raw_dir))\n    dest_dir = os.path.join(dest_dir, name)\n    os.makedirs(dest_dir, exist_ok=True)\n    tiles, _ = extract_tiles(raw_dir, dest_dir, size, scale, drop_thr)\n    if max_samples:\n        prune_tiles(tiles, max_samples)\n    gc.collect()\n    time.sleep(1)","metadata":{"execution":{"iopub.status.busy":"2023-11-07T06:19:14.876228Z","iopub.execute_input":"2023-11-07T06:19:14.876691Z","iopub.status.idle":"2023-11-07T06:19:14.884518Z","shell.execute_reply.started":"2023-11-07T06:19:14.876658Z","shell.execute_reply":"2023-11-07T06:19:14.883518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# process the train_images folder\n\nimport glob\nimport multiprocessing as mproc\nfrom tqdm.auto import tqdm\nfrom joblib import Parallel, delayed\n\n!mkdir -p train_tiles\n\nls = sorted(glob.glob(os.path.join(DATASET_IMAGES, '*.png')))\nprint(f\"found images: {len(ls)}\")\n\nls=ls[270:] #segment tasks into smaller chunk\n\n# this mothed uses an unordered queue, doesn't keep processing order\npool = mproc.Pool(3)\ntqdm_bar = tqdm(total=len(ls))\nfor _ in pool.imap_unordered(extract_prune_tiles, enumerate(ls)):\n    tqdm_bar.update()\npool.close()\npool.join()\n\n# # this mothed keeps the processing order, may get blocked by the longest job in batch\n# _= Parallel(n_jobs=os.cpu_count())(\n#     delayed(extract_prune_tiles)\n#     (id_pimg, \"train_tiles\", size=2048, drop_thr=0.75, scale=0.25)\n#     for id_pimg in tqdm(enumerate(ls))\n# )","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-11-07T06:19:18.795486Z","iopub.execute_input":"2023-11-07T06:19:18.795858Z","iopub.status.idle":"2023-11-07T06:20:23.454101Z","shell.execute_reply.started":"2023-11-07T06:19:18.79582Z","shell.execute_reply":"2023-11-07T06:20:23.4476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # show some samples\n\n# ls = [p for p in glob.glob('train_tiles/*') if os.path.isdir(p)]\n# print(f\"found folders: {len(ls)}\")\n# ls = glob.glob('train_tiles/*/*.png')\n# print(f\"found images: {len(ls)}\")\n\n# fig, axes = plt.subplots(nrows=5, ncols=5, figsize=(10, 10))\n# for i, p_img in enumerate(ls[:25]):\n#     img = plt.imread(p_img)\n#     ax = axes[i // 5, i % 5]\n#     ax.imshow(img)","metadata":{"execution":{"iopub.status.busy":"2023-11-05T17:12:27.314025Z","iopub.execute_input":"2023-11-05T17:12:27.314364Z","iopub.status.idle":"2023-11-05T17:12:27.341991Z","shell.execute_reply.started":"2023-11-05T17:12:27.314327Z","shell.execute_reply":"2023-11-05T17:12:27.340532Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# only keep train_tiles in output and delete all other files\n\nimport shutil\n\noutput_folder_path = '/kaggle/working'\n\nfor item in os.listdir(output_folder_path):\n    item_path = os.path.join(output_folder_path, item)\n    # check if the fold is 'train_tiles'\n    if os.path.isdir(item_path) and item != 'train_tiles':\n        shutil.rmtree(item_path)\n        print(f\"Deleted folder: {item_path}\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# zip output\n\nimport shutil\n\noutput_folder = '/kaggle/working/'\nzip_filename = '/kaggle/working/output'\nshutil.make_archive(zip_filename, 'zip', output_folder)\n\nprint(f\"Created zip file: {zip_filename}.zip\")","metadata":{},"execution_count":null,"outputs":[]}]}