{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\nimport time\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# import cv2\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\nimport shutil\nimport matplotlib.pyplot as plt\nfrom PIL import Image\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","scrolled":true,"execution":{"iopub.status.busy":"2023-10-16T13:52:53.267694Z","iopub.execute_input":"2023-10-16T13:52:53.268096Z","iopub.status.idle":"2023-10-16T13:52:53.715269Z","shell.execute_reply.started":"2023-10-16T13:52:53.268066Z","shell.execute_reply":"2023-10-16T13:52:53.714104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install --user pyvips\n!apt-get update -y\n!apt install libvips -y","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"scrolled":true,"execution":{"iopub.status.busy":"2023-10-16T13:52:55.02307Z","iopub.execute_input":"2023-10-16T13:52:55.023541Z","iopub.status.idle":"2023-10-16T13:53:50.826679Z","shell.execute_reply.started":"2023-10-16T13:52:55.023511Z","shell.execute_reply":"2023-10-16T13:53:50.825125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Reference\nMayo Clinic - STRIP AI\nHow To Interact With Large .tif Files\nhttps://www.kaggle.com/code/dschettler8845/mcsai-how-to-interact-with-large-tif-files","metadata":{"execution":{"iopub.status.busy":"2023-10-16T01:07:42.357729Z","iopub.execute_input":"2023-10-16T01:07:42.358251Z","iopub.status.idle":"2023-10-16T01:07:42.367129Z","shell.execute_reply.started":"2023-10-16T01:07:42.358219Z","shell.execute_reply":"2023-10-16T01:07:42.365136Z"}}},{"cell_type":"code","source":"import pyvips\n# import openslide","metadata":{"execution":{"iopub.status.busy":"2023-10-16T13:53:50.829065Z","iopub.execute_input":"2023-10-16T13:53:50.829394Z","iopub.status.idle":"2023-10-16T13:53:51.143027Z","shell.execute_reply.started":"2023-10-16T13:53:50.829366Z","shell.execute_reply":"2023-10-16T13:53:51.141744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Root_DIR = os.path.abspath(os.path.join(os.getcwd(),\"..\",\"input\"))\nDATA_DIR = os.path.join(Root_DIR,\"UBC-OCEAN\")\nTRAIN_DIR = os.path.join(DATA_DIR, \"train_images\")\nTRAIN_CSV = os.path.join(DATA_DIR, \"train.csv\")\ntrain_df = pd.read_csv(TRAIN_CSV)\ntrain_df[\"image_path\"] = train_df[\"image_id\"].apply(lambda x: os.path.join(TRAIN_DIR,str(x)+\".png\"))\nprint(\"\\n... TRAINING DATAFRAME... \\n\")\ndisplay(train_df)","metadata":{"execution":{"iopub.status.busy":"2023-10-16T13:53:51.145833Z","iopub.execute_input":"2023-10-16T13:53:51.146605Z","iopub.status.idle":"2023-10-16T13:53:51.196262Z","shell.execute_reply.started":"2023-10-16T13:53:51.146565Z","shell.execute_reply":"2023-10-16T13:53:51.195184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 2023/10/12 Plan B\nDEMO_PATHS = train_df.sample(5).image_path.tolist()\nDEMO_DOWNSCALE_BY = 20\n\ndef get_file_size(image_path):\n    return os.path.getsize(image_path)\n\ndef load_scaled_down_slide(image_path, downsample_by=10, to_numpy=True):\n    image = pyvips.Image.new_from_file(image_path)\n    original_size = get_file_size(image_path) / (1024 * 1024) # 获取原始图像大小 M\n    image = image.resize(1 / downsample_by)\n    if to_numpy:\n        buffer = image.write_to_memory()\n        numpy_array = np.frombuffer(buffer, dtype=np.uint8)\n        return numpy_array.reshape(image.height, image.width, image.bands), original_size\n    return image, original_size","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2023-10-16T13:53:51.207656Z","iopub.execute_input":"2023-10-16T13:53:51.208062Z","iopub.status.idle":"2023-10-16T13:53:51.222479Z","shell.execute_reply.started":"2023-10-16T13:53:51.208026Z","shell.execute_reply":"2023-10-16T13:53:51.221694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"HOW TO SHRINK WSI WITHOUT CRASHING DUE TO OOM RAM","metadata":{}},{"cell_type":"code","source":"for DEMO_PATH in DEMO_PATHS:\n    t1 = time.time()\n    DEMO_DOWNSCALED_SLIDE, original_size = load_scaled_down_slide(DEMO_PATH, DEMO_DOWNSCALE_BY)\n    t2 = time.time()\n    print(f\"\\nTime taken for processing: {t2 - t1:.2f} seconds\")\n    \n    # 添加原始大小和处理后大小到标题\n    demo_size = original_size/ (DEMO_DOWNSCALE_BY * DEMO_DOWNSCALE_BY)\n    title = f\"{DEMO_PATH.split(DATA_DIR)[-1]} Original Size: {original_size:.2f} MB Demo_size: {demo_size:.2f} MB\"\n\n    # image = Image.fromarray(DEMO_DOWNSCALED_SLIDE)\n    # image.save(DEMO_PATH.split('/')[-1])\n    \n    plt.figure(figsize=(10, 10))\n    plt.imshow(DEMO_DOWNSCALED_SLIDE)\n    plt.title(title, fontweight=\"bold\")\n    plt.tight_layout()\n    plt.show()","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2023-10-16T05:24:10.53208Z","iopub.execute_input":"2023-10-16T05:24:10.532516Z","iopub.status.idle":"2023-10-16T05:32:26.564005Z","shell.execute_reply.started":"2023-10-16T05:24:10.532487Z","shell.execute_reply":"2023-10-16T05:32:26.562934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\nRESIZE TRAINING DATA INTO PNG LOW-RESOLUTION PNG DATASET (0.n X)","metadata":{}},{"cell_type":"code","source":"t1 = time.time()\nDWN_SCALE_FACTOR = 10\nos.makedirs(f\"/kaggle/working/UBC-downscaled-by-{DWN_SCALE_FACTOR}/train_images\", exist_ok=True)\nos.makedirs(f\"/kaggle/working/UBC-downscaled-by-{DWN_SCALE_FACTOR}/test_images\", exist_ok=True)\nprint(\"\\ntime: \",time.time()-t1)","metadata":{"execution":{"iopub.status.busy":"2023-10-16T13:53:51.223373Z","iopub.execute_input":"2023-10-16T13:53:51.223726Z","iopub.status.idle":"2023-10-16T13:53:51.237513Z","shell.execute_reply.started":"2023-10-16T13:53:51.223694Z","shell.execute_reply":"2023-10-16T13:53:51.236404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_scaled_down(image_path, downsample_by=10, to_numpy=True):\n    # if not os.path.exists(image_path):\n    #     print(f\"Error: Image file '{image_path}' not found.\")\n    #     return None      \n    # image = pyvips.Image.new_from_file(image_path)\n    \n    # if image.width == 0 or image.height == 0:\n    #     print(f\"Error: Image dimensions are zero for '{image_path}'.\")\n    #     return None      \n    # image = image.resize(1 / downsample_by)\n    try:\n        # 尝试加载图像\n        image = pyvips.Image.new_from_file(image_path)\n        image = image.resize(1 / downsample_by)\n        # 其他处理代码\n    except Exception as e:\n        print(f\"Error loading image: {e}\")\n        \n    if to_numpy:\n        buffer = image.write_to_memory()\n        if len(buffer) == 0:\n            print(f\"Error: Empty buffer for '{image_path}'.\")\n            return None\n\n        numpy_array = np.frombuffer(buffer, dtype=np.uint8)\n        if numpy_array.size == 0:\n            print(f\"Error: Empty numpy array for '{image_path}'.\")\n            return None\n\n        return numpy_array.reshape(image.height, image.width, image.bands)\n\n    return image","metadata":{"execution":{"iopub.status.busy":"2023-10-16T05:32:58.98393Z","iopub.execute_input":"2023-10-16T05:32:58.98433Z","iopub.status.idle":"2023-10-16T05:32:58.991756Z","shell.execute_reply.started":"2023-10-16T05:32:58.984301Z","shell.execute_reply":"2023-10-16T05:32:58.990245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plan A\nnums = 0\nt1 = time.time()\ntotal_num = len(train_df)\nthreshold = 100#000000  # 阈值，以字节为单位（100MB）\n\nfor img_path in train_df.image_path.tolist():\n    nums+=1\n    dst_path = img_path.replace(\"input/UBC-OCEAN\", \n                                f\"working/UBC-downscaled-by-{DWN_SCALE_FACTOR}\")#\\\n#                         .replace(\".tif\", \n#                                  \".png\")\n    file_size = get_file_size(img_path) / (1024 * 1024) # 获取原始图像大小 M\n    \n    if file_size > threshold*20:\n        tmp_img = load_scaled_down(img_path, downsample_by=20)\n    elif file_size > threshold*15:\n        tmp_img = load_scaled_down(img_path, downsample_by=15)\n    elif file_size > threshold*10:\n        tmp_img = load_scaled_down(img_path, downsample_by=10)\n    elif file_size > threshold*5:\n        tmp_img = load_scaled_down(img_path, downsample_by=5)\n    else:\n        tmp_img = load_scaled_down(img_path, downsample_by=2)\n        \n    #to_numpy=True    \n    image = Image.fromarray(tmp_img)\n    image.save(dst_path,optimize=True)\n#     cv2.imwrite(dst_path, tmp_img)\n    \n    #to_numpy=False\n#     tmp_img.write_to_file(dst_path) \n    print(\"\\r[{}] processing [{}/{}]\".format(dst_path, nums, total_num), end=\"\")  \n    \nprint(\"time to resize dataset: \", time.time()-t1)","metadata":{"execution":{"iopub.status.busy":"2023-10-16T07:36:00.199484Z","iopub.execute_input":"2023-10-16T07:36:00.201563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"On my computer, when 'to_numpy' is set to either 'True' or 'False', an error occurs:\n'/home/woodman/Jupyter/wm/Kaggle/ubc/dataset/train_images/8279.png'. (x.png)\n\nThis means that the pyvips image failed to load successfully.","metadata":{}},{"cell_type":"code","source":"# Plan B\n\nImage.MAX_IMAGE_PIXELS = 4294967296\nthreshold = 100000000  # 100MB\nnums = 0\ntotal_num = len(train_df)\nDATA_DIR = os.path.join(Root_DIR,\"working\",f\"UBC-downscaled-by-{DWN_SCALE_FACTOR}\",\"train_images\")\nDATA_list = os.listdir(DATA_DIR)\n\nt1 = time.time()\n\nfor img_path in train_df.image_path.tolist():\n    nums+=1\n    t2 = time.time()\n    dst_path = img_path.replace(\"input/UBC-OCEAN\", \n                                f\"working/UBC-downscaled-by-{DWN_SCALE_FACTOR}\")\n    filename = img_path.split('/')[-1]\n    \n    if filename in DATA_list:\n        continue\n    else:\n        file_size = os.path.getsize(img_path)\n        img = Image.open(img_path)\n        width, height = img.size\n   \n        if file_size > threshold*20:\n            img = img.resize((width//20, height//20),Image.ANTIALIAS)\n        elif file_size > threshold*15:\n            img = img.resize((width//15, height//15),Image.ANTIALIAS)\n        elif file_size > threshold*10:\n            img = img.resize((width//10, height//10),Image.ANTIALIAS)\n        elif file_size > threshold*5:\n            img = img.resize((width//5, height//5),Image.ANTIALIAS)\n        else:\n            img = img.resize((width//2, height//2),Image.ANTIALIAS)\n            \n        img.save(dst_path,optimize=True)#,format=\"PNG\"\n\n    print(\"\\r[{}] processing [{}/{}] [{:.2f}]s\".format(dst_path, nums, total_num, time.time()-t2), end=\"\")  \n    \nprint(\"time to resize dataset: \", time.time()-t1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!du -sh /kaggle/working/UBC-downscaled-by-10/train_images\n!ls -lrSh /kaggle/working/UBC-downscaled-by-10/train_images/*","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2023-10-16T05:11:34.324635Z","iopub.status.idle":"2023-10-16T05:11:34.325072Z","shell.execute_reply.started":"2023-10-16T05:11:34.324846Z","shell.execute_reply":"2023-10-16T05:11:34.324872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!zip -qr UBC.zip /kaggle/working/UBC-downscaled-by-10/train_images","metadata":{"execution":{"iopub.status.busy":"2023-10-16T04:59:26.498617Z","iopub.execute_input":"2023-10-16T04:59:26.499148Z","iopub.status.idle":"2023-10-16T05:02:18.504186Z","shell.execute_reply.started":"2023-10-16T04:59:26.499109Z","shell.execute_reply":"2023-10-16T05:02:18.502082Z"},"trusted":true},"execution_count":null,"outputs":[]}]}