{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"}],"dockerImageVersionId":30587,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Tiles extraction from Whole slide Images using Pyvips and save it in labels-specific directories\nThis notebook focuses on a tile generation process essential for handling large-scale image datasets, frequently encountered in fields such as medical imaging. The primary goal is to efficiently preprocess high-resolution Whole Slide Images (WSI) into smaller, manageable tiles for downstream analysis and machine learning tasks.\n\n**Advantages Over Other Approaches:**\n\n**-Handling Large Image Dimensions (Overcoming Size Limitations):**\n\nThis approach addresses the challenge posed by image sizes beyond the constraints of traditional image processing libraries like OpenCV (cv2), which typically have limitations on image dimensions (e.g., 32,000x32,000 pixels). Unlike cv2, Pyvips facilitates the handling of WSIs that exceed such dimensions.\n\n**-Supporting WSIs in PNG Format:**\n\nMany datasets in medical imaging and pathology comprise Whole Slide Images in formats like PNG, which aren't directly supported by libraries such as OpenSlide. While OpenSlide can manage huge images, it does not support WSIs in PNG format. Pyvips enables PNG image processing tasks.\n\n","metadata":{}},{"cell_type":"code","source":"%%capture\n!apt-get update\n!apt-get install -y libvips\n!pip install pyvips","metadata":{"execution":{"iopub.status.busy":"2024-03-30T21:03:07.174181Z","iopub.execute_input":"2024-03-30T21:03:07.174614Z","iopub.status.idle":"2024-03-30T21:04:13.221962Z","shell.execute_reply.started":"2024-03-30T21:03:07.174579Z","shell.execute_reply":"2024-03-30T21:04:13.220579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %%capture\n# # to install vips locally\n# !dpkg -i --force-depends /kaggle/input/pyvips-local/libvips-apt/libvips-apt/*.deb >/dev/null 2>&1\n# !pip install /kaggle/input/pyvips-local/cffi-1.14.4-cp37-cp37m-manylinux1_x86_64.whl\n# !pip install /kaggle/input/pyvips-local/pycparser-2.20-py2.py3-none-any.whl\n# !pip install /kaggle/input/pyvips-local/pyvips-2.1.13-py2.py3-none-any.whl","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport pyvips\nimport numpy as np\nimport pandas as pd","metadata":{"execution":{"iopub.status.busy":"2024-03-30T21:04:18.163769Z","iopub.execute_input":"2024-03-30T21:04:18.164165Z","iopub.status.idle":"2024-03-30T21:04:18.953352Z","shell.execute_reply.started":"2024-03-30T21:04:18.164131Z","shell.execute_reply":"2024-03-30T21:04:18.952359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Function to generate tiles from images based on provided parameters\ndef generate_tiles(image_id, label, image_dir, output_dir, tile_size, threshold_mean, threshold_std):\n    # Create label-specific output directory\n    label_output_dir = os.path.join(output_dir, label)\n    os.makedirs(label_output_dir, exist_ok=True)\n\n    # Construct the path for the image using its ID\n    image_path = os.path.join(image_dir, f\"{image_id}.png\")\n    \n    # Check if the image exists\n    if not os.path.exists(image_path):\n        print(f\"Image not found: {image_path}\")\n        return\n\n    try:\n        # Read the input image using pyvips\n        img = pyvips.Image.new_from_file(image_path)\n\n        # Extract image dimensions (height, width)\n        height = img.height\n        width = img.width\n    except Exception as e:\n        print(f\"Error reading image {image_path}: {e}\")\n        return\n\n    # Calculate the number of rows and columns for tiles\n    rows_count = height // tile_size\n    cols_count = width // tile_size\n\n    # Iterate through rows and columns to generate tiles\n    for row_idx in range(rows_count):\n        for col_idx in range(cols_count):\n            # Calculate coordinates for each tile\n            x = col_idx * tile_size\n            y = row_idx * tile_size\n\n            # Extract the tile from the image\n            tile = img.crop(x, y, tile_size, tile_size)\n\n            # Define the name for the tiles\n            tile_name = f\"{image_id}_tile_{col_idx}_{row_idx}.png\"\n            \n            # Convert the tile to a NumPy array for mean and std calculation\n            tile_np = np.array(tile)\n\n            # Check if the tile is not mostly blank based on mean and std thresholds\n            if tile_np.mean() >= threshold_mean and tile_np.std() >= threshold_std:\n                # Define the path for the tiles\n                tile_path = os.path.join(label_output_dir, tile_name)\n                # Save the tile\n                tile.write_to_file(tile_path)","metadata":{"execution":{"iopub.status.busy":"2024-03-30T21:04:21.636769Z","iopub.execute_input":"2024-03-30T21:04:21.637376Z","iopub.status.idle":"2024-03-30T21:04:21.650764Z","shell.execute_reply.started":"2024-03-30T21:04:21.637337Z","shell.execute_reply":"2024-03-30T21:04:21.649621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define directory paths and CSV file\nimage_dir = \"/kaggle/input/UBC-OCEAN/train_images\"\noutput_dir = \"/kaggle/working/output_tiles\"\ncsv_file_path = \"/kaggle/input/UBC-OCEAN/train.csv\"\n\n# Read the CSV file containing image IDs and labels\ndata = pd.read_csv(csv_file_path)\n\n# Iterate through each row in the CSV file\nfor index, dataset_row in data.iterrows():\n#     if dataset_row['is_tma'] == True: #remove this line if you want to apply it on the entire train_images, but you need to have enough output capacity \n    # Extract image ID, label, and generate tiles for each entry\n        generate_tiles(dataset_row['image_id'], dataset_row['label'], image_dir, output_dir, tile_size=1024, threshold_mean=170, threshold_std=15)\n        break","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2024-03-30T21:04:25.659609Z","iopub.execute_input":"2024-03-30T21:04:25.659999Z","iopub.status.idle":"2024-03-30T21:05:33.298082Z","shell.execute_reply.started":"2024-03-30T21:04:25.659971Z","shell.execute_reply":"2024-03-30T21:05:33.296299Z"},"trusted":true},"execution_count":null,"outputs":[]}]}