{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# <center><strong>Introduction</strong></center>\n<p>The train images are quite large and directly feeding the images to the neural network needs huge computational power and resizing the images will reduce the quality of the image. So, the best option which is remaining is dividing the images in small tiles.</p>","metadata":{}},{"cell_type":"code","source":"import os\nfrom glob import glob\nimport zipfile\nfrom typing import Optional\nfrom tqdm import tqdm\nimport random\nimport math\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport PIL\nfrom PIL import Image\nPIL.Image.MAX_IMAGE_PIXELS = None","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-11-03T06:49:19.770739Z","iopub.execute_input":"2023-11-03T06:49:19.771393Z","iopub.status.idle":"2023-11-03T06:49:19.780536Z","shell.execute_reply.started":"2023-11-03T06:49:19.771346Z","shell.execute_reply":"2023-11-03T06:49:19.779001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Intialization of parameters\nclass Config:\n    csv_path = '/kaggle/input/UBC-OCEAN/train.csv'\n    image_dir_path = '/kaggle/input/UBC-OCEAN/train_images'\n    tile_height = 512\n    tile_width = 512\n    tile_save_dir = 'images_tiles'\n    overlap_factor = 0.1\n    save = True\nconfig = Config()","metadata":{"execution":{"iopub.status.busy":"2023-11-03T06:49:19.784076Z","iopub.execute_input":"2023-11-03T06:49:19.784999Z","iopub.status.idle":"2023-11-03T06:49:19.797118Z","shell.execute_reply.started":"2023-11-03T06:49:19.784927Z","shell.execute_reply":"2023-11-03T06:49:19.795899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read metadata\ndf = pd.read_csv(config.csv_path)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-03T06:49:19.799176Z","iopub.execute_input":"2023-11-03T06:49:19.800521Z","iopub.status.idle":"2023-11-03T06:49:19.831985Z","shell.execute_reply.started":"2023-11-03T06:49:19.800468Z","shell.execute_reply":"2023-11-03T06:49:19.830954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Scatter plot of width and height of images\nsns.scatterplot(x=\"image_width\", y=\"image_height\", hue='label', data=df)\nplt.xlabel(\"Image Width\")\nplt.ylabel(\"Image Height\")\nplt.title(\"Dimension of Images\")\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-03T06:49:19.834082Z","iopub.execute_input":"2023-11-03T06:49:19.835257Z","iopub.status.idle":"2023-11-03T06:49:20.347027Z","shell.execute_reply.started":"2023-11-03T06:49:19.835201Z","shell.execute_reply":"2023-11-03T06:49:20.345395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <strong><center>Image to Tiles</center></strong>","metadata":{}},{"cell_type":"code","source":"def is_invalid_image(tile, threshold=0.7):\n    # Convert the tile to RGBA mode to handle transparency\n    tile = tile.convert('RGBA')\n    # Get the pixel data as a list\n    pixel_data = list(tile.getdata())\n    # Count the number of transparent pixels\n    num_empty_pixels = sum(1 for pixel in pixel_data if sum(pixel[:3]) == 0)\n    num_white_pixels = sum(1 for pixel in pixel_data if sum(pixel[:3]) > 720)\n    # Calculate the percentage of empty pixels and white pixels\n    empty_percentage = num_empty_pixels / len(pixel_data)\n    white_percentage = num_white_pixels / len(pixel_data)\n    # Check invalid or not\n    flag = (empty_percentage >= threshold) or (white_percentage >= threshold) or(empty_percentage+white_percentage >= threshold)\n    return flag","metadata":{"execution":{"iopub.status.busy":"2023-11-03T06:49:20.350723Z","iopub.execute_input":"2023-11-03T06:49:20.351128Z","iopub.status.idle":"2023-11-03T06:49:20.360995Z","shell.execute_reply.started":"2023-11-03T06:49:20.351095Z","shell.execute_reply":"2023-11-03T06:49:20.359424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def image_to_tiles(image_path: str, \n                   tile_height: Optional[int] = config.tile_height, \n                   tile_width: Optional[int] = config.tile_width, \n                   overlap_factor: Optional[float] = config.overlap_factor, \n                   save: Optional[bool] = config.save, \n                   save_dir: Optional[str] = config.tile_save_dir):\n    # Load the image\n    image = Image.open(image_path)\n    \n    # Get the dimensions of the image\n    image_width, image_height = image.size\n    \n    # Calculate the number of rows and columns of tiles to cover the entire image\n    overlap_height = int(tile_height * overlap_factor)\n    overlap_width = int(tile_width * overlap_factor)\n    \n    stride_height = tile_height - overlap_height\n    stride_width = tile_width - overlap_width\n    \n    num_rows = int(math.ceil((image_height + overlap_height) / stride_height))\n    num_cols = int(math.ceil((image_width + overlap_width) / stride_width))\n\n    # Create a directory to save the tiles\n    if save:\n        os.makedirs(save_dir, exist_ok=True)\n    \n    # Loop through the image and extract tiles with overlapping\n    image_tiles = []\n    for row in range(num_rows):\n        for col in range(num_cols):\n            left = max(0, col * stride_width)\n            upper = max(0, row * stride_height)\n            right = min(image_width, left + tile_width)\n            lower = min(image_height, upper + tile_height)\n\n            # Crop the tile from the image\n            tile = image.crop((left, upper, right, lower))\n\n            # Save the tile with a unique name\n            if is_invalid_image(tile):     # Check the image has valid data or not\n                pass\n            else:\n                # Check the image has same dimensions of other images, otherwise resize it\n                if tile.size[1] == tile_height and tile.size[0] == tile_width:\n                    tile = tile.resize((tile_width, tile_height))\n                # Save the image or store it into a list\n                if save:\n                    tile.save(os.path.join(save_dir, 'tile_{}_{}.png'.format(row, col)))\n                else:\n                    image_tiles.append(Image.fromarray(tile))\n    return np.array(image_tiles)","metadata":{"execution":{"iopub.status.busy":"2023-11-03T06:49:20.362712Z","iopub.execute_input":"2023-11-03T06:49:20.363047Z","iopub.status.idle":"2023-11-03T06:49:20.379023Z","shell.execute_reply.started":"2023-11-03T06:49:20.363019Z","shell.execute_reply":"2023-11-03T06:49:20.377536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create proper path of the image\ndf['image_id'] = df['image_id'].apply(lambda x: os.path.join(config.image_dir_path, str(x) + '.png'))\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-03T06:49:20.381394Z","iopub.execute_input":"2023-11-03T06:49:20.381854Z","iopub.status.idle":"2023-11-03T06:49:20.410231Z","shell.execute_reply.started":"2023-11-03T06:49:20.381819Z","shell.execute_reply":"2023-11-03T06:49:20.408777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Divide the first image into tiles\nimage_to_tiles(df['image_id'].to_list()[0])","metadata":{"execution":{"iopub.status.busy":"2023-11-03T06:49:20.411973Z","iopub.execute_input":"2023-11-03T06:49:20.412998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <strong><center>Convert Results to Zip</center></strong>","metadata":{}},{"cell_type":"code","source":"# Function to zip a folder\ndef zip_folder(folder_path, output_path):\n    with zipfile.ZipFile(output_path, 'w', zipfile.ZIP_DEFLATED) as zipf:\n        for root, _, files in os.walk(folder_path):\n            for file in files:\n                file_path = os.path.join(root, file)\n                # Add the file to the zip archive\n                zipf.write(file_path, os.path.relpath(file_path, folder_path))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Zip the result folder\nzip_folder('image_tiles', 'image_tiles.zip')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <strong><center>Plot Images</center></strong>","metadata":{}},{"cell_type":"code","source":"# Function to plot images\ndef plot_images(folder_path, images_per_row=4, n_images=32):\n    # Get a list of image file paths in the folder\n    image_files = [f for f in os.listdir(folder_path) if f.lower().endswith(('.png', '.jpg', '.jpeg', '.gif', '.bmp', '.tif', '.tiff'))]\n    image_files = random.sample(image_files, n_images)\n    # Create a Matplotlib figure to display the images\n    num_images = len(image_files)\n    num_rows = (num_images + images_per_row - 1) // images_per_row\n    plt.figure(figsize=(15, 3 * num_rows))  # Adjust the figure size as needed\n    for i, image_file in enumerate(image_files):\n        try:\n            image_path = os.path.join(folder_path, image_file)\n            image = Image.open(image_path)\n            plt.subplot(num_rows, images_per_row, i + 1)\n            plt.imshow(image)\n            plt.title(image_file)\n            plt.axis('off')\n        except Exception as e:\n            continue\n    plt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot images\nplot_images(config.tile_save_dir)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}