{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"},{"sourceId":6774400,"sourceType":"datasetVersion","datasetId":3895136},{"sourceId":6844262,"sourceType":"datasetVersion","datasetId":3934666}],"dockerImageVersionId":30684,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-09T07:22:12.888Z","iopub.execute_input":"2024-04-09T07:22:12.888893Z","iopub.status.idle":"2024-04-09T07:22:12.944936Z","shell.execute_reply.started":"2024-04-09T07:22:12.888816Z","shell.execute_reply":"2024-04-09T07:22:12.943886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport cv2 as cv\n\nimport PIL.Image as Image\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\n\nimport tensorflow as tf\nfrom tensorflow.keras.layers import Dense, MaxPooling2D, Dropout, Conv2D, Flatten\nfrom tensorflow.keras.callbacks import EarlyStopping\nfrom keras.models import Sequential\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\nimport PIL\nPIL.Image.MAX_IMAGE_PIXELS = None","metadata":{"execution":{"iopub.status.busy":"2024-04-09T07:22:12.946929Z","iopub.execute_input":"2024-04-09T07:22:12.947916Z","iopub.status.idle":"2024-04-09T07:22:12.958037Z","shell.execute_reply.started":"2024-04-09T07:22:12.947864Z","shell.execute_reply":"2024-04-09T07:22:12.956654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv = pd.read_csv('/kaggle/input/UBC-OCEAN/train.csv')\npath_folder = \"/kaggle/input/UBC-OCEAN/train_images\"\nimg_files = [os.path.join(path_folder, str(f[1].iloc[0])+\".png\") for f in train_csv.iterrows()]\n\ntrain_csv['path'] = pd.Series(img_files).astype(str)","metadata":{"execution":{"iopub.status.busy":"2024-04-09T07:22:12.960006Z","iopub.execute_input":"2024-04-09T07:22:12.961304Z","iopub.status.idle":"2024-04-09T07:22:13.029882Z","shell.execute_reply.started":"2024-04-09T07:22:12.961259Z","shell.execute_reply":"2024-04-09T07:22:13.027734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Removing the WSI images\nistma_false = train_csv[train_csv[\"is_tma\"]==False]\nprint(len(istma_false))\nistma_false.tail()","metadata":{"execution":{"iopub.status.busy":"2024-04-09T07:22:13.034315Z","iopub.execute_input":"2024-04-09T07:22:13.034946Z","iopub.status.idle":"2024-04-09T07:22:13.059165Z","shell.execute_reply.started":"2024-04-09T07:22:13.034892Z","shell.execute_reply":"2024-04-09T07:22:13.057996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_size = 450\n\n# Split data into test and train\ntrain_df = train_csv[0:train_size]\ntest_df = train_csv[train_size:]","metadata":{"execution":{"iopub.status.busy":"2024-04-09T07:22:13.061624Z","iopub.execute_input":"2024-04-09T07:22:13.062532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(train_df))\ntrain_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(test_df))\ntest_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## pyvips","metadata":{}},{"cell_type":"code","source":"!pip install pyvips","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls /kaggle/input/pyvips-python-and-deb-package-gpu\n# intall the deb packages\n!yes | dpkg -i --force-depends /kaggle/input/pyvips-python-and-deb-package-gpu/linux_packages/archives/*.deb\n# install the python wrapper\n!pip install pyvips -f /kaggle/input/pyvips-python-and-deb-package-gpu/python_packages/ --no-index","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!sudo apt-get update\n!sudo apt-get install libvips-dev -y --no-install-recommends --download-only -o dir::cache='./'","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir ./libvips\n!mv ./archives/* ./libvips\n!rm -rf ./archives\n!ls ./libvips","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!yes | sudo dpkg -i ./libvips/*.deb","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pyvips","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Tiling","metadata":{}},{"cell_type":"code","source":"tile_size = (224, 224)  # Define the size of each tile\nbackground_threshold = 0.85  # Define the threshold for considering a tile as background\n\n\ndef process_image_in_tiles(image_path, output_path, tile_size, background_threshold, max_tiles, tiles, paths):\n    # Open the input image using Pyvips\n    input_image = pyvips.Image.new_from_file(image_path)\n\n    file_name = os.path.basename(image_path)\n    id, _ = os.path.splitext(file_name)\n\n    # Get the dimensions of the input image\n    image_width = input_image.width\n    image_height = input_image.height\n\n    # Calculate the number of tiles in the x and y directions\n    num_tiles_x = int(np.ceil(image_width / tile_size[0]))\n    num_tiles_y = int(np.ceil(image_height / tile_size[1]))\n\n    # Limit the number of tiles if it exceeds the maximum\n    num_tiles = min(max_tiles, num_tiles_x * num_tiles_y)\n    num_tiles_x = min(num_tiles_x, int(np.ceil(np.sqrt(num_tiles))))\n    num_tiles_y = min(num_tiles_y, int(np.ceil(num_tiles / num_tiles_x)))\n\n    # Segment the input image into tiles\n    tile_count = 0\n    for i in range(num_tiles_x):\n        for j in range(num_tiles_y):\n            # Calculate the coordinates of the current tile\n            left = i * tile_size[0]\n            upper = j * tile_size[1]\n            right = min(left + tile_size[0], image_width)\n            lower = min(upper + tile_size[1], image_height)\n\n            # Extract the current tile from the input image using Pyvips\n            tile = input_image.crop(left, upper, right - left, lower - upper)\n\n            # Convert the tile to a numpy array\n            tile_array = np.ndarray(buffer=tile.write_to_memory(),\n                                     dtype=np.uint8,\n                                     shape=[tile.height, tile.width, tile.bands])\n\n            # Convert the tile to a PIL image\n            tile_pil = Image.fromarray(tile_array)\n\n            # Calculate the percentage of background pixels\n            num_background_pixels = np.sum((tile_array == 0) | (tile_array == 255))  # Count black and white pixels\n            background_percentage = num_background_pixels / (tile_size[0] * tile_size[1])\n\n            # Check if the tile is predominantly background\n            if background_percentage >= background_threshold:\n                continue\n\n            # Save the tile to the output folder\n            tile_filename = f'tile_{i}_{j}.jpg'  # You can adjust the filename format as needed\n            tile_pil.save(os.path.join(output_path, tile_filename))\n            tile_count += 1\n\n            # Break out of the loop if maximum number of tiles reached\n            if tile_count >= max_tiles:\n                break\n        else:\n            continue  # Continue to the next iteration of outer loop if no break\n        break  # Break out of the outer loop if maximum number of tiles reached\n\n    print(f\"{id}: Image segmentation into tiles complete. Total tiles: {tile_count}\")\n    tiles.append(tile_count)\n    paths.append(output_path)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"parent_dir = '/kaggle/working/'\ndirectory = 'img_tiles'\n\npath = os.path.join(parent_dir, directory)\nos.makedirs(path, exist_ok=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"max_tiles_per_image = 3000\ntiles = []\npaths = []\n\n# Process each image in the dataset\nfor index, row in train_df.iterrows():\n    image_path = row.iloc[5]\n    label = row.iloc[1]\n    output_folder = os.path.join(path, os.path.splitext(os.path.basename(image_path))[0])\n    os.makedirs(output_folder, exist_ok=True)\n    process_image_in_tiles(image_path, output_folder, tile_size, background_threshold, max_tiles_per_image, tiles, paths)\n\ntrain_csv['totTile'] = pd.Series(tiles).astype(int)\ntrain_csv['tilePath'] = pd.Series(paths).astype(str)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tiles_test = []\npaths_test = []\n\n# Process each image in the dataset\nfor index, row in test_df.iterrows():\n    image_path = row.iloc[5]\n    label = row.iloc[1]\n    output_folder = os.path.join(path, os.path.splitext(os.path.basename(image_path))[0])\n    os.makedirs(output_folder, exist_ok=True)\n    process_image_in_tiles(image_path, output_folder, tile_size, background_threshold, max_tiles_per_image, tiles_test, paths_test)\n\ntest_csv['totTile'] = pd.Series(tiles_test).astype(int)\ntest_csv['tilePath'] = pd.Series(paths_test).astype(str)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.to_csv('trainTiles.csv')\ntest_df.to_csv('testTiles.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}