{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"},{"sourceId":6774400,"sourceType":"datasetVersion","datasetId":3895136},{"sourceId":6844262,"sourceType":"datasetVersion","datasetId":3934666},{"sourceId":7092658,"sourceType":"datasetVersion","datasetId":4087402},{"sourceId":7120688,"sourceType":"datasetVersion","datasetId":4107067},{"sourceId":7465291,"sourceType":"datasetVersion","datasetId":4276337},{"sourceId":7736460,"sourceType":"datasetVersion","datasetId":4521196}],"dockerImageVersionId":30673,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-02T19:11:00.219373Z","iopub.execute_input":"2024-04-02T19:11:00.220095Z","iopub.status.idle":"2024-04-02T19:11:01.19522Z","shell.execute_reply.started":"2024-04-02T19:11:00.220064Z","shell.execute_reply":"2024-04-02T19:11:01.194402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !pip install rl-benchmark","metadata":{"execution":{"iopub.status.busy":"2024-04-02T19:11:22.690131Z","iopub.execute_input":"2024-04-02T19:11:22.691227Z","iopub.status.idle":"2024-04-02T19:11:22.695265Z","shell.execute_reply.started":"2024-04-02T19:11:22.691194Z","shell.execute_reply":"2024-04-02T19:11:22.694261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from openslide import OpenSlide","metadata":{"execution":{"iopub.status.busy":"2024-04-02T19:11:23.177855Z","iopub.execute_input":"2024-04-02T19:11:23.178231Z","iopub.status.idle":"2024-04-02T19:11:23.182259Z","shell.execute_reply.started":"2024-04-02T19:11:23.178198Z","shell.execute_reply":"2024-04-02T19:11:23.181264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls /kaggle/input/pyvips-python-and-deb-package-gpu\n# intall the deb packages\n!yes | dpkg -i --force-depends /kaggle/input/pyvips-python-and-deb-package-gpu/linux_packages/archives/*.deb\n# install the python wrapper\n!pip install pyvips -f /kaggle/input/pyvips-python-and-deb-package-gpu/python_packages/ --no-index","metadata":{"execution":{"iopub.status.busy":"2024-04-02T19:11:23.213231Z","iopub.execute_input":"2024-04-02T19:11:23.213512Z","iopub.status.idle":"2024-04-02T19:11:27.270559Z","shell.execute_reply.started":"2024-04-02T19:11:23.213487Z","shell.execute_reply":"2024-04-02T19:11:27.269561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !pip install accelerate datasets evaluate transformers peft scikit-learn torch torchvision evaluate --quiet","metadata":{"execution":{"iopub.status.busy":"2024-04-02T19:11:27.273013Z","iopub.execute_input":"2024-04-02T19:11:27.273843Z","iopub.status.idle":"2024-04-02T19:11:27.277861Z","shell.execute_reply.started":"2024-04-02T19:11:27.273794Z","shell.execute_reply":"2024-04-02T19:11:27.277081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv('/kaggle/input/UBC-OCEAN/train.csv')\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-02T19:11:27.278873Z","iopub.execute_input":"2024-04-02T19:11:27.279148Z","iopub.status.idle":"2024-04-02T19:11:27.324884Z","shell.execute_reply.started":"2024-04-02T19:11:27.279125Z","shell.execute_reply":"2024-04-02T19:11:27.323968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dir_path = \"/kaggle/input/UBC-OCEAN/train_images/\"\nimg_path = []\nfor i in df_train['image_id']:\n    img = str(i) + \".png\"\n    img_path.append(os.path.join(dir_path, img))\ndf_train['image_path'] = img_path","metadata":{"execution":{"iopub.status.busy":"2024-04-02T19:11:27.327078Z","iopub.execute_input":"2024-04-02T19:11:27.32736Z","iopub.status.idle":"2024-04-02T19:11:27.336312Z","shell.execute_reply.started":"2024-04-02T19:11:27.327337Z","shell.execute_reply":"2024-04-02T19:11:27.335364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX = df_train[['image_id', 'image_width', 'image_height', 'is_tma', 'image_path']]\ny = df_train['label']\nX_train, X_test, y_train, y_test = train_test_split(X, y, random_state=0, train_size = .80)","metadata":{"execution":{"iopub.status.busy":"2024-04-02T19:11:27.337471Z","iopub.execute_input":"2024-04-02T19:11:27.337794Z","iopub.status.idle":"2024-04-02T19:11:28.427708Z","shell.execute_reply.started":"2024-04-02T19:11:27.337764Z","shell.execute_reply":"2024-04-02T19:11:28.426734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Number of rows in training:\", X_train.shape[0])\nprint(\"Number of rows in testing:\", X_test.shape[0])","metadata":{"execution":{"iopub.status.busy":"2024-04-02T19:11:28.429784Z","iopub.execute_input":"2024-04-02T19:11:28.430399Z","iopub.status.idle":"2024-04-02T19:11:28.435848Z","shell.execute_reply.started":"2024-04-02T19:11:28.430364Z","shell.execute_reply":"2024-04-02T19:11:28.434873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Segmenting images","metadata":{}},{"cell_type":"code","source":"from PIL import Image\nimport cv2","metadata":{"execution":{"iopub.status.busy":"2024-04-02T19:11:28.437125Z","iopub.execute_input":"2024-04-02T19:11:28.437409Z","iopub.status.idle":"2024-04-02T19:11:28.639228Z","shell.execute_reply.started":"2024-04-02T19:11:28.437364Z","shell.execute_reply":"2024-04-02T19:11:28.638376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"parent_dir = '/kaggle/working/'\ndirectory = 'img_tiles'\n\npath = os.path.join(parent_dir, directory) \nos.mkdir(path) ","metadata":{"execution":{"iopub.status.busy":"2024-04-02T19:11:28.640362Z","iopub.execute_input":"2024-04-02T19:11:28.64066Z","iopub.status.idle":"2024-04-02T19:11:28.645549Z","shell.execute_reply.started":"2024-04-02T19:11:28.640633Z","shell.execute_reply":"2024-04-02T19:11:28.644534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Define your dataset and dataloaders\n# # dir_path already has the path to train images\n# # path already has the path to output images\n\n\n# for ipath in df_train['image_path']:\n\n#     # Open the input image\n# #     input_image = Image.open(ipath)\n#     input_image = cv2.imread(ipath)\n    \n    \n#     tile_size = (224, 224)  # Define the size of each tile\n#     background_threshold = 0.9  # Define the threshold for considering a tile as background\n    \n#     file_name = os.path.basename(ipath)\n#     id, _ = os.path.splitext(file_name)\n\n#     # Get the dimensions of the input image\n#     if input_image is None:\n#         print(f\"Error: Unable to read image from {ipath}\")\n#     else:\n#         # Get the dimensions of the input image\n#         image_height, image_width, _ = input_image.shape\n\n#     # Calculate the number of tiles in the x and y directions\n#     num_tiles_x = image_width // tile_size[0]\n#     num_tiles_y = image_height // tile_size[1]\n    \n#     out_path = os.path.join(path, id)\n#     os.mkdir(out_path)\n\n#     # Ensure output folder exists\n#     os.makedirs(out_path, exist_ok=True)\n    \n\n#     # Segment the input image into tiles\n#     for i in range(num_tiles_x):\n#         for j in range(num_tiles_y):\n#             # Calculate the coordinates of the current tile\n#             left = i * tile_size[0]\n#             upper = j * tile_size[1]\n#             right = left + tile_size[0]\n#             lower = upper + tile_size[1]\n\n#             # Crop the current tile from the input image\n# #             tile = input_image.crop((left, upper, right, lower))\n            \n#             # Crop the current tile from the input image using OpenCV\n#             tile = input_image[upper:lower, left:right]\n\n#             # Convert the tile to a PIL image\n#             tile_pil = Image.fromarray(cv2.cvtColor(tile, cv2.COLOR_BGR2RGB))\n\n\n#             # Convert the tile to a numpy array\n#             tile_array = np.array(tile)\n\n#             # Calculate the percentage of background pixels\n#             num_background_pixels = np.sum((tile_array == 0) | (tile_array == 255))  # Count black and white pixels\n#             background_percentage = num_background_pixels / (tile_size[0] * tile_size[1])\n\n#             # Check if the tile is predominantly background\n#             if background_percentage >= background_threshold:\n#                 print(f\"Skipping tile ({i}, {j}) as it is predominantly background.\")\n#                 continue\n\n#             # Save the tile to the output folder\n# #             tile_filename = f'tile_{i}_{j}.jpg'  # You can adjust the filename format as needed\n# #             tile.save(os.path.join(out_path, tile_filename))\n#             tile_filename = f'tile_{i}_{j}.jpg'  # You can adjust the filename format as needed\n#             cv2.imwrite(os.path.join(out_path, tile_filename), tile)\n\n#     print(id + 'Image segmentation into tiles complete.')\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nos.environ[\"OPENCV_IO_MAX_IMAGE_PIXELS\"] = str(pow(2,40))\nimport cv2","metadata":{"execution":{"iopub.status.busy":"2024-04-02T19:12:08.650004Z","iopub.execute_input":"2024-04-02T19:12:08.65064Z","iopub.status.idle":"2024-04-02T19:12:08.655688Z","shell.execute_reply.started":"2024-04-02T19:12:08.650608Z","shell.execute_reply":"2024-04-02T19:12:08.654455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from PIL import Image\n\n# # Define your dataset and dataloaders\n# # dir_path already has the path to train images\n# # path already has the path to output images\n\n# max_tiles = 200  # Maximum number of tiles to extract\n# tile_size = (224, 224)  # Define the size of each tile \n# background_threshold = 0.9  # Define the threshold for considering a tile as background\n\n# for ipath in df_train['image_path']:\n#     # Open the input image\n# #     input_image = Image.open(open(ipath, 'rb'))\n#     input_image = cv2.imread(ipath)\n\n#     file_name = os.path.basename(ipath)\n#     id, _ = os.path.splitext(file_name)\n\n#     # Get the dimensions of the input image\n#     if input_image is None:\n#         print(f\"Error: Unable to read image from {ipath}\")\n#         continue  # Skip to the next image if unable to read\n#     else:\n#         # Get the dimensions of the input image\n#         image_height, image_width, _ = input_image.shape\n\n#     # Calculate the tile size based on the number of tiles required\n#     num_tiles = min(max_tiles, (image_width // tile_size[0]) * (image_height // tile_size[1]))\n\n#     # Calculate the number of tiles in the x and y directions\n#     num_tiles_x = min(image_width // tile_size[0], int(np.ceil(np.sqrt(num_tiles))))\n#     num_tiles_y = min(image_height // tile_size[1], int(np.ceil(np.sqrt(num_tiles))))\n\n#     out_path = os.path.join(path, id)\n#     os.makedirs(out_path, exist_ok=True)\n\n#     # Segment the input image into tiles\n#     tile_count = 0\n#     for i in range(num_tiles_x):\n#         for j in range(num_tiles_y):\n#             # Calculate the coordinates of the current tile\n#             left = i * tile_size[0]\n#             upper = j * tile_size[1]\n#             right = min(left + tile_size[0], image_width)  # Ensure right doesn't exceed image boundary\n#             lower = min(upper + tile_size[1], image_height)  # Ensure lower doesn't exceed image boundary\n\n#             # Crop the current tile from the input image using OpenCV\n#             tile = input_image[upper:lower, left:right]\n\n#             # Convert the tile to a numpy array\n#             tile_array = np.array(tile)\n\n#             # Calculate the percentage of background pixels\n#             num_background_pixels = np.sum((tile_array == 0) | (tile_array == 255))  # Count black and white pixels\n#             background_percentage = num_background_pixels / (tile_size[0] * tile_size[1])\n\n#             # Check if the tile is predominantly background\n#             if background_percentage >= background_threshold:\n#                 print(f\"Skipping tile ({i}, {j}) as it is predominantly background.\")\n#                 continue\n\n#             # Save the tile to the output folder\n#             tile_filename = f'tile_{i}_{j}.jpg'  # You can adjust the filename format as needed\n#             cv2.imwrite(os.path.join(out_path, tile_filename), tile)\n#             tile_count += 1\n\n#             # Break out of the loop if maximum number of tiles reached\n#             if tile_count >= max_tiles:\n#                 break\n#         else:\n#             continue  # Continue to the next iteration of outer loop if no break\n#         break  # Break out of the outer loop if maximum number of tiles reached\n\n#     print(f\"{id}: Image segmentation into tiles complete. Total tiles: {tile_count}\")","metadata":{"execution":{"iopub.status.busy":"2024-04-02T19:12:10.145184Z","iopub.execute_input":"2024-04-02T19:12:10.14555Z","iopub.status.idle":"2024-04-02T19:12:10.152142Z","shell.execute_reply.started":"2024-04-02T19:12:10.14552Z","shell.execute_reply":"2024-04-02T19:12:10.151205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!sudo apt-get update\n!sudo apt-get install libvips-dev -y --no-install-recommends --download-only -o dir::cache='./'","metadata":{"execution":{"iopub.status.busy":"2024-04-02T19:12:10.721973Z","iopub.execute_input":"2024-04-02T19:12:10.722365Z","iopub.status.idle":"2024-04-02T19:12:42.446499Z","shell.execute_reply.started":"2024-04-02T19:12:10.722337Z","shell.execute_reply":"2024-04-02T19:12:42.445334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir ./libvips\n!mv ./archives/* ./libvips\n!rm -rf ./archives\n!ls ./libvips","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-04-02T19:12:42.448565Z","iopub.execute_input":"2024-04-02T19:12:42.448878Z","iopub.status.idle":"2024-04-02T19:12:46.239738Z","shell.execute_reply.started":"2024-04-02T19:12:42.448852Z","shell.execute_reply":"2024-04-02T19:12:46.238592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!yes | sudo dpkg -i ./libvips/*.deb","metadata":{"execution":{"iopub.status.busy":"2024-04-02T19:12:46.24109Z","iopub.execute_input":"2024-04-02T19:12:46.24138Z","iopub.status.idle":"2024-04-02T19:13:13.765342Z","shell.execute_reply.started":"2024-04-02T19:12:46.241353Z","shell.execute_reply":"2024-04-02T19:13:13.764165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install pyvips\n!pip wheel pyvips\n!mkdir pyvips\n!mv *.whl ./pyvips","metadata":{"execution":{"iopub.status.busy":"2024-04-02T19:13:13.768301Z","iopub.execute_input":"2024-04-02T19:13:13.768685Z","iopub.status.idle":"2024-04-02T19:13:37.276111Z","shell.execute_reply.started":"2024-04-02T19:13:13.768648Z","shell.execute_reply":"2024-04-02T19:13:37.274854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport pyvips\nimport numpy as np\nfrom PIL import Image","metadata":{"execution":{"iopub.status.busy":"2024-04-02T19:13:37.277543Z","iopub.execute_input":"2024-04-02T19:13:37.277843Z","iopub.status.idle":"2024-04-02T19:13:37.690477Z","shell.execute_reply.started":"2024-04-02T19:13:37.277816Z","shell.execute_reply":"2024-04-02T19:13:37.689618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define your dataset and dataloaders\n# dir_path already has the path to train images\n# path already has the path to output images\n\ntile_size = (224, 224)  # Define the size of each tile\nbackground_threshold = 0.9  # Define the threshold for considering a tile as background\n\ndef process_image_in_tiles(image_path, output_path, tile_size, background_threshold, max_tiles):\n    # Open the input image using Pyvips\n    input_image = pyvips.Image.new_from_file(image_path)\n\n    file_name = os.path.basename(image_path)\n    id, _ = os.path.splitext(file_name)\n\n    # Get the dimensions of the input image\n    image_width = input_image.width\n    image_height = input_image.height\n\n    # Calculate the number of tiles in the x and y directions\n    num_tiles_x = int(np.ceil(image_width / tile_size[0]))\n    num_tiles_y = int(np.ceil(image_height / tile_size[1]))\n\n    # Limit the number of tiles if it exceeds the maximum\n    num_tiles = min(max_tiles, num_tiles_x * num_tiles_y)\n    num_tiles_x = min(num_tiles_x, int(np.ceil(np.sqrt(num_tiles))))\n    num_tiles_y = min(num_tiles_y, int(np.ceil(num_tiles / num_tiles_x)))\n\n    # Segment the input image into tiles\n    tile_count = 0\n    for i in range(num_tiles_x):\n        for j in range(num_tiles_y):\n            # Calculate the coordinates of the current tile\n            left = i * tile_size[0]\n            upper = j * tile_size[1]\n            right = min(left + tile_size[0], image_width)\n            lower = min(upper + tile_size[1], image_height)\n\n            # Extract the current tile from the input image using Pyvips\n            tile = input_image.crop(left, upper, right - left, lower - upper)\n\n            # Convert the tile to a numpy array\n            tile_array = np.ndarray(buffer=tile.write_to_memory(),\n                                     dtype=np.uint8,\n                                     shape=[tile.height, tile.width, tile.bands])\n\n            # Convert the tile to a PIL image\n            tile_pil = Image.fromarray(tile_array)\n\n            # Calculate the percentage of background pixels\n            num_background_pixels = np.sum((tile_array == 0) | (tile_array == 255))  # Count black and white pixels\n            background_percentage = num_background_pixels / (tile_size[0] * tile_size[1])\n\n            # Check if the tile is predominantly background\n            if background_percentage >= background_threshold:\n#                 print(f\"Skipping tile ({i}, {j}) as it is predominantly background.\")\n                continue\n\n            # Save the tile to the output folder\n            tile_filename = f'tile_{i}_{j}.jpg'  # You can adjust the filename format as needed\n            tile_pil.save(os.path.join(output_path, tile_filename))\n            tile_count += 1\n\n            # Break out of the loop if maximum number of tiles reached\n            if tile_count >= max_tiles:\n                break\n        else:\n            continue  # Continue to the next iteration of outer loop if no break\n        break  # Break out of the outer loop if maximum number of tiles reached\n\n    print(f\"{id}: Image segmentation into tiles complete. Total tiles: {tile_count}\")\n \n# Process each image in the dataset\nfor index, row in df_train.iterrows():\n    image_path = row['image_path']\n    output_folder = os.path.join(path, os.path.splitext(os.path.basename(image_path))[0])\n    os.makedirs(output_folder, exist_ok=True)\n    process_image_in_tiles(image_path, output_folder, tile_size, background_threshold, max_tiles=2000)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-02T19:13:37.691931Z","iopub.execute_input":"2024-04-02T19:13:37.692558Z","iopub.status.idle":"2024-04-03T03:40:13.133179Z","shell.execute_reply.started":"2024-04-02T19:13:37.692523Z","shell.execute_reply":"2024-04-03T03:40:13.125943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"hello\")","metadata":{"execution":{"iopub.status.busy":"2024-04-03T03:40:13.136848Z","iopub.execute_input":"2024-04-03T03:40:13.137248Z","iopub.status.idle":"2024-04-03T03:40:13.148982Z","shell.execute_reply.started":"2024-04-03T03:40:13.137209Z","shell.execute_reply":"2024-04-03T03:40:13.147859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"kaggle kernels output hellydhamesha/tiling -p /path/to/dest","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ### Dependencies\n# # Base Dependencies\n# import os\n# import pickle\n# import sys\n\n# # LinAlg / Stats / Plotting Dependencies\n# import h5py\n# import matplotlib.pyplot as plt\n# import numpy as np\n# import pandas as pd\n# from PIL import Image\n# import umap\n# import umap.plot\n# from tqdm import tqdm\n\n# # Torch Dependencies\n# import torch\n# import torch.multiprocessing\n# import torchvision\n# from torch.utils.data.dataset import Dataset\n# from torchvision import transforms\n# from pl_bolts.models.self_supervised import resnets\n# from pl_bolts.utils.semi_supervised import Identity\n# device = torch.device('cuda:0')\n# torch.multiprocessing.set_sharing_strategy('file_system')\n\n# # Model Architectures\n# from nn_encoder_arch.vision_transformer import vit_small\n# from nn_encoder_arch.resnet_trunc import resnet50_trunc_baseline\n\n\n# ### Helper Functions for Normalization + Loading in pytorch_lightning SSL encoder (for SimCLR)\n# def eval_transforms(pretrained=False):\n#     if pretrained:\n#         mean, std = (0.485, 0.456, 0.406), (0.229, 0.224, 0.225)\n#     else:\n#         mean, std = (0.5,0.5,0.5), (0.5,0.5,0.5)\n#     trnsfrms_val = transforms.Compose([transforms.ToTensor(), transforms.Normalize(mean = mean, std = std)])\n#     return trnsfrms_val\n\n\n# def torchvision_ssl_encoder(name: str, pretrained: bool = False, return_all_feature_maps: bool = False):\n#     pretrained_model = getattr(resnets, name)(pretrained=pretrained, return_all_feature_maps=return_all_feature_maps)\n#     pretrained_model.fc = Identity()\n#     return pretrained_model\n\n# ### Wrapper Classes for loading in patch datasets for BreastPathQ + BCSS (CRC100K uses the ImageFolder Dataset Class)\n# class CSVDataset_BreastPathQ(Dataset):    \n#     def __init__(self, dataroot, csv_path, transforms_eval=eval_transforms()):\n#         self.csv = pd.read_csv(csv_path)\n#         self.csv['img_path'] = dataroot+self.csv['slide'].astype(str) + \"_\" + self.csv['rid'].astype(str) + '.tif'\n#         self.transforms = transforms_eval\n        \n#     def __getitem__(self, index):\n#         img = Image.open(self.csv['img_path'][index])\n#         return self.transforms(img), self.csv['y'][index]\n    \n#     def __len__(self):\n#         return self.csv.shape[0]\n\n\n# class CSVDataset_BCSS(Dataset):    \n#     def __init__(self, dataset_csv, is_train=1, transforms_eval=eval_transforms()):\n#         self.csv = dataset_csv\n#         self.csv = self.csv[self.csv['train']==is_train]\n#         self.transforms = transforms_eval   \n        \n#     def __getitem__(self, index):\n#         img = Image.open(self.csv.index[index])\n#         return self.transforms(img), self.csv.iloc[index]['label']\n    \n#     def __len__(self):\n#         return self.csv.shape[0]\n\n# ### Functions for Loading + Saving + Visualizing Patch Embeddings\n# def save_embeddings(model, fname, dataloader, dataset=None, is_imagefolder=False, \n#                     save_patches=False, sprite_dim=128, overwrite=False):\n\n#     if os.path.isfile('%s.pkl' % fname) and (overwrite == False):\n#         return None\n\n#     embeddings, labels = [], []\n#     patches = []\n\n#     for batch, target in tqdm(dataloader):\n#         if save_patches:\n#             for img in batch:\n#                 patches.append(tensor2im(input_image=img).resize(sprite_dim))\n        \n#         with torch.no_grad():\n#             batch = batch.to(device)\n#             embeddings.append(model(batch).detach().cpu().numpy())\n#             labels.append(target.numpy())\n            \n            \n#     embeddings = np.vstack(embeddings)\n#     labels = np.vstack(labels).squeeze()\n    \n#     if is_imagefolder:\n#         id2label = dict(map(reversed, dataset.class_to_idx.items()))\n#         labels = np.array(list(map(id2label.get, labels.ravel())))\n\n#     asset_dict = {'embeddings': embeddings, 'labels': labels}\n#     if save_patches:\n#         asset_dict.update({'patches': patches})\n#     with open('%s.pkl' % (fname), 'wb') as handle:\n#         pickle.dump(asset_dict, handle, protocol=pickle.HIGHEST_PROTOCOL)\n\n        \n# def create_UMAP(library_path, enc_name, dataset, n=50, d=0.2):\n\n    \n#     path = os.path.join(library_path, '%s_%s.pkl' % (dataset, enc_name))\n#     with open(path, 'rb') as handle:\n#         asset_dict = pickle.load(handle)\n#         embeddings, labels = asset_dict['embeddings'], asset_dict['labels']\n#     mapper = umap.UMAP(n_neighbors=n, min_dist=d).fit(embeddings)\n#     fig = plt.figure(figsize=(10, 10), dpi=100)\n#     umap.plot.points(mapper, labels=labels, width=600, height=600)\n#     plt.tight_layout()\n#     plt.savefig(os.path.join(library_path, 'UMAPs', '%s_%s_umap_n%d_d%0.2f.jpg' % (dataset, enc_name, n, d)))\n\n\n# def create_embeddings(embeddings_dir, enc_name, dataset, save_patches=False, sprite_dim=128, \n#                       patch_datasets='path/to/patch/datasets', assets_dir ='./ckpts/',\n#                       disentangle=-1, stage=-1):\n#     print(\"Extracting Features for '%s' via '%s'\" % (dataset, enc_name))\n#     if enc_name == 'resnet50_trunc':\n#         model = resnet50_trunc_baseline(pretrained=True)\n#         eval_t = eval_transforms(pretrained=True)\n#     elif 'dino' in enc_name:\n#         ckpt_path = os.path.join(assets_dir, enc_name+'.pt')\n#         assert os.path.isfile(ckpt_path)\n#         model = vit_small(patch_size=16)\n#         state_dict = torch.load(ckpt_path, map_location=\"cpu\")['teacher']\n#         state_dict = {k.replace(\"module.\", \"\"): v for k, v in state_dict.items()}\n#         state_dict = {k.replace(\"backbone.\", \"\"): v for k, v in state_dict.items()}\n#         missing_keys, unexpected_keys = model.load_state_dict(state_dict, strict=False)\n#         #print(\"Missing Keys:\", missing_keys)\n#         #print(\"Unexpected Keys:\", unexpected_keys)\n#         eval_t = eval_transforms(pretrained=False)\n#     elif 'simclr' in enc_name:\n#         ckpt_path = os.path.join(assets_dir, enc_name+'.pt')\n#         assert os.path.isfile(ckpt_path)\n#         model = torchvision_ssl_encoder('resnet50', pretrained=True)\n#         missing_keys, unexpected_keys = model.load_state_dict(torch.load(ckpt_path), strict=False)\n#         #print(\"Missing Keys:\", missing_keys)\n#         #print(\"Unexpected Keys:\", unexpected_keys)\n#         eval_t = eval_transforms(pretrained=False)\n#     else:\n#         pass\n\n#     model = model.to(device)\n#     model.eval()\n\n#     if 'simclr' in enc_name or 'simsiam' in enc_name:\n#         _model = model\n#         model = lambda x: _model.forward(x)[0]\n#     elif 'dino' in enc_name:\n#         _model = model\n#         if stage == -1:\n#             model = _model\n#         else:\n#             model = lambda x: torch.cat([x[:, 0] for x in _model.get_intermediate_layers(x, stage)], dim=-1)\n\n#     if stage != -1:\n#         _stage = '_s%d' % stage\n#     else:\n#         _stage = ''\n    \n#     if dataset == 'crc100k':\n#         ### Train\n#         dataroot = os.path.join(patch_datasets, 'NCT-CRC-HE-100K/')\n#         dataset = torchvision.datasets.ImageFolder(dataroot, transform=eval_t)\n#         dataloader = torch.utils.data.DataLoader(dataset, batch_size=10, shuffle=False, num_workers=4)\n#         fname = os.path.join(embeddings_dir, 'crc100k_train_%s%s' % (enc_name, _stage))\n#         save_embeddings(model=model, fname=fname, dataloader=dataloader, dataset=dataset,\n#                         save_patches=save_patches, sprite_dim=sprite_dim, is_imagefolder=True)\n        \n#         ### Test\n#         dataroot = os.path.join(patch_datasets, 'CRC-VAL-HE-7K/')\n#         dataset = torchvision.datasets.ImageFolder(dataroot, transform=eval_t)\n#         dataloader = torch.utils.data.DataLoader(dataset, batch_size=1, shuffle=False, num_workers=4)\n#         fname = os.path.join(embeddings_dir, 'crc100k_val_%s%s' % (enc_name, _stage))\n#         save_embeddings(model=model, fname=fname, dataloader=dataloader, dataset=dataset,\n#                         save_patches=save_patches, sprite_dim=sprite_dim, is_imagefolder=True)\n\n#     elif dataset == 'crc100knonorm':\n#         ### Train\n#         dataroot = os.path.join(patch_datasets, 'NCT-CRC-HE-100K-NONORM/')\n#         dataset = torchvision.datasets.ImageFolder(dataroot, transform=eval_t)\n#         dataloader = torch.utils.data.DataLoader(dataset, batch_size=10, shuffle=False, num_workers=4)\n#         fname = os.path.join(embeddings_dir, 'crc100knonorm_train_%s%s' % (enc_name, _stage))\n#         save_embeddings(model=model, fname=fname, dataloader=dataloader, dataset=dataset,\n#                         save_patches=save_patches, sprite_dim=sprite_dim, is_imagefolder=True)\n        \n#         ### Test\n#         dataroot = os.path.join(patch_datasets, 'CRC-VAL-HE-7K/')\n#         dataset = torchvision.datasets.ImageFolder(dataroot, transform=eval_t)\n#         dataloader = torch.utils.data.DataLoader(dataset, batch_size=1, shuffle=False, num_workers=4)\n#         fname = os.path.join(embeddings_dir, 'crc100knonorm_val_%s%s' % (enc_name, _stage))\n#         save_embeddings(model=model, fname=fname, dataloader=dataloader, dataset=dataset,\n#                         save_patches=save_patches, sprite_dim=sprite_dim, is_imagefolder=True)\n\n#     elif dataset == 'breastpathq':\n#         train_dataroot = os.path.join(patch_datasets, 'BreastPathQ/breastpathq/datasets/train/')\n#         val_dataroot = os.path.join(patch_datasets, 'BreastPathQ/breastpathq/datasets/validation/')\n#         train_csv = os.path.join(patch_datasets, 'BreastPathQ/breastpathq/datasets/train_labels.csv')\n#         val_csv = os.path.join(patch_datasets, 'BreastPathQ/breastpathq/datasets/val_labels.csv')\n\n#         train_dataset = CSVDataset_BreastPathQ(dataroot=train_dataroot, csv_path=train_csv, transforms_eval=eval_t)\n#         train_dataloader = torch.utils.data.DataLoader(train_dataset, batch_size=1, shuffle=False, num_workers=4)\n#         val_dataset = CSVDataset_BreastPathQ(dataroot=val_dataroot, csv_path=val_csv, transforms_eval=eval_t)\n#         val_dataloader = torch.utils.data.DataLoader(val_dataset, batch_size=1, shuffle=False, num_workers=4)\n        \n#         train_fname = os.path.join(embeddings_dir, 'breastq_train_%s%s' % (enc_name, _stage))\n#         val_fname = os.path.join(embeddings_dir, 'breastq_val_%s%s' % (enc_name, _stage))\n#         save_embeddings(model=model, fname=train_fname, dataloader=train_dataloader, \n#                         save_patches=save_patches, sprite_dim=sprite_dim)\n#         save_embeddings(model=model, fname=val_fname, dataloader=val_dataloader, \n#                         save_patches=save_patches, sprite_dim=sprite_dim)\n\n    \n#     elif dataset == 'bcss':\n#         dataroot = os.path.join(patch_datasets, 'BCSS/40x/patches/All/')\n#         csv_path = os.path.join(patch_datasets, 'BCSS/40x/patches/summary.csv')\n        \n#         dataset_csv = pd.read_csv(csv_path, sep=' ')['filename,train'].str.split(',', expand=True).astype(int)\n#         dataset_csv.columns = ['label', 'train']\n#         dataset_csv = dataset_csv[dataset_csv['label'].isin([0,1,2,3])]\n#         dataset_csv.index = [os.path.join(dataroot, fname+'.png') for fname in dataset_csv.index]\n\n#         train_dataset = CSVDataset_BCSS(dataset_csv=dataset_csv, is_train=1, transforms_eval=eval_t)\n#         train_dataloader = torch.utils.data.DataLoader(train_dataset, batch_size=1, shuffle=False, num_workers=1)\n#         val_dataset = CSVDataset_BCSS(dataset_csv=dataset_csv, is_train=0, transforms_eval=eval_t)\n#         val_dataloader = torch.utils.data.DataLoader(val_dataset, batch_size=1, shuffle=False, num_workers=1)\n\n#         train_fname = os.path.join(embeddings_dir, 'bcss_train_%s%s' % (enc_name, _stage))\n#         val_fname = os.path.join(embeddings_dir, 'bcss_val_%s%s' % (enc_name, _stage))\n#         save_embeddings(model=model, fname=train_fname, dataloader=train_dataloader, \n#                         save_patches=save_patches, sprite_dim=sprite_dim)\n#         save_embeddings(model=model, fname=val_fname, dataloader=val_dataloader, \n#                         save_patches=save_patches, sprite_dim=sprite_dim)","metadata":{"execution":{"iopub.status.busy":"2024-04-02T19:11:58.675684Z","iopub.status.idle":"2024-04-02T19:11:58.676068Z","shell.execute_reply.started":"2024-04-02T19:11:58.675855Z","shell.execute_reply":"2024-04-02T19:11:58.675869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}