{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Getting started with the PANDA dataset\n\nThis notebook shows a few methods to load and display images from the PANDA challenge dataset. The dataset consists of around 11.000 whole-slide images (WSI) of prostate biopsies from Radboud University Medical Center and the Karolinska Institute. \n","metadata":{}},{"cell_type":"markdown","source":"example code for m rcnn","metadata":{}},{"cell_type":"markdown","source":"> code to resize ground truth based on orginal image","metadata":{}},{"cell_type":"code","source":"mkdir masks","metadata":{"execution":{"iopub.status.busy":"2023-04-02T16:34:17.619781Z","iopub.execute_input":"2023-04-02T16:34:17.620247Z","iopub.status.idle":"2023-04-02T16:34:18.705484Z","shell.execute_reply.started":"2023-04-02T16:34:17.620208Z","shell.execute_reply":"2023-04-02T16:34:18.703712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"from PIL import Image\nimport os\nimport csv\nImage.MAX_IMAGE_PIXELS = None\ncount = 0\n\n# Set the directory where the images and the CSV file are located\ndirectory = \"/kaggle/input/isup-training/images\"\ngt_directory = '/kaggle/input/panda2/train_label_masks'\n# Load the CSV file containing the image IDs\nwith open(\"/kaggle/input/prostate-cancer-grade-assessment/train.csv\", \"r\") as file:\n    reader = csv.reader(file)\n    next(reader) # skip the header row\n    print('started')\n    for row in reader:\n        # Extract the image ID from the current row of the CSV file\n        image_id = row[0]\n        \n        # Construct the paths for the actual image and the ground truth image\n        actual_img_path = os.path.join(directory, f\"{image_id}.png\")\n        gt_img_path = os.path.join(gt_directory, f\"{image_id}.png\")\n        \n        # Check if the actual image file and the ground truth image file exist\n        if os.path.isfile(actual_img_path) and os.path.isfile(gt_img_path):\n            # Load actual image and ground truth image\n            actual_img = Image.open(actual_img_path)\n            gt_img = Image.open(gt_img_path)\n\n            # Get size of actual image and ground truth image\n            actual_size = actual_img.size\n            gt_size = gt_img.size\n\n            # Calculate scaling factor\n            scale_factor = actual_size[0] / gt_size[0]\n\n            # Resize ground truth image\n            #resized_gt_img = gt_img.resize((int(gt_size[0] * scale_factor), int(gt_size[1] * scale_factor)),resample=Image.NEAREST)\n            resized_gt_img = gt_img.resize(actual_size)\n            \n            # Save resized ground truth image\n            resized_gt_img.save(os.path.join('/kaggle/working/masks/',f\"{image_id}.png\"))\n            count +=1\n            print(image_id)\n            print(count)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"code to make the image and ground truth into tiles","metadata":{}},{"cell_type":"code","source":"mkdir tiles/masks","metadata":{"execution":{"iopub.status.busy":"2023-04-02T16:50:03.619153Z","iopub.execute_input":"2023-04-02T16:50:03.619612Z","iopub.status.idle":"2023-04-02T16:50:04.73136Z","shell.execute_reply.started":"2023-04-02T16:50:03.619551Z","shell.execute_reply":"2023-04-02T16:50:04.729738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_tiles(image, mask, tile_size, images_output_folder, masks_output_folder, image_filename, threshold=0.05):\n    height, width = image.shape[:2]\n\n    for y in range(0, height, tile_size):\n        for x in range(0, width, tile_size):\n            tile_image = image[y:y + tile_size, x:x + tile_size]\n            tile_mask = mask[y:y + tile_size, x:x + tile_size]\n\n            # Compute mean color for each channel\n            mean_color = np.mean(tile_image, axis=(0, 1))\n\n            # Calculate the mean color value across all channels\n            mean_color_value = np.mean(mean_color)\n\n            # Check if the mean color value is below the white threshold\n            if mean_color_value < 255 * (1 - threshold):\n                tile_image_path = os.path.join(images_output_folder, f'{image_filename}_tile_{y}_{x}.png')\n                tile_mask_path = os.path.join(masks_output_folder, f'{image_filename}_tile_{y}_{x}.png')\n                cv2.imwrite(tile_image_path, tile_image)\n                cv2.imwrite(tile_mask_path, tile_mask)\n\n\n\ndef process_images(images_folder, masks_folder, tile_size, images_output_folder, masks_output_folder):\n    image_files = os.listdir(images_folder)\n\n    for image_file in image_files:\n        if image_file.endswith('.png'):\n            image_path = os.path.join(images_folder, image_file)\n            mask_path = os.path.join(masks_folder, image_file)\n\n            image = cv2.imread(image_path, cv2.IMREAD_COLOR)\n            mask = cv2.imread(mask_path, cv2.IMREAD_COLOR)\n            image_filename = os.path.splitext(image_file)[0]\n\n            create_tiles(image, mask, tile_size, images_output_folder, masks_output_folder, image_filename)\n\n\n# Input parameters\nimages_folder = '/kaggle/input/isup-training/images/'\nmasks_folder = '/kaggle/working/masks'\ntile_size = 256\nimages_output_folder = '/kaggle/working/tiles/images'\nmasks_output_folder = '/kaggle/working/tiles/masks'\n\n# Process the images\nprocess_images(images_folder, masks_folder, tile_size, images_output_folder, masks_output_folder)\n","metadata":{"execution":{"iopub.status.busy":"2023-04-02T08:51:40.396594Z","iopub.execute_input":"2023-04-02T08:51:40.39715Z","iopub.status.idle":"2023-04-02T08:51:41.509006Z","shell.execute_reply.started":"2023-04-02T08:51:40.397102Z","shell.execute_reply":"2023-04-02T08:51:41.507411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport numpy as np\nimport os\n\n\ndef create_tiles(image, mask, tile_size, images_output_folder, masks_output_folder, image_filename, threshold=0.05):\n    height, width = image.shape[:2]\n\n    for y in range(0, height, tile_size):\n        for x in range(0, width, tile_size):\n            tile_image = image[y:y + tile_size, x:x + tile_size]\n            tile_mask = mask[y:y + tile_size, x:x + tile_size]\n\n            mean_color = np.mean(tile_image, axis=(0, 1))\n            mean_color_value = np.mean(mean_color)\n\n            if mean_color_value < 255 * (1 - threshold):\n                tile_image_path = os.path.join(images_output_folder, f'{image_filename}_tile_{y}_{x}.png')\n                tile_mask_path = os.path.join(masks_output_folder, f'{image_filename}_tile_{y}_{x}.png')\n\n                cv2.imwrite(tile_image_path, tile_image)\n                \n                # Check if the mask is a single-channel image and save it accordingly\n             \n                cv2.imwrite(tile_mask_path, tile_mask[:, :, 0])\n              \ndef process_images(images_folder, masks_folder, tile_size, images_output_folder, masks_output_folder):\n    image_files = os.listdir(images_folder)\n\n    for image_file in image_files:\n        if image_file.endswith('.png'):\n            image_path = os.path.join(images_folder, image_file)\n            mask_path = os.path.join(masks_folder, image_file)\n\n            image = cv2.imread(image_path, cv2.IMREAD_COLOR)\n            mask = cv2.imread(mask_path, cv2.IMREAD_COLOR)\n            image_filename = os.path.splitext(image_file)[0]\n\n            create_tiles(image, mask, tile_size, images_output_folder, masks_output_folder, image_filename)\n\n\n# Input parameters\nimages_folder = '/kaggle/input/isup-training/images/'\nmasks_folder = '/kaggle/working/masks'\ntile_size = 256\nimages_output_folder = '/kaggle/working/tiles/images'\nmasks_output_folder = '/kaggle/working/tiles/masks'\n\n# Process the images\nprocess_images(images_folder, masks_folder, tile_size, images_output_folder, masks_output_folder)","metadata":{"execution":{"iopub.status.busy":"2023-04-02T17:11:08.122752Z","iopub.execute_input":"2023-04-02T17:11:08.123212Z","iopub.status.idle":"2023-04-02T18:05:02.917816Z","shell.execute_reply.started":"2023-04-02T17:11:08.123174Z","shell.execute_reply":"2023-04-02T18:05:02.911818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tree /kaggle/working/tiles/images --filelimit=30","metadata":{"execution":{"iopub.status.busy":"2023-04-02T17:07:15.26618Z","iopub.execute_input":"2023-04-02T17:07:15.266724Z","iopub.status.idle":"2023-04-02T17:07:15.276609Z","shell.execute_reply.started":"2023-04-02T17:07:15.26667Z","shell.execute_reply":"2023-04-02T17:07:15.274502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ls /kaggle/working/tiles/images | head -600\n","metadata":{"execution":{"iopub.status.busy":"2023-04-02T18:13:34.253803Z","iopub.execute_input":"2023-04-02T18:13:34.255421Z","iopub.status.idle":"2023-04-02T18:13:36.844119Z","shell.execute_reply.started":"2023-04-02T18:13:34.255358Z","shell.execute_reply":"2023-04-02T18:13:36.842153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-04-02T13:27:58.333257Z","iopub.execute_input":"2023-04-02T13:27:58.334234Z","iopub.status.idle":"2023-04-02T13:27:58.343733Z","shell.execute_reply.started":"2023-04-02T13:27:58.334188Z","shell.execute_reply":"2023-04-02T13:27:58.341889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cp /kaggle/input/panda-2020-level-1-2/train_images/train_images/00412139e6b04d1e1cee8421f38f6e90_1.jpeg /kaggle/working/","metadata":{"execution":{"iopub.status.busy":"2023-04-02T08:56:35.110939Z","iopub.execute_input":"2023-04-02T08:56:35.111989Z","iopub.status.idle":"2023-04-02T08:56:36.209713Z","shell.execute_reply.started":"2023-04-02T08:56:35.111949Z","shell.execute_reply":"2023-04-02T08:56:36.208274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport numpy as np\n\n# Load the image\nimage_path = '/kaggle/input/panda-2020-level-1-2/train_images/train_images/00412139e6b04d1e1cee8421f38f6e90_1.jpeg'\nimage = cv2.imread(image_path, cv2.IMREAD_COLOR)\n\n# Convert the image to float32 for accurate normalization\nimage = image.astype(np.float32)\n\n# Calculate the mean and standard deviation of the image\nmean, std_dev = cv2.meanStdDev(image)\n\n# Normalize the image\nnormalized_image = (image - mean.reshape(1, 1, -1)) / std_dev.reshape(1, 1, -1)\nrescaled_image = cv2.normalize(normalized_image, None, alpha=0, beta=255, norm_type=cv2.NORM_MINMAX, dtype=cv2.CV_8U)\n# Save the normalized image\noutput_path = '/kaggle/working/img/rescaled_image1.png'\ncv2.imwrite(output_path, rescaled_image)\n","metadata":{"execution":{"iopub.status.busy":"2023-04-02T08:55:21.868915Z","iopub.execute_input":"2023-04-02T08:55:21.869361Z","iopub.status.idle":"2023-04-02T08:55:22.560906Z","shell.execute_reply.started":"2023-04-02T08:55:21.869325Z","shell.execute_reply":"2023-04-02T08:55:22.559641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow_hub as hub\n    \nkeras_layer = hub.KerasLayer('https://kaggle.com/models/tensorflow/mask-rcnn-inception-resnet-v2/frameworks/TensorFlow2/variations/1024x1024/versions/1')\n","metadata":{"execution":{"iopub.status.busy":"2023-03-05T03:25:56.507982Z","iopub.execute_input":"2023-03-05T03:25:56.508791Z","iopub.status.idle":"2023-03-05T03:26:50.630561Z","shell.execute_reply.started":"2023-03-05T03:25:56.508758Z","shell.execute_reply":"2023-03-05T03:26:50.629591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install keras-segmentation","metadata":{"execution":{"iopub.status.busy":"2023-03-29T09:47:59.737916Z","iopub.execute_input":"2023-03-29T09:47:59.738321Z","iopub.status.idle":"2023-03-29T09:48:16.627741Z","shell.execute_reply.started":"2023-03-29T09:47:59.738283Z","shell.execute_reply":"2023-03-29T09:48:16.626336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras_segmentation.models.unet import vgg_unet\nfrom keras_segmentation.models.model_utils import load_model\n\n# Load the saved weights into a new model\nmodel = vgg_unet(n_classes=6, input_height=416, input_width=608)\nload_model(model,\"/kaggle/input/test-model/vgg_unet (2).0\")\n\n# Use the model for inference or training\n","metadata":{"execution":{"iopub.status.busy":"2023-03-29T09:55:25.251183Z","iopub.execute_input":"2023-03-29T09:55:25.251626Z","iopub.status.idle":"2023-03-29T09:55:25.769005Z","shell.execute_reply.started":"2023-03-29T09:55:25.251584Z","shell.execute_reply":"2023-03-29T09:55:25.767507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install patchify","metadata":{"execution":{"iopub.status.busy":"2023-03-25T06:32:32.262886Z","iopub.execute_input":"2023-03-25T06:32:32.263441Z","iopub.status.idle":"2023-03-25T06:32:45.760295Z","shell.execute_reply.started":"2023-03-25T06:32:32.263387Z","shell.execute_reply":"2023-03-25T06:32:45.758711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import csv\nimport shutil\nimport os\nimport tifffile\nfrom PIL import Image\n\npng_mask = pil_mask.convert('RGB').save('/kaggle/working/mask/'+row['image_id']+'.png')","metadata":{"execution":{"iopub.status.busy":"2023-03-25T16:37:14.196635Z","iopub.execute_input":"2023-03-25T16:37:14.197795Z","iopub.status.idle":"2023-03-25T16:37:14.244151Z","shell.execute_reply.started":"2023-03-25T16:37:14.197752Z","shell.execute_reply":"2023-03-25T16:37:14.241991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from PIL import Image\nimport matplotlib.pyplot as plt\nimport matplotlib\nimport numpy as np\nImage.MAX_IMAGE_PIXELS = None\nimport cv2 as cv2\n%matplotlib inline\n\noriginal_image = \"/kaggle/working/tiles/images0018ae58b01bdadc8e347995b69f99aa_tile_0_1024.png\"\nlabel_image_semantic = \"/kaggle/working/normalized_image.png\"\n\nfig, axs = plt.subplots(1, 2, figsize=(16, 8), constrained_layout=True)\n\ncmap = matplotlib.colors.ListedColormap(['black', 'gray', 'green', 'yellow', 'orange', 'red'])\n\naxs[0].imshow( Image.open(original_image))\naxs[0].grid(False)\n\nlabel_image_semantic = cv2.imread(label_image_semantic)\nlabel_image_semantic = np.asarray(label_image_semantic)\nprint(label_image_semantic.min(), label_image_semantic.max())\nlabel_image_semantic = label_image_semantic[:, :, :3]\nlabel_image_semantic = label_image_semantic / 255.0\naxs[1].imshow(label_image_semantic,cmap=cmap,interpolation='nearest')\naxs[1].grid(False)\n","metadata":{"execution":{"iopub.status.busy":"2023-04-02T15:23:05.88482Z","iopub.execute_input":"2023-04-02T15:23:05.885268Z","iopub.status.idle":"2023-04-02T15:23:06.804624Z","shell.execute_reply.started":"2023-04-02T15:23:05.885232Z","shell.execute_reply":"2023-04-02T15:23:06.803537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from PIL import Image\nimport matplotlib.pyplot as plt\nimport matplotlib\nimport numpy as np\n\n%matplotlib inline\nimage_id = 'ffe9bcababc858e04840669e788065a1'\n\noriginal_image = \"/kaggle/working/tiles/images/004dd32d9cd167d9cc31c13b704498af_tile_11008_4352.png\"\nlabel_image_semantic = \"/kaggle/working/tiles/masks/004dd32d9cd167d9cc31c13b704498af_tile_11008_4352.png\"\n\nfig, axs = plt.subplots(1, 2, figsize=(16, 8), constrained_layout=True)\n\ncmap = matplotlib.colors.ListedColormap(['black', 'gray', 'green', 'yellow', 'orange', 'red'])\n\naxs[0].imshow( Image.open(original_image))\naxs[0].grid(False)\n\nlabel_image_semantic = Image.open(label_image_semantic)\nlabel_image_semantic = np.asarray(label_image_semantic)\naxs[1].imshow(label_image_semantic,cmap=cmap,interpolation='nearest', vmin=0, vmax=5)\naxs[1].grid(False)","metadata":{"execution":{"iopub.status.busy":"2023-04-02T18:14:07.160149Z","iopub.execute_input":"2023-04-02T18:14:07.160711Z","iopub.status.idle":"2023-04-02T18:14:08.445001Z","shell.execute_reply.started":"2023-04-02T18:14:07.160656Z","shell.execute_reply":"2023-04-02T18:14:08.442916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport matplotlib.pyplot as plt\n\nmask = cv2.imread('/kaggle/working/tiles/masks/0018ae58b01bdadc8e347995b69f99aa_tile_10496_1024.png', cv2.IMREAD_COLOR)\nred_channel = mask[:, :, 2]\ncmap = matplotlib.colors.ListedColormap(['black', 'gray', 'green', 'yellow', 'orange', 'red'])\nplt.imshow(red_channel, cmap=cmap)\nplt.colorbar()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-04-02T15:30:38.519712Z","iopub.execute_input":"2023-04-02T15:30:38.520177Z","iopub.status.idle":"2023-04-02T15:30:38.834938Z","shell.execute_reply.started":"2023-04-02T15:30:38.520135Z","shell.execute_reply":"2023-04-02T15:30:38.833456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"actual_img_path = \"/kaggle/input/isup-training/images/0018ae58b01bdadc8e347995b69f99aa.png\"\ngt_img_path = \"/kaggle/input/panda2/train_label_masks/0018ae58b01bdadc8e347995b69f99aa.png\"     \n    \nactual_img = Image.open(actual_img_path)\ngt_img = Image.open(gt_img_path)\n\n            # Get size of actual image and ground truth image\nactual_size = actual_img.size\ngt_size = gt_img.size\n\n            # Calculate scaling factor\nscale_factor = actual_size[0] / gt_size[0]\n\n            # Resize ground truth image\n        #resized_gt_img = gt_img.resize((int(gt_size[0] * scale_factor), int(gt_size[1] * scale_factor)),resample=Image.NEAREST)\nresized_gt_img = gt_img.resize(actual_size)\nresized_gt_img.save(os.path.join('/kaggle/working/1.png'))","metadata":{"execution":{"iopub.status.busy":"2023-04-02T15:17:50.382134Z","iopub.execute_input":"2023-04-02T15:17:50.382785Z","iopub.status.idle":"2023-04-02T15:17:54.217032Z","shell.execute_reply.started":"2023-04-02T15:17:50.382732Z","shell.execute_reply":"2023-04-02T15:17:54.215743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport matplotlib.colors as mcolors\nimport matplotlib\nimport numpy as np\nfrom PIL import Image\nImage.MAX_IMAGE_PIXELS = None\n# Load your image as a numpy array\nimg = Image.open(\"/kaggle/input/learning/mask/006f6aa35a78965c92fffd1fbd53a058.png\")\nimg_array = np.array(img)\n\n# Convert the image to grayscale\ngray_array = np.mean(img_array, axis=-1)\n\n# Rescale the grayscale image to the range of 0 to 1\ngray_array_rescaled = gray_array / np.max(gray_array)\n\n# Define the colormap\ncmap = matplotlib.colors.ListedColormap(['black', 'gray', 'green', 'yellow', 'orange', 'red'])\n# Apply the colormap to the grayscale image\ncolored_array = cmap(gray_array_rescaled)\n\n# Display the colored image\nplt.imshow(colored_array)\nplt.show()\nprint('done')\n","metadata":{"execution":{"iopub.status.busy":"2023-03-26T09:47:33.079456Z","iopub.execute_input":"2023-03-26T09:47:33.080023Z","iopub.status.idle":"2023-03-26T09:47:37.507907Z","shell.execute_reply.started":"2023-03-26T09:47:33.079975Z","shell.execute_reply":"2023-03-26T09:47:37.506653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from PIL import Image\nimport matplotlib.pyplot as plt\nimport matplotlib\n\n%matplotlib inline\n\noriginal_image = \"/kaggle/input/panda2/train_images/008069b542b0439ed69b194674051964.png\"\nlabel_image_semantic = \"/kaggle/input/panda2/train_label_masks/008069b542b0439ed69b194674051964.png\"\n\nfig, axs = plt.subplots(1, 2, figsize=(16, 8), constrained_layout=True)\n\ncmap = matplotlib.colors.ListedColormap(['black', 'gray', 'green', 'yellow', 'orange', 'red'])\n\naxs[0].imshow( Image.open(original_image))\naxs[0].grid(False)\n\nlabel_image_semantic = Image.open(label_image_semantic)\nlabel_image_semantic = np.asarray(label_image_semantic)\naxs[1].imshow(label_image_semantic,cmap=cmap,interpolation='nearest', vmin=0, vmax=5)\naxs[1].grid(False)","metadata":{"execution":{"iopub.status.busy":"2023-03-26T09:47:43.190195Z","iopub.execute_input":"2023-03-26T09:47:43.191394Z","iopub.status.idle":"2023-03-26T09:47:44.095682Z","shell.execute_reply.started":"2023-03-26T09:47:43.19133Z","shell.execute_reply":"2023-03-26T09:47:44.094311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from PIL import Image\nImage.MAX_IMAGE_PIXELS = None\n# Load actual image and ground truth image\nactual_img = Image.open(\"/kaggle/input/learning/images/0018ae58b01bdadc8e347995b69f99aa.png\")\ngt_img = Image.open(\"/kaggle/input/panda2/train_label_masks/0018ae58b01bdadc8e347995b69f99aa.png\")\n\n# Get size of actual image and ground truth image\nactual_size = actual_img.size\ngt_size = gt_img.size\n\n# Calculate scaling factor\nscale_factor = actual_size[0] / gt_size[0]\n\n# Resize ground truth image\nresized_gt_img = gt_img.resize((int(gt_size[0] * scale_factor), int(gt_size[1] * scale_factor)))\n\n# Save resized ground truth image\n\nprint(\"Image shape:\", actual_img.size)\nprint(\"Mask shape:\", resized_gt_img.size)\n\nfig, axs = plt.subplots(1, 2, figsize=(16, 8), constrained_layout=True)\n\ncmap = matplotlib.colors.ListedColormap(['black', 'gray', 'green', 'yellow', 'orange', 'red'])\nlabel_image_semantic = np.asarray(resized_gt_img)\naxs[1].imshow(label_image_semantic,cmap=cmap,interpolation='nearest', vmin=0, vmax=5)\naxs[1].grid(False)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-03-26T10:22:53.738341Z","iopub.execute_input":"2023-03-26T10:22:53.738847Z","iopub.status.idle":"2023-03-26T10:22:57.94996Z","shell.execute_reply.started":"2023-03-26T10:22:53.738797Z","shell.execute_reply":"2023-03-26T10:22:57.948374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"mkdir mask","metadata":{"execution":{"iopub.status.busy":"2023-03-26T10:00:07.364015Z","iopub.execute_input":"2023-03-26T10:00:07.364424Z","iopub.status.idle":"2023-03-26T10:00:08.484404Z","shell.execute_reply.started":"2023-03-26T10:00:07.364389Z","shell.execute_reply":"2023-03-26T10:00:08.482568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"from PIL import Image\nimport os\nimport csv\nImage.MAX_IMAGE_PIXELS = None\ncount = 0\n\n# Set the directory where the images and the CSV file are located\ndirectory = \"/kaggle/input/learning/images/\"\ngt_directory = '/kaggle/input/panda2/train_label_masks'\n# Load the CSV file containing the image IDs\nwith open(\"/kaggle/input/prostate-cancer-grade-assessment/train.csv\", \"r\") as file:\n    reader = csv.reader(file)\n    next(reader) # skip the header row\n    for row in reader:\n        # Extract the image ID from the current row of the CSV file\n        image_id = row[0]\n        \n        # Construct the paths for the actual image and the ground truth image\n        actual_img_path = os.path.join(directory, f\"{image_id}.png\")\n        gt_img_path = os.path.join(gt_directory, f\"{image_id}.png\")\n        \n        # Check if the actual image file and the ground truth image file exist\n        if os.path.isfile(actual_img_path) and os.path.isfile(gt_img_path):\n            # Load actual image and ground truth image\n            actual_img = Image.open(actual_img_path)\n            gt_img = Image.open(gt_img_path)\n\n            # Get size of actual image and ground truth image\n            actual_size = actual_img.size\n            gt_size = gt_img.size\n\n            # Calculate scaling factor\n            scale_factor = actual_size[0] / gt_size[0]\n\n            # Resize ground truth image\n            resized_gt_img = gt_img.resize((int(gt_size[0] * scale_factor), int(gt_size[1] * scale_factor)))\n\n            # Save resized ground truth image\n            resized_gt_img.save(os.path.join('/kaggle/working/mask/',f\"{image_id}.png\"))\n            count +=1\n            print(image_id)\n            print(count)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport numpy as np\n\nann_img = np.zeros((30,30,3)).astype('uint8')\nann_img[ 3 , 4 ] = 1 # this would set the label of pixel 3,4 as 1\nann_img[ 0 , 0 ] = 2 \n\n","metadata":{"execution":{"iopub.status.busy":"2023-03-26T10:20:07.551351Z","iopub.execute_input":"2023-03-26T10:20:07.55253Z","iopub.status.idle":"2023-03-26T10:20:07.80309Z","shell.execute_reply.started":"2023-03-26T10:20:07.552461Z","shell.execute_reply":"2023-03-26T10:20:07.801689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install keras_segmentation ","metadata":{"execution":{"iopub.status.busy":"2023-03-26T10:20:32.079413Z","iopub.execute_input":"2023-03-26T10:20:32.079866Z","iopub.status.idle":"2023-03-26T10:20:49.202638Z","shell.execute_reply.started":"2023-03-26T10:20:32.079828Z","shell.execute_reply":"2023-03-26T10:20:49.201171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\n\nimage_path = \"/kaggle/input/learning/images/008069b542b0439ed69b194674051964.png\"\nmask_path = \"/kaggle/working/mask/008069b542b0439ed69b194674051964.png\"\n\nimage = Image.open(image_path)\nmask = Image.open(mask_path)\n\nprint(\"Image shape:\", image.size)\nprint(\"Mask shape:\", mask.size)\n","metadata":{"execution":{"iopub.status.busy":"2023-03-26T10:24:03.743131Z","iopub.execute_input":"2023-03-26T10:24:03.743933Z","iopub.status.idle":"2023-03-26T10:24:03.762657Z","shell.execute_reply.started":"2023-03-26T10:24:03.743863Z","shell.execute_reply":"2023-03-26T10:24:03.760665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras_segmentation.models.unet import vgg_unet\n\nn_classes = 7 # Aerial Semantic Segmentation Drone Dataset tree, gras, other vegetation, dirt, gravel, rocks, water, paved area, pool, person, dog, car, bicycle, roof, wall, fence, fence-pole, window, door, obstacle\nmodel = vgg_unet(n_classes=n_classes ,  input_height=416, input_width=608  )\n\nmodel.train( \n    train_images =  \"/kaggle/input/learning/images\",\n    train_annotations = \"/kaggle/working/mask\",\n    checkpoints_path = \"vgg_unet\" , epochs=1\n)\n","metadata":{"execution":{"iopub.status.busy":"2023-03-26T10:26:36.731483Z","iopub.execute_input":"2023-03-26T10:26:36.731947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* > *****> code to covert tiff to png with full clarity*","metadata":{}},{"cell_type":"code","source":"import csv\nimport shutil\nimport os\nimport tifffile\nfrom PIL import Image\n\n# Define the path to the CSV file\ncsv_file_path = \"/kaggle/input/prostate-cancer-grade-assessment/train.csv\"\n\n# Define the path to the folder where the images are stored\nimages_folder_path = \"/kaggle/input/panda2/train_label_masks\"\n\n# Define the provider you want to filter\nprovider_name = \"radboud\"\n\n# Define the path to the folder where the images with the specific provider will be copied\noutput_folder_path = \"/kaggle/working/images\"\n\n# Create the output folder if it doesn't exist\nif not os.path.exists(output_folder_path):\n    os.makedirs(output_folder_path)\ncount = 0\n# Open the CSV file and read its contents\nwith open(csv_file_path, newline='') as csvfile:\n    reader = csv.DictReader(csvfile)\n    \n    # Loop through each row in the CSV file\n    for row in reader:\n        \n        # Check if the provider in the current row matches the provider you want to filter\n        if row['data_provider'] == 'radboud' and row['gleason_score']!='negative' and os.path.isfile('/kaggle/input/radboud-mask/preprocessed/'+row['image_id']+'.png'):\n            \n            # Define the path to the image file\n            #image_file_path = os.path.join(images_folder_path, row['image_id']+'.png')\n            \n            # Check if the image file exists\n            #if os.path.exists(image_file_path):\n                \n                # Define the path to the output file\n                #output_file_path = os.path.join(output_folder_path, row['image_id']+'.png')\n                \n                # Copy the image file to the output folder\n                #shutil.copyfile(image_file_path, output_file_path)\n                #count+=1\n                #print(count)\n                # Print a message to confirm that the file has been copied\n                \n                \n            tiff_image = tifffile.imread('/kaggle/input/prostate-cancer-grade-assessment/train_images/'+row['image_id']+'.tiff')\n            #tiff_mask = tifffile.imread('/kaggle/input/prostate-cancer-grade-assessment/train_label_masks/'+row['image_id']+'_mask.tiff')\n            # Convert the image to PIL image object\n            pil_image = Image.fromarray(tiff_image)\n           # pil_mask = Image.fromarray(tiff_mask)\n            # Convert the image to PNG\n            png_image = pil_image.convert('RGB').save('/kaggle/working/images/'+row['image_id']+'.png')\n            #png_mask = pil_mask.convert('RGB').save('/kaggle/working/mask/'+row['image_id']+'.png')\n            print(\"Image file '{}' has been copied to '{}'\".format(row['image_id'], output_folder_path))\n            count+=1\n            print(count)\n            if count == 200:\n                break\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install openslide","metadata":{"execution":{"iopub.status.busy":"2023-03-24T04:48:53.284483Z","iopub.execute_input":"2023-03-24T04:48:53.285265Z","iopub.status.idle":"2023-03-24T04:48:55.543163Z","shell.execute_reply.started":"2023-03-24T04:48:53.285217Z","shell.execute_reply":"2023-03-24T04:48:55.541688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"resizing ground truth","metadata":{}},{"cell_type":"code","source":"mkdir masks","metadata":{"execution":{"iopub.status.busy":"2023-04-02T11:11:34.699132Z","iopub.execute_input":"2023-04-02T11:11:34.699587Z","iopub.status.idle":"2023-04-02T11:11:35.885609Z","shell.execute_reply.started":"2023-04-02T11:11:34.699548Z","shell.execute_reply":"2023-04-02T11:11:35.883884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tile preprocessing","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rm *.png","metadata":{"execution":{"iopub.status.busy":"2023-04-02T11:39:28.847775Z","iopub.execute_input":"2023-04-02T11:39:28.848252Z","iopub.status.idle":"2023-04-02T11:39:29.959115Z","shell.execute_reply.started":"2023-04-02T11:39:28.848205Z","shell.execute_reply":"2023-04-02T11:39:29.957589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mkdir tiles","metadata":{"execution":{"iopub.status.busy":"2023-04-02T13:37:17.541942Z","iopub.execute_input":"2023-04-02T13:37:17.542412Z","iopub.status.idle":"2023-04-02T13:37:19.585555Z","shell.execute_reply.started":"2023-04-02T13:37:17.542362Z","shell.execute_reply":"2023-04-02T13:37:19.582142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mkdir tiles/masks","metadata":{"execution":{"iopub.status.busy":"2023-04-02T13:37:27.175163Z","iopub.execute_input":"2023-04-02T13:37:27.175694Z","iopub.status.idle":"2023-04-02T13:37:28.283697Z","shell.execute_reply.started":"2023-04-02T13:37:27.175646Z","shell.execute_reply":"2023-04-02T13:37:28.282104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-04-02T13:38:41.779559Z","iopub.execute_input":"2023-04-02T13:38:41.780132Z","iopub.status.idle":"2023-04-02T14:35:10.833592Z","shell.execute_reply.started":"2023-04-02T13:38:41.780088Z","shell.execute_reply":"2023-04-02T14:35:10.831846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n\n# There are two ways to load the data from the PANDA dataset:\n# Option 1: Load images using openslide\nimport openslide\n# Option 2: Load images using skimage (requires that tifffile is installed)\nimport skimage.io\n\n# General packages\nimport pandas as pd\nimport numpy as np\nimport matplotlib\nimport matplotlib.pyplot as plt\nimport PIL\nfrom IPython.display import Image, display\n\n# Plotly for the interactive viewer (see last section)\nimport plotly.graph_objs as go\n","metadata":{"execution":{"iopub.status.busy":"2023-03-25T16:37:48.843837Z","iopub.execute_input":"2023-03-25T16:37:48.844255Z","iopub.status.idle":"2023-03-25T16:37:49.112831Z","shell.execute_reply.started":"2023-03-25T16:37:48.844217Z","shell.execute_reply":"2023-03-25T16:37:49.11156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tifffile\nimport numpy as np\nfrom PIL import Image\n\n# Load the ground truth data using tifffile\nground_truth = tifffile.imread('/kaggle/input/prostate-cancer-grade-assessment/train_label_masks/0bd231c85b2695e2cf021299e67a6afc_mask.tiff')\n\n# Convert the ground truth data to a grayscale image\nground_truth = np.mean(ground_truth, axis=-1)\n\n# Rescale the ground truth data to the range of 0-255\nground_truth = (ground_truth - ground_truth.min()) / (ground_truth.max() - ground_truth.min()) * 255\nground_truth = ground_truth.astype(np.uint8)\n\n# Convert the ground truth data to a PIL image object\npil_image = Image.fromarray(ground_truth)\n\n# Convert the PIL image object to PNG format and save it\npil_image.save('/kaggle/working/final1.png')\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{"execution":{"iopub.status.busy":"2023-03-25T08:25:05.793301Z","iopub.execute_input":"2023-03-25T08:25:05.794464Z","iopub.status.idle":"2023-03-25T08:25:06.265972Z","shell.execute_reply.started":"2023-03-25T08:25:05.79441Z","shell.execute_reply":"2023-03-25T08:25:06.264273Z"}}},{"cell_type":"code","source":"import cv2\nimport matplotlib.pyplot as plt\nfrom matplotlib.colors import ListedColormap\n\ndef convert_tiff_to_png(input_file, output_file):\n    image = cv2.imread(input_file, cv2.IMREAD_UNCHANGED)\n    cv2.imwrite(output_file, image)\n    return image\n\ninput_file = \"/kaggle/input/prostate-cancer-grade-assessment/train_label_masks/00a7fb880dc12c5de82df39b30533da9_mask.tiff\"\noutput_file = \"/kaggle/working/output1.png\"\n\nimage = convert_tiff_to_png(input_file, output_file)\n\n# Convert the image to grayscale if it has 3 channels (possibly RGB)\nif len(image.shape) == 3 and image.shape[2] == 3:\n    gray_image = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)\nelse:\n    gray_image = image\n\n# Define the colormap\ncustom_cmap = ListedColormap(['black', 'gray', 'green', 'yellow', 'orange', 'red'])\n\n# Display the grayscale image using the custom colormap\nplt.imshow(gray_image, cmap=custom_cmap)\nplt.show()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tifffile\nimport cv2\nimport matplotlib.pyplot as plt\nfrom matplotlib.colors import ListedColormap\n\ndef convert_tiff_to_png(input_file, output_file):\n    image = tifffile.imread(input_file)\n    cv2.imwrite(output_file, image)\n    return image\n\ninput_file = \"/kaggle/input/prostate-cancer-grade-assessment/train_label_masks/00a26aaa82c959624d90dfb69fcf259c_mask.tiff\"\noutput_file = \"/kaggle/working/output.png\"\n\nimage = convert_tiff_to_png(input_file, output_file)\n\n# Select the R channel (assuming the R channel is the first channel)\nr_channel = image[:, :, 0]\n\n# Define the colormap\ncustom_cmap = matplotlib.colors.ListedColormap(['black', 'gray', 'green', 'orange', 'white', 'red'])\n\n# Display the R channel using the custom colormap\nplt.imshow(r_channel, cmap=custom_cmap,interpolation='nearest')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-03-29T04:52:06.471621Z","iopub.execute_input":"2023-03-29T04:52:06.473037Z","iopub.status.idle":"2023-03-29T04:52:10.731805Z","shell.execute_reply.started":"2023-03-29T04:52:06.472967Z","shell.execute_reply":"2023-03-29T04:52:10.730438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"creating a zip file","metadata":{}},{"cell_type":"code","source":"import shutil\nimport os\n\ndef zip_folder(folder_path, output_path):\n    shutil.make_archive(output_path, 'zip', folder_path)\n\n# Example usage\nfolder_path = 'path/to/folder'\noutput_path = 'path/to/output/file_without_extension'\n\nzip_folder(folder_path, output_path)\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tifffile\nfrom PIL import Image\n\n# Load the original TIFF image using tifffile\nimage = tifffile.imread('/kaggle/input/prostate-cancer-grade-assessment/train_label_masks/0bd231c85b2695e2cf021299e67a6afc_mask.tiff')\n\n# Extract the ground truth data from the red channel of the image\nground_truth = image[:, :, 0]\n\n# Rescale the ground truth data to the range of 0-255\nground_truth = (ground_truth - ground_truth.min()) / (ground_truth.max() - ground_truth.min()) * 255\nground_truth = ground_truth.astype('uint8')\n\n# Convert the ground truth data to a PIL image object\npil_image = Image.fromarray(ground_truth)\n\n# Convert the PIL image object to PNG format and save it\npil_image.save('/kaggle/working/2.png')\n","metadata":{"execution":{"iopub.status.busy":"2023-03-25T10:03:53.77417Z","iopub.execute_input":"2023-03-25T10:03:53.774586Z","iopub.status.idle":"2023-03-25T10:03:54.209256Z","shell.execute_reply.started":"2023-03-25T10:03:53.77455Z","shell.execute_reply":"2023-03-25T10:03:54.207976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Location of the training images\ndata_dir = '/kaggle/input/prostate-cancer-grade-assessment/train_images'\nmask_dir = '/kaggle/input/prostate-cancer-grade-assessment/train_label_masks'\n\n# Location of training labels\ntrain_labels = pd.read_csv('/kaggle/input/prostate-cancer-grade-assessment/train.csv').set_index('image_id')\ntrain_labels['data_provider'].value_counts()['radboud']","metadata":{"execution":{"iopub.status.busy":"2023-03-24T10:44:03.516824Z","iopub.execute_input":"2023-03-24T10:44:03.5173Z","iopub.status.idle":"2023-03-24T10:44:03.601761Z","shell.execute_reply.started":"2023-03-24T10:44:03.517262Z","shell.execute_reply":"2023-03-24T10:44:03.600404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from PIL import Image\nimport matplotlib.pyplot as plt\nimport matplotlib\n\n%matplotlib inline\n\noriginal_image = \"/kaggle/input/panda2/train_images/00a7fb880dc12c5de82df39b30533da9.png\"\nlabel_image_semantic = \"/kaggle/working/ground_truth2.png\"\n\nfig, axs = plt.subplots(1, 2, figsize=(16, 8), constrained_layout=True)\n\ncmap = matplotlib.colors.ListedColormap(['black', 'gray', 'green', 'orange', 'white', 'red'])\n\naxs[0].imshow( Image.open(original_image))\naxs[0].grid(False)\n\nlabel_image_semantic = Image.open(label_image_semantic)\nlabel_image_semantic = np.asarray(label_image_semantic)\naxs[1].imshow(label_image_semantic,cmap=cmap,interpolation='nearest')\naxs[1].grid(False)","metadata":{"execution":{"iopub.status.busy":"2023-03-25T16:34:02.570788Z","iopub.execute_input":"2023-03-25T16:34:02.572417Z","iopub.status.idle":"2023-03-25T16:34:03.426869Z","shell.execute_reply.started":"2023-03-25T16:34:02.572345Z","shell.execute_reply":"2023-03-25T16:34:03.42505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Using OpenSlide to load the data\n\nIn the following sections we will load data from the slides with [OpenSlide](https://openslide.org/api/python/). The benefit of OpenSlide is that we can load arbitrary regions of the slide, without loading the whole image in memory. Want to interactively view a slide? We have added an [interactive viewer](#Interactive-viewer-for-slides) to this notebook in the last section.\n\nYou can read more about the OpenSlide python bindings in the documentation: https://openslide.org/api/python/\n\n## Loading a slide\n\nBefore we can load data from a slide, we need to open it. After a file is open we can retrieve data from it at arbitratry positions and levels.\n\n```python\nbiopsy = openslide.OpenSlide(path)\n# do someting with the slide here\nbiopsy.close()\n```","metadata":{}},{"cell_type":"markdown","source":"For this tutorial, we created a small function to show some basic information about a slide. Additionally, this function display a small thumbnail of the slide. All images in the dataset contain this metadata and you can use this in your data pipeline.","metadata":{}},{"cell_type":"code","source":"def print_slide_details(slide, show_thumbnail=True, max_size=(600,400)):\n    \"\"\"Print some basic information about a slide\"\"\"\n    # Generate a small image thumbnail\n    if show_thumbnail:\n        display(slide.get_thumbnail(size=max_size))\n\n    # Here we compute the \"pixel spacing\": the physical size of a pixel in the image.\n    # OpenSlide gives the resolution in centimeters so we convert this to microns.\n    spacing = 1 / (float(slide.properties['tiff.XResolution']) / 10000)\n    \n    print(f\"File id: {slide}\")\n    print(f\"Dimensions: {slide.dimensions}\")\n    print(f\"Microns per pixel / pixel spacing: {spacing:.3f}\")\n    print(f\"Number of levels in the image: {slide.level_count}\")\n    print(f\"Downsample factor per level: {slide.level_downsamples}\")\n    print(f\"Dimensions of levels: {slide.level_dimensions}\")\n","metadata":{"execution":{"iopub.status.busy":"2023-03-24T04:49:16.030968Z","iopub.execute_input":"2023-03-24T04:49:16.031436Z","iopub.status.idle":"2023-03-24T04:49:16.039858Z","shell.execute_reply.started":"2023-03-24T04:49:16.031396Z","shell.execute_reply":"2023-03-24T04:49:16.038749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"#A code to copy specific files for testing\nimport shutil\nimport os\n\n# Define the paths of the source and destination directories\nsource_dir = '/kaggle/input/prostate-cancer-grade-assessment/train_images'\ndest_dir = '/kaggle/working/'\n\n# Define the name of the image file to be copied\nimage_name = '2bb4f3202acd0b3cd10ffae292b6edf8.tiff'\n\n# Create the full paths of the source and destination files\nsource_file = os.path.join(source_dir, image_name)\ndest_file = os.path.join(dest_dir, image_name)\n\n# Copy the image file from the source directory to the destination directory\nshutil.copyfile(source_file, dest_file)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-03-14T06:13:13.25615Z","iopub.execute_input":"2023-03-14T06:13:13.258239Z","iopub.status.idle":"2023-03-14T06:13:13.275262Z","shell.execute_reply.started":"2023-03-14T06:13:13.258183Z","shell.execute_reply":"2023-03-14T06:13:13.274047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#to zip neccessary images and files for downloading\nimport os\nimport zipfile\n\ndirectory = '/kaggle/working/'\nfiles = os.listdir(directory)\n\nwith zipfile.ZipFile('/kaggle/working/preprocessed.zip', 'r') as zip:\n        zip.extractall('/kaggle/working/')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Running the cell below loads four example biopsies using OpenSlide. Some things you can notice:\n\n- The image dimensions are quite large (typically between 5.000 and 40.000 pixels in both x and y).\n- Each slide has 3 levels you can load, corresponding to a downsampling of 1, 4 and 16. Intermediate levels can be created by downsampling a higher resolution level.\n- The dimensions of each level differ based on the dimensions of the original image.\n- Biopsies can be in different rotations. This rotation has no clinical value, and is only dependent on how the biopsy was collected in the lab.\n- There are noticable color differences between the biopsies, this is very common within pathology and is caused by different laboratory procedures.\n","metadata":{}},{"cell_type":"code","source":"example_slides = [\n    '005e66f06bce9c2e49142536caf2f6ee',\n    '00928370e2dfeb8a507667ef1d4efcbb',\n    '007433133235efc27a39f11df6940829',\n    '024ed1244a6d817358cedaea3783bbde',\n]\n\nfor case_id in example_slides:\n    biopsy = openslide.OpenSlide(os.path.join(data_dir, f'{case_id}.tiff'))\n    print_slide_details(biopsy)\n    biopsy.close()\n    \n    # Print the case-level label\n    print(f\"ISUP grade: {train_labels.loc[case_id, 'isup_grade']}\")\n    print(f\"Gleason score: {train_labels.loc[case_id, 'gleason_score']}\\n\\n\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Loading image regions/patches\n\nWith OpenSlide we can easily extract patches from the slide from arbitrary locations. Loading a specific region is done using the [read_region](https://openslide.org/api/python/#openslide.OpenSlide.read_region) function.\n\nAfter opening the slide we can, for example, load a 512x512 patch from the lowest level (level 0) at a specific coordinate.\n","metadata":{}},{"cell_type":"code","source":"# Open the image (does not yet read the image into memory)\nimage = openslide.OpenSlide(os.path.join(data_dir, '00bbc1482301d16de3ff63238cfd0b34.tiff'))\n\n# Read a specific region of the image starting at upper left coordinate (x=17800, y=19500) on level 0 and extracting a 256*256 pixel patch.\n# At this point image data is read from the file and loaded into memory.\npatch = image.read_region((62,120), 1, (193, 1008))\n\n# Display the image\ndisplay(patch)\n\n# Close the opened slide after use\nimage.close()","metadata":{"execution":{"iopub.status.busy":"2023-03-24T10:44:30.287272Z","iopub.execute_input":"2023-03-24T10:44:30.287704Z","iopub.status.idle":"2023-03-24T10:44:30.354413Z","shell.execute_reply.started":"2023-03-24T10:44:30.287673Z","shell.execute_reply":"2023-03-24T10:44:30.35286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"biopsy = openslide.OpenSlide(os.path.join(data_dir, '00928370e2dfeb8a507667ef1d4efcbb.tiff'))\n\nx = 5150\ny = 21000\nlevel = 1\nwidth = 512\nheight = 512\n\nregion = biopsy.read_region((x,y), level, (width, height))\ndisplay(region)\n","metadata":{"execution":{"iopub.status.busy":"2023-03-24T10:44:55.223103Z","iopub.execute_input":"2023-03-24T10:44:55.224809Z","iopub.status.idle":"2023-03-24T10:44:55.449967Z","shell.execute_reply.started":"2023-03-24T10:44:55.224716Z","shell.execute_reply":"2023-03-24T10:44:55.449121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from PIL import Image\n\n# Open the original image\noriginal_image = Image.open('/kaggle/input/panda2/train_images/00928370e2dfeb8a507667ef1d4efcbb.png')\n\n# Upscale the image\nupscaled_image = original_image.resize((original_image.width * 2, original_image.height * 2), resample=Image.BICUBIC)\n\n# Save the upscaled image\nupscaled_image.save('upscaled.png')","metadata":{"execution":{"iopub.status.busy":"2023-03-24T06:12:55.66246Z","iopub.execute_input":"2023-03-24T06:12:55.66313Z","iopub.status.idle":"2023-03-24T06:12:56.618081Z","shell.execute_reply.started":"2023-03-24T06:12:55.663087Z","shell.execute_reply":"2023-03-24T06:12:56.616735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\n\nimg = cv2.imread('/kaggle/input/panda2/train_images/000920ad0b612851f8e01bcc880d9b3d.png')\nimg.save()\n","metadata":{"execution":{"iopub.status.busy":"2023-03-07T03:32:57.629146Z","iopub.execute_input":"2023-03-07T03:32:57.629954Z","iopub.status.idle":"2023-03-07T03:32:57.656603Z","shell.execute_reply.started":"2023-03-07T03:32:57.629914Z","shell.execute_reply":"2023-03-07T03:32:57.655025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Using the `level` argument we can easily load in data from any level that is present in the slide. Coordinates passed to `read_region` are always relative to level 0 (the highest resolution).","metadata":{}},{"cell_type":"code","source":"biopsy.close()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Loading label masks\n\nApart from the slide-level label (present in the csv file), almost all slides in the training set have an associated mask with additional label information. These masks directly indicate which parts of the tissue are healthy and which are cancerous. The information in the masks differ from the two centers:\n\n- **Radboudumc**: Prostate glands are individually labelled. Valid values are:\n  - 0: background (non tissue) or unknown\n  - 1: stroma (connective tissue, non-epithelium tissue)\n  - 2: healthy (benign) epithelium\n  - 3: cancerous epithelium (Gleason 3)\n  - 4: cancerous epithelium (Gleason 4)\n  - 5: cancerous epithelium (Gleason 5)\n- **Karolinska**: Regions are labelled. Valid values:\n  - 0: background (non tissue) or unknown\n  - 1: benign tissue (stroma and epithelium combined)\n  - 2: cancerous tissue (stroma and epithelium combined)\n\nThe label masks of Radboudumc were semi-automatically generated by several deep learning algorithms, contain noise, and can be considered as weakly-supervised labels. The label masks of Karolinska were semi-autotomatically generated based on annotations by a pathologist.\n\nThe label masks are stored in an RGB format so that they can be easily opened by image readers. The label information is stored in the red (R) channel, the other channels are set to zero and can be ignored. As with the slides itself, the label masks can be opened using OpenSlide.","metadata":{}},{"cell_type":"markdown","source":"### Visualizing the masks (using PIL)\n\nUsing a small helper function we can display some basic information about a mask. To more easily inspect the masks, we map the int labels to RGB colors using a color palette. If you prefer something like `matplotlib` you can also use `plt.imshow()` to directly show a mask (without converting it to an RGB image).","metadata":{}},{"cell_type":"code","source":"def print_mask_details(slide, center='radboud', show_thumbnail=True, max_size=(400,400)):\n    \"\"\"Print some basic information about a slide\"\"\"\n\n    if center not in ['radboud', 'karolinska']:\n        raise Exception(\"Unsupported palette, should be one of [radboud, karolinska].\")\n\n    # Generate a small image thumbnail\n    if show_thumbnail:\n        # Read in the mask data from the highest level\n        # We cannot use thumbnail() here because we need to load the raw label data.\n        mask_data = slide.read_region((0,0), slide.level_count - 1, slide.level_dimensions[-1])\n        # Mask data is present in the R channel\n        mask_data = mask_data.split()[0]\n\n        # To show the masks we map the raw label values to RGB values\n        preview_palette = np.zeros(shape=768, dtype=int)\n        if center == 'radboud':\n            # Mapping: {0: background, 1: stroma, 2: benign epithelium, 3: Gleason 3, 4: Gleason 4, 5: Gleason 5}\n            preview_palette[0:18] = (np.array([0, 0, 0, 0.5, 0.5, 0.5, 0, 1, 0, 1, 1, 0.7, 1, 0.5, 0, 1, 0, 0]) * 255).astype(int)\n        elif center == 'karolinska':\n            # Mapping: {0: background, 1: benign, 2: cancer}\n            preview_palette[0:9] = (np.array([0, 0, 0, 0.5, 0.5, 0.5, 1, 0, 0]) * 255).astype(float)\n        mask_data.putpalette(data=preview_palette.tolist())\n        mask_data = mask_data.convert(mode='RGB')\n        #mask_data.thumbnail(size=max_size, resample=0)\n        display(mask_data)\n\n    # Compute microns per pixel (openslide gives resolution in centimeters)\n    spacing = 1 / (float(slide.properties['tiff.XResolution']) / 10000)\n    \n    print(f\"File id: {slide}\")\n    print(f\"Dimensions: {slide.dimensions}\")\n    print(f\"Microns per pixel / pixel spacing: {spacing:.3f}\")\n    print(f\"Number of levels in the image: {slide.level_count}\")\n    print(f\"Downsample factor per level: {slide.level_downsamples}\")\n    print(f\"Dimensions of levels: {slide.level_dimensions}\")\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#testing 1\n","metadata":{}},{"cell_type":"markdown","source":"testing 2","metadata":{}},{"cell_type":"code","source":"from PIL import Image # (pip install Pillow)\n\ndef create_sub_masks(mask_image):\n    width, height = mask_image.size\n\n    # Initialize a dictionary of sub-masks indexed by RGB colors\n    sub_masks = {}\n    for x in range(width):\n        for y in range(height):\n            # Get the RGB values of the pixel\n            pixel = mask_image.getpixel((x,y))[:3]\n\n            # If the pixel is not black...\n            if pixel != (0, 0, 0):\n                # Check to see if we've created a sub-mask...\n                pixel_str = str(pixel)\n                sub_mask = sub_masks.get(pixel_str)\n                if sub_mask is None:\n                   # Create a sub-mask (one bit per pixel) and add to the dictionary\n                    # Note: we add 1 pixel of padding in each direction\n                    # because the contours module doesn't handle cases\n                    # where pixels bleed to the edge of the image\n                    sub_masks[pixel_str] = Image.new('1', (width+2, height+2))\n\n                # Set the pixel value to 1 (default is 0), accounting for padding\n                sub_masks[pixel_str].putpixel((x+1, y+1), 1)\n\n    return sub_masks","metadata":{"execution":{"iopub.status.busy":"2023-03-25T16:35:58.959633Z","iopub.execute_input":"2023-03-25T16:35:58.960594Z","iopub.status.idle":"2023-03-25T16:35:58.968941Z","shell.execute_reply.started":"2023-03-25T16:35:58.960552Z","shell.execute_reply":"2023-03-25T16:35:58.968004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np                                 # (pip install numpy)\nfrom skimage import measure                        # (pip install scikit-image)\nfrom shapely.geometry import Polygon, MultiPolygon # (pip install Shapely)\n\ndef create_sub_mask_annotation(sub_mask, image_id, category_id, annotation_id, is_crowd):\n    # Find contours (boundary lines) around each sub-mask\n    # Note: there could be multiple contours if the object\n    # is partially occluded. (E.g. an elephant behind a tree)\n    contours = measure.find_contours(sub_mask, 0.5, positive_orientation='low')\n\n    segmentations = []\n    polygons = []\n    for contour in contours:\n        # Flip from (row, col) representation to (x, y)\n        # and subtract the padding pixel\n        for i in range(len(contour)):\n            row, col = contour[i]\n            contour[i] = (col - 1, row - 1)\n\n        # Make a polygon and simplify it\n        poly = Polygon(contour)\n        poly = poly.simplify(1.0, preserve_topology=False)\n        polygons.append(poly)\n        segmentation = np.array(poly.exterior.coords).ravel().tolist()\n        segmentations.append(segmentation)\n\n    # Combine the polygons to calculate the bounding box and area\n    multi_poly = MultiPolygon(polygons)\n    x, y, max_x, max_y = multi_poly.bounds\n    width = max_x - x\n    height = max_y - y\n    bbox = (x, y, width, height)\n    area = multi_poly.area\n    \n    annotation = {\n        \"segmentation\": segmentations,\n        \"iscrowd\": is_crowd,\n        \"image_id\": image_id,\n        \n       \n        \"category_id\": category_id,\n        \n        \"id\": annotation_id,\n        \"bbox\": bbox,\n        \"area\": area\n    }\n\n    return annotation","metadata":{"execution":{"iopub.status.busy":"2023-03-25T16:36:07.484688Z","iopub.execute_input":"2023-03-25T16:36:07.48583Z","iopub.status.idle":"2023-03-25T16:36:08.115989Z","shell.execute_reply.started":"2023-03-25T16:36:07.485777Z","shell.execute_reply":"2023-03-25T16:36:08.114981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rm test.json\n","metadata":{"execution":{"iopub.status.busy":"2023-03-08T06:49:34.878025Z","iopub.execute_input":"2023-03-08T06:49:34.878873Z","iopub.status.idle":"2023-03-08T06:49:35.880662Z","shell.execute_reply.started":"2023-03-08T06:49:34.878829Z","shell.execute_reply":"2023-03-08T06:49:35.879328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Hope","metadata":{}},{"cell_type":"code","source":"import json\nimport numpy as np\nfile_name = \"0068d4c7529e34fd4c9da863ce01a161.png\"\n\nplant_book_mask_image = Image.open('/kaggle/input/learning/karolinska/'+file_name)\n\n\nmask_images = [plant_book_mask_image]\n\n# Define which colors match which categories in the images\nbackground, Cell, cancer = range(7)\ncategory_ids = {\n    1:{\n        '(0, 0, 0)': background,\n        '(255, 255, 178)': stroma,\n        '(255, 0, 0)': benign_epithelium,\n        '(0, 255, 0)': Gleason_3,\n        '(255, 127, 0)': Gleason_4,\n        '(0, 255, 255)': Gleason_5,\n        '(127, 127, 127)': Cell\n    }\n}\n\nis_crowd = True\n\n# These ids will be automatically increased as we go\nannotation_id = 1\nimage_id = 1\n\n# Create the annotations\nannotations = []\nfor mask_image in mask_images:\n    sub_masks = create_sub_masks(mask_image)\n    for color, sub_mask in sub_masks.items():\n        category_id = category_ids[image_id][color]\n        print(\"sub mask is\")\n        print(color)\n        annotation = create_sub_mask_annotation(np.array(sub_mask), image_id, category_id, annotation_id, is_crowd)\n        annotations.append(annotation)\n        annotation_id += 1\n    image_id +=1 \nvgg_output_file = os.path.join(\"/kaggle/working/\", f'test1.json')\n\nwith open(vgg_output_file, 'w') as f:\n    json.dump(annotations, f)","metadata":{"execution":{"iopub.status.busy":"2023-03-17T11:59:52.38476Z","iopub.execute_input":"2023-03-17T11:59:52.386065Z","iopub.status.idle":"2023-03-17T11:59:52.44676Z","shell.execute_reply.started":"2023-03-17T11:59:52.385999Z","shell.execute_reply":"2023-03-17T11:59:52.445189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rm annotations.json","metadata":{"execution":{"iopub.status.busy":"2023-03-12T13:25:43.535957Z","iopub.execute_input":"2023-03-12T13:25:43.536964Z","iopub.status.idle":"2023-03-12T13:25:44.540954Z","shell.execute_reply.started":"2023-03-12T13:25:43.53692Z","shell.execute_reply":"2023-03-12T13:25:44.539539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"trying to annotate for multiple images","metadata":{}},{"cell_type":"code","source":"import json\nimport os\nimport pandas as pd\nfrom PIL import Image\nimport numpy as np\n\n# Define which colors match which categories in the images\nbackground, Cell,Cancer = range(3)\ncategory_ids = {\n    1:{\n        '(0, 0, 0)': background,\n        \n        '(0, 255, 0)':  Cancer,\n       \n        '(127, 127, 127)': Cell\n    }\n}\n\nis_crowd = 0\n\n# These ids will be automatically increased as we go\nannotation_id = 1\nimage_id = 1\n\n# Read the CSV file with image IDs\ncsv_file = \"/kaggle/input/prostate-cancer-grade-assessment/train.csv\"\ndf = pd.read_csv(csv_file)\n\n# Create the annotations for each image in the CSV file\nannotations = []\nimages = []\nfor index, row in df.iterrows():\n    fimage_id = row['image_id']\n    file_name = f\"{fimage_id}.png\"\n    if not os.path.isfile('/kaggle/input/learning/karolinska/'+file_name):\n        #print(f\"Image not found: {file_name}\")\n        continue\n    img = Image.open('/kaggle/input/learning/karolinska/'+file_name)\n    print(f\"Image found: {file_name}\")\n    mask_images = [img]\n    for mask_image in mask_images:\n        sub_masks = create_sub_masks(mask_image)\n        for color, sub_mask in sub_masks.items():\n            category_id = category_ids[1][color]\n            \n            if category_id == 2 or 3:\n                category_id = row['isup_grade'] + 1\n                    \n            annotation = create_sub_mask_annotation(np.array(sub_mask), image_id, category_id, annotation_id, is_crowd)\n            annotations.append(annotation)\n            annotation_id += 1\n            \n            \n        \n\n        print(image_id)\n             \n        w,h = img.size\n        image_data = {\n            \"id\": image_id,\n            \"file_name\":file_name,\n            \"width\":w,\n            \"height\":h\n\n        }\n        images.append(image_data)\n        image_id = image_id+1\n\n    if image_id > 1000:\n        break\ncategories= [\n        {\n            \"id\": 1,\n            \"name\": \"isup0\"\n        },\n        {\n            \"id\": 2,\n            \"name\": \"isup_1\"\n        },\n        {\n            \"id\": 3,\n            \"name\": \"isup2\"\n        },\n        {\n            \"id\": 4,\n            \"name\": \"isup3\"\n        },\n            {\n            \"id\": 5,\n            \"name\": \"isup4\"\n        },\n            {\n            \"id\": 6,\n            \"name\": \"isup5\"\n        },\n    ]\n        \nresult = {\n    \"images\":images,\n    \"annotations\":annotations,\n    \"categories\":categories\n}\n# Write the annotations to a JSON file\nvgg_output_file = os.path.join(\"/kaggle/working/\", f'annotations.json')\nwith open(vgg_output_file, 'w') as f:\n    \n    json.dump(result, f)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mkdir preprocessed","metadata":{"execution":{"iopub.status.busy":"2023-03-07T19:29:10.181644Z","iopub.execute_input":"2023-03-07T19:29:10.182259Z","iopub.status.idle":"2023-03-07T19:29:11.211878Z","shell.execute_reply.started":"2023-03-07T19:29:10.182214Z","shell.execute_reply":"2023-03-07T19:29:11.210631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#to zip neccessary images and files for downloading\nimport os\nimport zipfile\n\ndirectory = '/kaggle/working/'\nfiles = os.listdir(directory)\n\nwith zipfile.ZipFile('/kaggle/working/preprocessed.zip', 'r') as zip:\n        zip.extractall('/kaggle/working/')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rm example_coco.json","metadata":{"execution":{"iopub.status.busy":"2023-03-06T05:19:25.582677Z","iopub.execute_input":"2023-03-06T05:19:25.583589Z","iopub.status.idle":"2023-03-06T05:19:26.331719Z","shell.execute_reply.started":"2023-03-06T05:19:25.583553Z","shell.execute_reply":"2023-03-06T05:19:26.330499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The cells below shows two example masks from the dataset. The first mask is from Radboudumc and shows two different grades of cancer (shown in yellow and orange). The second mask is from Karolinska, the region that contains cancer is higlighted in red.\n\nNote that, eventhough a biopsy contains cancer, not all epithelial tissue has to be cancerous. Biopsies can contain a mix of cancerous and healthy tissue.","metadata":{}},{"cell_type":"markdown","source":"for png image","metadata":{}},{"cell_type":"markdown","source":"### Visualizing masks (using matplotlib)\n\nGiven that the masks are just integer matrices, you can also use other packages to display the masks. For example, using matplotlib and a custom color map we can quickly visualize the different cancer regions:","metadata":{}},{"cell_type":"code","source":"mask = openslide.OpenSlide(os.path.join(mask_dir, '00928370e2dfeb8a507667ef1d4efcbb_mask.tiff'))\nmask_data = mask.read_region((0,0), mask.level_count - 1, mask.level_dimensions[-1])\n\nplt.figure()\nplt.title(\"Mask with default cmap\")\nplt.imshow(np.asarray(mask_data)[:,:,0], interpolation='nearest')\nplt.show()\n\nplt.figure()\nplt.title(\"Mask with custom cmap\")\n# Optional: create a custom color map\ncmap = matplotlib.colors.ListedColormap(['black', 'gray', 'green', 'yellow', 'orange', 'red'])\nplt.imshow(np.asarray(mask_data)[:,:,0], cmap=cmap, interpolation='nearest', vmin=0, vmax=5)\nplt.show()\n\nmask.close()","metadata":{"execution":{"iopub.status.busy":"2023-03-24T05:37:04.69816Z","iopub.execute_input":"2023-03-24T05:37:04.6986Z","iopub.status.idle":"2023-03-24T05:37:05.174857Z","shell.execute_reply.started":"2023-03-24T05:37:04.69856Z","shell.execute_reply":"2023-03-24T05:37:05.173711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mask = openslide.OpenSlide(os.path.join(mask_dir, '00928370e2dfeb8a507667ef1d4efcbb_mask.tiff'))\nprint_mask_details(mask, center='radboud')\nmask.close()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mask = openslide.OpenSlide(os.path.join(mask_dir, '00928370e2dfeb8a507667ef1d4efcbb_mask.tiff'))\nprint_mask_details(mask, center='radboud')\nmask.close()","metadata":{"execution":{"iopub.status.busy":"2023-03-25T16:37:03.308266Z","iopub.execute_input":"2023-03-25T16:37:03.308719Z","iopub.status.idle":"2023-03-25T16:37:03.327313Z","shell.execute_reply.started":"2023-03-25T16:37:03.308674Z","shell.execute_reply":"2023-03-25T16:37:03.326065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> ####preprocessing  Mask","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport json\nimport os\nimport pandas as pd\nimport openslide\nfrom PIL import Image\n\ncount = 0\n# Define the mapping of color codes to object classes\n\n\n\n\n# Define the COCO annotation structure\n\n\n\n# Define the path to the folder containing the TIFF mask files\nfolder_path = '/kaggle/input/prostate-cancer-grade-assessment/train_label_masks'\n\n# Read the CSV file containing the image IDs\ncsv_path = '/kaggle/input/prostate-cancer-grade-assessment/train.csv'\ndf = pd.read_csv(csv_path)\n\n# Iterate over the rows in the CSV file and process the corresponding mask file\nfor index, row in df.iterrows():\n    # Extract the image ID from the CSV row\n    image_id = row['image_id']\n    image_provider = row['data_provider']\n    \n    if image_provider != 'karolinska':\n        continue\n\n    try:\n        mask_path = os.path.join(folder_path, image_id + '_mask.tiff')\n        # Load the mask image\n        print(image_id)\n        slide = openslide.OpenSlide(mask_path)\n\n        #mask data save\n        mask_data = slide.read_region((0,0), slide.level_count - 1, slide.level_dimensions[-1])\n            # Mask data is present in the R channel\n        mask_data = mask_data.split()[0]\n\n            # To show the masks we map the raw label values to RGB values\n        preview_palette = np.zeros(shape=768, dtype=int)\n\n                # Mapping: {0: background, 1: stroma, 2: benign epithelium, 3: Gleason 3, 4: Gleason 4, 5: Gleason 5}\n        preview_palette[0:18] = (np.array([0, 0, 0, 0.5, 0.5, 0.5, 0, 1, 0, 1, 1, 0.7, 1, 0.5, 0, 1, 0, 0]) * 255).astype(int)\n        mask_data.putpalette(data=preview_palette.tolist())\n        mask_data = mask_data.convert(mode='RGB')\n            #mask_data.thumbnail(size=max_size, resample=0)\n        count=count+1\n        print(count)\n        save_dir = os.path.join('/kaggle/working/karolinska',(image_id+\".png\"))\n        mask_data.save(save_dir,'PNG')\n    except openslide.OpenSlideUnsupportedFormatError:\n        print(f\"Unsupported or missing image file: {image_id}. Skipping...\")\n        continue\n    except FileNotFoundError:\n        print(f\"File not found: {image_id}. Skipping...\")\n        continue\n    except Exception as e:\n        print(f\"Error processing image {image_id}: {e}\")\n        continue","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mkdir karolinska","metadata":{"execution":{"iopub.status.busy":"2023-03-12T04:27:27.172479Z","iopub.execute_input":"2023-03-12T04:27:27.1729Z","iopub.status.idle":"2023-03-12T04:27:28.259368Z","shell.execute_reply.started":"2023-03-12T04:27:27.172862Z","shell.execute_reply":"2023-03-12T04:27:28.257759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Overlaying masks on the slides\n\nAs the masks have the same dimension as the slides, we can overlay the masks on the tissue to directly see which areas are cancerous. This overlay can help you identifying the different growth patterns. To do this, we load both the mask and the biopsy and merge them using PIL.\n\n**Tip:** Want to view the slides in a more interactive way? Using a WSI viewer you can interactively view the slides. Examples of open source viewers that can open the PANDA dataset are [ASAP](https://github.com/computationalpathologygroup/ASAP) and [QuPath](https://qupath.github.io/). ASAP can also overlay the masks on top of the images using the \"Overlay\" functionality. If you use Qupath, and the images do not load, try changing the file extension to `.vtif`.","metadata":{}},{"cell_type":"code","source":"def overlay_mask_on_slide(slide, mask, center='radboud', alpha=0.8, max_size=(800, 800)):\n    \"\"\"Show a mask overlayed on a slide.\"\"\"\n\n    if center not in ['radboud', 'karolinska']:\n        raise Exception(\"Unsupported palette, should be one of [radboud, karolinska].\")\n\n    # Load data from the highest level\n    slide_data = slide.read_region((0,0), slide.level_count - 1, slide.level_dimensions[-1])\n    mask_data = mask.read_region((0,0), mask.level_count - 1, mask.level_dimensions[-1])\n\n    # Mask data is present in the R channel\n    mask_data = mask_data.split()[0]\n\n    # Create alpha mask\n    alpha_int = int(round(255*alpha))\n    if center == 'radboud':\n        alpha_content = np.less(mask_data.split()[0], 2).astype('uint8') * alpha_int + (255 - alpha_int)\n    elif center == 'karolinska':\n        alpha_content = np.less(mask_data.split()[0], 1).astype('uint8') * alpha_int + (255 - alpha_int)\n    \n    alpha_content = PIL.Image.fromarray(alpha_content)\n    preview_palette = np.zeros(shape=768, dtype=int)\n    \n    if center == 'radboud':\n        # Mapping: {0: background, 1: stroma, 2: benign epithelium, 3: Gleason 3, 4: Gleason 4, 5: Gleason 5}\n        preview_palette[0:18] = (np.array([0, 0, 0, 0.5, 0.5, 0.5, 0, 1, 0, 1, 1, 0.7, 1, 0.5, 0, 1, 0, 0]) * 255).astype(int)\n    elif center == 'karolinska':\n        # Mapping: {0: background, 1: benign, 2: cancer}\n        preview_palette[0:9] = (np.array([0, 0, 0, 0, 1, 0, 1, 0, 0]) * 255).astype(int)\n    \n    mask_data.putpalette(data=preview_palette.tolist())\n    mask_rgb = mask_data.convert(mode='RGB')\n\n    overlayed_image = PIL.Image.composite(image1=slide_data, image2=mask_rgb, mask=alpha_content)\n    overlayed_image.thumbnail(size=max_size, resample=0)\n\n    display(overlayed_image)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T11:29:07.974611Z","iopub.execute_input":"2023-03-11T11:29:07.975052Z","iopub.status.idle":"2023-03-11T11:29:08.015393Z","shell.execute_reply.started":"2023-03-11T11:29:07.975012Z","shell.execute_reply":"2023-03-11T11:29:08.014271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> Note: In the example below you can also observe a few pen markings on the slide (dark green smudges). These markings are not part of the tissue but were made by the pathologists who originally checked this case. These pen markings are available on some slides in the training set.","metadata":{}},{"cell_type":"code","source":"slide = openslide.OpenSlide(os.path.join(data_dir, '08ab45297bfe652cc0397f4b37719ba1.tiff'))\nmask = openslide.OpenSlide(os.path.join(mask_dir, '08ab45297bfe652cc0397f4b37719ba1_mask.tiff'))\noverlay_mask_on_slide(slide, mask, center='radboud')\nslide.close()\nmask.close()","metadata":{"execution":{"iopub.status.busy":"2023-03-11T11:34:32.537771Z","iopub.execute_input":"2023-03-11T11:34:32.538174Z","iopub.status.idle":"2023-03-11T11:34:32.722652Z","shell.execute_reply.started":"2023-03-11T11:34:32.53814Z","shell.execute_reply":"2023-03-11T11:34:32.721647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"slide = openslide.OpenSlide(os.path.join(data_dir, '090a77c517a7a2caa23e443a77a78bc7.tiff'))\nmask = openslide.OpenSlide(os.path.join(mask_dir, '090a77c517a7a2caa23e443a77a78bc7_mask.tiff'))\noverlay_mask_on_slide(slide, mask, center='karolinska', alpha=0.6)\nslide.close()\nmask.close()","metadata":{"execution":{"iopub.status.busy":"2023-03-11T11:34:57.834002Z","iopub.execute_input":"2023-03-11T11:34:57.834413Z","iopub.status.idle":"2023-03-11T11:34:58.242899Z","shell.execute_reply.started":"2023-03-11T11:34:57.834379Z","shell.execute_reply":"2023-03-11T11:34:58.241564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow_hub as hub\n    \nkeras_layer = hub.KerasLayer('https://kaggle.com/models/tensorflow/mask-rcnn-inception-resnet-v2/frameworks/TensorFlow2/variations/1024x1024/versions/1')\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!python --version","metadata":{"execution":{"iopub.status.busy":"2023-03-12T17:53:29.493776Z","iopub.execute_input":"2023-03-12T17:53:29.494953Z","iopub.status.idle":"2023-03-12T17:53:30.616235Z","shell.execute_reply.started":"2023-03-12T17:53:29.494893Z","shell.execute_reply":"2023-03-12T17:53:30.615015Z"},"trusted":true},"execution_count":null,"outputs":[]}]}