{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"}],"dockerImageVersionId":30627,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nos.environ[\"KERAS_BACKEND\"] = \"jax\"\n\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nfrom tensorflow.keras.models import Sequential\nimport keras_cv\n\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nplt.rcParams[\"axes.grid\"] = False\n# Set the style for the plot\nsns.set(style=\"whitegrid\")\n\nimport itertools\n\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    print(os.path.join(dirname))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-01-02T13:04:39.735534Z","iopub.execute_input":"2024-01-02T13:04:39.736128Z","iopub.status.idle":"2024-01-02T13:05:03.544872Z","shell.execute_reply.started":"2024-01-02T13:04:39.736032Z","shell.execute_reply":"2024-01-02T13:05:03.543832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Configurations","metadata":{}},{"cell_type":"code","source":"is_submission = False\n\n# Reproducibility\nSEED = 42\n\n# Training\ntrain_csv_path = \"/kaggle/input/UBC-OCEAN/train.csv\"\ntrain_thumbnail_paths = \"/kaggle/input/UBC-OCEAN/train_thumbnails\"\nimg_height = 256\nimg_width = 256\nbatch_size = 16\nlearning_rate = 1e-3\nepochs = 20\n\n# Inference\ntest_csv_path = \"/kaggle/input/UBC-OCEAN/test.csv\"\ntest_thumbnail_paths = \"/kaggle/input/UBC-OCEAN/test_thumbnails\"\n\n# For reproducibility\nkeras.utils.set_random_seed(seed=SEED)","metadata":{"execution":{"iopub.status.busy":"2024-01-02T13:05:03.546654Z","iopub.execute_input":"2024-01-02T13:05:03.548041Z","iopub.status.idle":"2024-01-02T13:05:03.556305Z","shell.execute_reply.started":"2024-01-02T13:05:03.547996Z","shell.execute_reply":"2024-01-02T13:05:03.555261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Utils","metadata":{}},{"cell_type":"code","source":"mapping = {\"HGSC\": 0, \"LGSC\": 1, \"EC\": 2, \"CC\": 3, \"MC\": 4}\nclass_names = {}\nfor name, val in mapping.items():\n    class_names[val] = name\n\ndef read_image(path):\n    file = tf.io.read_file(path)\n    image = tf.io.decode_png(file, 3)\n    #image = tf.image.convert_image_dtype(image, dtype=tf.float32)\n    image = tf.image.resize(image, (img_height, img_width))\n    # Rescale with mean ~= 0 and variance ~= 1\n    #image = tf.image.per_image_standardization(image)\n    # Rescale with values between 0 and 1\n    image = tf.keras.layers.Rescaling(1./255)(image)\n    return image\n\n\ndef load_df():\n    df = pd.read_csv(train_csv_path)\n\n    # Create the thumbnail df where is_tma == False\n    df = df[df[\"is_tma\"] == False]\n    \n    # Define paths to thumbnails\n    df[\"image_thumbnail_path\"] = df[\"image_id\"].apply(lambda x: f\"{train_thumbnail_paths}/{x}_thumbnail.png\")\n    \n    # Label encoding\n    df[\"label_encoded\"] = df[\"label\"].map(mapping)\n    \n    # Reordering columns\n    df = df[[\"image_id\", \"image_width\", \"image_height\", \"is_tma\", \"image_thumbnail_path\", \"label\", \"label_encoded\"]]\n    \n    # Class weights\n    labels = df[\"label\"].values\n    label_names = df[\"label\"].unique()\n    class_weights = (1 / df[\"label_encoded\"].value_counts()) * (labels.size / label_names.size)\n    class_weights.name = \"weights\"\n    \n    return df, class_weights\n    \n    \ndef create_tf_dataset(df, ohe=False):\n    x = (\n        tf.data.Dataset.from_tensor_slices(df[\"image_thumbnail_path\"].values)\n        .map(read_image, num_parallel_calls=tf.data.AUTOTUNE)\n    )\n    \n    if ohe:\n        labels_ohe = pd.get_dummies(df[\"label\"], dtype=np.int8)\n        y = tf.data.Dataset.from_tensor_slices(labels_ohe.values)\n    else:\n        y = tf.data.Dataset.from_tensor_slices(df[\"label_encoded\"].values)\n\n    # Zip the x and y together\n    ds = tf.data.Dataset.zip((x, y))\n    ds = ds.batch(batch_size)\n    \n    return ds","metadata":{"execution":{"iopub.status.busy":"2024-01-02T13:05:03.557493Z","iopub.execute_input":"2024-01-02T13:05:03.558044Z","iopub.status.idle":"2024-01-02T13:05:03.60635Z","shell.execute_reply.started":"2024-01-02T13:05:03.558011Z","shell.execute_reply":"2024-01-02T13:05:03.604776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create datasets","metadata":{}},{"cell_type":"code","source":"df, class_weights = load_df()\nds = create_tf_dataset(df)","metadata":{"execution":{"iopub.status.busy":"2024-01-02T13:05:03.609108Z","iopub.execute_input":"2024-01-02T13:05:03.609741Z","iopub.status.idle":"2024-01-02T13:05:03.90876Z","shell.execute_reply.started":"2024-01-02T13:05:03.609696Z","shell.execute_reply":"2024-01-02T13:05:03.907323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's visualize some images","metadata":{}},{"cell_type":"code","source":"nrows = 3\nncols = 3\nplt.figure(figsize=(10, 10))\nfor images, labels in ds.take(1):\n    for i in range(nrows*ncols):\n        ax = plt.subplot(nrows, ncols, i + 1)\n#         plt.imshow(images[i].numpy().astype(\"uint8\"))\n        plt.imshow(images[i].numpy())\n        plt.title(class_names[labels[i].numpy()])\n        plt.axis(\"off\")","metadata":{"execution":{"iopub.status.busy":"2024-01-02T13:05:03.910139Z","iopub.execute_input":"2024-01-02T13:05:03.91054Z","iopub.status.idle":"2024-01-02T13:05:08.536703Z","shell.execute_reply.started":"2024-01-02T13:05:03.910506Z","shell.execute_reply":"2024-01-02T13:05:08.535501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It's hard to see differences between categories from these images.\nFurthermore, when resizing the image with a lower size we lose information because the image quality decreases.","metadata":{}},{"cell_type":"code","source":"plt.rcParams[\"axes.grid\"] = False","metadata":{"execution":{"iopub.status.busy":"2024-01-02T13:05:08.538628Z","iopub.execute_input":"2024-01-02T13:05:08.539311Z","iopub.status.idle":"2024-01-02T13:05:08.545292Z","shell.execute_reply.started":"2024-01-02T13:05:08.539271Z","shell.execute_reply":"2024-01-02T13:05:08.543941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_path = \"../input/UBC-OCEAN/train_thumbnails/10077_thumbnail.png\"\nimage = tf.image.decode_image(tf.io.read_file(image_path), channels=3)\nimage = tf.image.convert_image_dtype(image, dtype=tf.float32)\nresized_image = tf.image.resize(image, (256,256))\n\nfig, axs = plt.subplots(nrows=1, ncols=2, figsize=(16,12))\naxs[0].imshow(image.numpy())\naxs[0].set_title(f\"Original image of shape {image.shape[0]}x{image.shape[1]}\")\naxs[1].imshow(resized_image.numpy())\naxs[1].set_title(f\"Resized image of shape {resized_image.shape[0]}x{resized_image.shape[1]}\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-02T13:05:08.547601Z","iopub.execute_input":"2024-01-02T13:05:08.54813Z","iopub.status.idle":"2024-01-02T13:05:11.514199Z","shell.execute_reply.started":"2024-01-02T13:05:08.548068Z","shell.execute_reply":"2024-01-02T13:05:11.512643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As it has been mentioned in other notebooks, in order for the classification model to gain more insight into the data, we are going to feed it with tiles extracted from the images.","metadata":{}},{"cell_type":"markdown","source":"# Extracting tiles with Tensorflow","metadata":{}},{"cell_type":"markdown","source":"## Example usage on a single image","metadata":{}},{"cell_type":"code","source":"@tf.function\ndef tile_image(image, patch_size, stride):\n    # Expand dimensions to make it a batch of 1\n    image = tf.expand_dims(image, 0)\n    \n    # Extract patches using tf.image.extract_patches\n    patches = tf.image.extract_patches(\n        images=image,\n        sizes=[1, patch_size, patch_size, 1],\n        strides=[1, stride, stride, 1],\n        rates=[1, 1, 1, 1],\n        padding=\"SAME\"\n    )\n    nx = patches.shape[1] # number of patches along x\n    ny = patches.shape[2] # number of patches along y\n    \n    # Reshape the patches to a 2D tensor\n    num_patches = tf.reduce_prod(tf.shape(patches)[1:3])\n    patches = tf.reshape(patches, [num_patches, patch_size, patch_size, tf.shape(image)[-1]])\n    \n    return patches, nx, ny\n\n# Example usage\nimage_path = \"../input/UBC-OCEAN/train_thumbnails/10077_thumbnail.png\"\nimage = tf.image.decode_image(tf.io.read_file(image_path), channels=3)\nimage = tf.image.convert_image_dtype(image, dtype=tf.float32)\n\n# Define patch size and stride\npatch_size = 512\nstride = 512\n\n# Tile the image\npatches, nx, ny = tile_image(image, patch_size, stride)\n\nprint(\"Original image shape:\", image.shape)\nprint(\"Patched image shape:\", patches.shape)\n\n# Display the image\nplt.imshow(image)\nplt.title(\"Original image\", va=\"bottom\")\nplt.axis(\"off\")\n\n# Display the patches\nedge_color = \"red\"\nnrows = int(nx)\nncols = int(ny)\nfig, axs = plt.subplots(nrows, ncols, sharex=True, sharey=True, figsize=(6,5))\nfig.suptitle(\"Extracted tiles\")\nplt.subplots_adjust(hspace=0, wspace=0)\nn = 0\nfor p in itertools.product(range(nrows), range(ncols)):\n    ax = axs[p[0], p[1]]\n    ax.imshow(patches[n].numpy())\n    ax.tick_params(left=False, right=False, labelleft=False, labelbottom=False, bottom=False)\n    for spine in ax.spines.values():\n        spine.set_color(edge_color)\n#     ax.axis(\"off\")\n    n += 1","metadata":{"execution":{"iopub.status.busy":"2024-01-02T13:05:11.516019Z","iopub.execute_input":"2024-01-02T13:05:11.516523Z","iopub.status.idle":"2024-01-02T13:05:22.820561Z","shell.execute_reply.started":"2024-01-02T13:05:11.516489Z","shell.execute_reply":"2024-01-02T13:05:22.819188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Application to the whole dataset","metadata":{}},{"cell_type":"code","source":"# We import the images with a size of 2048x2048 to preserve image quality\nimg_height = 2048\nimg_width = 2048\nbatch_size = 16\nimage_ds = (\n    tf.data.Dataset.from_tensor_slices(df[\"image_thumbnail_path\"].values)\n    .map(read_image, num_parallel_calls=tf.data.AUTOTUNE)\n)\nimage_ds = image_ds.batch(batch_size)\nlabels = df[\"label\"].values\nids = df[\"image_id\"].values.astype(str)","metadata":{"execution":{"iopub.status.busy":"2024-01-02T13:05:22.822652Z","iopub.execute_input":"2024-01-02T13:05:22.823882Z","iopub.status.idle":"2024-01-02T13:05:22.868298Z","shell.execute_reply.started":"2024-01-02T13:05:22.823828Z","shell.execute_reply":"2024-01-02T13:05:22.866796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%bash\nrm -r */\nmkdir -p HGSC LGSC EC CC MC\nls","metadata":{"execution":{"iopub.status.busy":"2024-01-02T13:05:22.873737Z","iopub.execute_input":"2024-01-02T13:05:22.874258Z","iopub.status.idle":"2024-01-02T13:05:22.905595Z","shell.execute_reply.started":"2024-01-02T13:05:22.874211Z","shell.execute_reply":"2024-01-02T13:05:22.904493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\npatch_size = 512\nstride = 512\nn = 0\nfor image_batch in image_ds:\n    for i in range(image_batch.shape[0]):\n        patches, nx, ny = tile_image(image_batch[i], patch_size, stride)\n        for j in range(patches.shape[0]):\n            keras.utils.save_img(\n                path=f\"{labels[n]}/{ids[n]}_thumbnail_patch_{j}.png\", x=patches[j].numpy(), data_format=\"channels_last\", scale=True\n            )\n        n += 1","metadata":{"execution":{"iopub.status.busy":"2024-01-02T13:05:22.907178Z","iopub.execute_input":"2024-01-02T13:05:22.907733Z","iopub.status.idle":"2024-01-02T13:15:21.96112Z","shell.execute_reply.started":"2024-01-02T13:05:22.907685Z","shell.execute_reply":"2024-01-02T13:15:21.95898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's check out we have the correct number of patches","metadata":{}},{"cell_type":"code","source":"n_patches = 0\nfor directory in os.listdir():\n    if directory[0] == \".\" or len(directory.split(\".\"))>1:\n        continue\n    n_patches += len(os.listdir(directory))\nn_patches","metadata":{"execution":{"iopub.status.busy":"2024-01-02T13:15:21.964018Z","iopub.execute_input":"2024-01-02T13:15:21.964833Z","iopub.status.idle":"2024-01-02T13:15:21.984103Z","shell.execute_reply.started":"2024-01-02T13:15:21.964785Z","shell.execute_reply":"2024-01-02T13:15:21.982526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(df) * 16)","metadata":{"execution":{"iopub.status.busy":"2024-01-02T13:15:21.986507Z","iopub.execute_input":"2024-01-02T13:15:21.987625Z","iopub.status.idle":"2024-01-02T13:15:21.99707Z","shell.execute_reply.started":"2024-01-02T13:15:21.987571Z","shell.execute_reply":"2024-01-02T13:15:21.99513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The number of patches effectively correspond to the number of images multiplied by the number of patches per image. <br>\nLet's check out the size of the dataset we just created :","metadata":{}},{"cell_type":"code","source":"!du -sh ./","metadata":{"execution":{"iopub.status.busy":"2024-01-02T13:15:21.999341Z","iopub.execute_input":"2024-01-02T13:15:21.999798Z","iopub.status.idle":"2024-01-02T13:15:23.21528Z","shell.execute_reply.started":"2024-01-02T13:15:21.999763Z","shell.execute_reply":"2024-01-02T13:15:23.213672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!du -sh ./*","metadata":{"execution":{"iopub.status.busy":"2024-01-02T13:15:23.218052Z","iopub.execute_input":"2024-01-02T13:15:23.218985Z","iopub.status.idle":"2024-01-02T13:15:24.401046Z","shell.execute_reply.started":"2024-01-02T13:15:23.218922Z","shell.execute_reply":"2024-01-02T13:15:24.398666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The dataset is `2.1G` and the original dataset of thumbnail images was about `3.4G`. This is because we resized the images by 2048x2048 which is, for most of images, lower than the original image size.","metadata":{}},{"cell_type":"markdown","source":"# Test\n\nLet's verify that we correctly saved the patches by randomly selecting an image and comparing it with the patches we extracted.","metadata":{}},{"cell_type":"code","source":"i = np.random.choice(len(ids))\nprint(i)\nnum_patches = (nx*ny).numpy()\npatches_paths = [f\"./{labels[i]}/{ids[i]}_thumbnail_patch_{j}.png\" for j in range(num_patches)]\npatches_paths","metadata":{"execution":{"iopub.status.busy":"2024-01-02T13:15:24.404639Z","iopub.execute_input":"2024-01-02T13:15:24.405359Z","iopub.status.idle":"2024-01-02T13:15:24.435712Z","shell.execute_reply.started":"2024-01-02T13:15:24.405294Z","shell.execute_reply":"2024-01-02T13:15:24.434092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"loaded_patches = (\n    tf.data.Dataset.from_tensor_slices(patches_paths)\n    .map(read_image, num_parallel_calls=tf.data.AUTOTUNE)\n)\nloaded_patches = loaded_patches.batch(num_patches)\nloaded_patches = next(iter(loaded_patches))","metadata":{"execution":{"iopub.status.busy":"2024-01-02T13:15:24.437922Z","iopub.execute_input":"2024-01-02T13:15:24.438534Z","iopub.status.idle":"2024-01-02T13:15:25.498678Z","shell.execute_reply.started":"2024-01-02T13:15:24.438488Z","shell.execute_reply":"2024-01-02T13:15:25.496916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot image\nimage_path = f\"{train_thumbnail_paths}/{ids[i]}_thumbnail.png\"\nimage = tf.image.decode_image(tf.io.read_file(image_path), channels=3)\nimage = tf.image.convert_image_dtype(image, dtype=tf.float32)\nimage = tf.image.resize(image, (img_height,img_width))\nplt.figure(figsize=(5,5))\nplt.imshow(image)\nplt.tick_params(left=False, right=False, labelleft=False, labelbottom=False, bottom=False)\n\n# plot patches\nedge_color = \"red\"\nnrows = nx.numpy()\nncols = ny.numpy()\nfig, axs = plt.subplots(nrows, ncols, sharex=True, sharey=True, figsize=(5,5))\nplt.subplots_adjust(hspace=0, wspace=0)\nn = 0\nfor p in itertools.product(range(nrows), range(ncols)):\n    ax = axs[p[0], p[1]]\n    ax.imshow(loaded_patches[n].numpy(), vmin=0, vmax=255)\n    ax.tick_params(left=False, right=False, labelleft=False, labelbottom=False, bottom=False)\n    for spine in ax.spines.values():\n        spine.set_color(edge_color)\n#     ax.axis(\"off\")\n    n += 1","metadata":{"execution":{"iopub.status.busy":"2024-01-02T13:15:25.501458Z","iopub.execute_input":"2024-01-02T13:15:25.502055Z","iopub.status.idle":"2024-01-02T13:15:49.061792Z","shell.execute_reply.started":"2024-01-02T13:15:25.502006Z","shell.execute_reply":"2024-01-02T13:15:49.060167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It looks good !","metadata":{"execution":{"iopub.status.busy":"2023-12-26T20:28:36.270955Z","iopub.execute_input":"2023-12-26T20:28:36.273201Z","iopub.status.idle":"2023-12-26T20:28:38.311261Z","shell.execute_reply.started":"2023-12-26T20:28:36.273126Z","shell.execute_reply":"2023-12-26T20:28:38.309966Z"}}},{"cell_type":"markdown","source":"# Outlook\n\n- Use the [ supplemental masks](https://www.kaggle.com/datasets/sohier/ubc-ovarian-cancer-competition-supplemental-masks) to classify patches as cancerous (specifying which cancer type), healthy, or necrotic.\n\n- Train models\n\n- Write an algorithm so that when one or more tiles of an image are classified as cancerous, then classify the image as it is.\n\n- process raw images. Notice that TMA images are at 40x magnification compare to WSI images that are at 20x (thumbnails only correspond to WSIs).","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}