{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"}],"dockerImageVersionId":30558,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport torch\nimport torchvision\nimport os\nimport json\nimport plotly.express as px\nimport PIL\nimport regex\nimport matplotlib.pyplot as plt\nimport random","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-11-19T17:47:09.241927Z","iopub.execute_input":"2023-11-19T17:47:09.242371Z","iopub.status.idle":"2023-11-19T17:47:14.554998Z","shell.execute_reply.started":"2023-11-19T17:47:09.242336Z","shell.execute_reply":"2023-11-19T17:47:14.553814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA","metadata":{"execution":{"iopub.status.busy":"2023-11-04T22:32:48.905838Z","iopub.execute_input":"2023-11-04T22:32:48.906373Z","iopub.status.idle":"2023-11-04T22:32:49.387119Z","shell.execute_reply.started":"2023-11-04T22:32:48.906336Z","shell.execute_reply":"2023-11-04T22:32:49.385724Z"}}},{"cell_type":"markdown","source":"**If you find this notebook useful, remember to upvote** 😀","metadata":{}},{"cell_type":"markdown","source":"## Overview","metadata":{}},{"cell_type":"code","source":"DATASET_FOLDER = \"/kaggle/input/UBC-OCEAN\"","metadata":{"execution":{"iopub.status.busy":"2023-11-19T17:47:14.557383Z","iopub.execute_input":"2023-11-19T17:47:14.557908Z","iopub.status.idle":"2023-11-19T17:47:14.563666Z","shell.execute_reply.started":"2023-11-19T17:47:14.557878Z","shell.execute_reply":"2023-11-19T17:47:14.562454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(os.path.join(DATASET_FOLDER, \"train.csv\"))\n\ndisplay(train_df.head())\ndisplay(train_df.info())\n\ntrain_num_images = len(os.listdir(os.path.join(DATASET_FOLDER, \"train_images\")))\ntrain_num_thumbnails = len(os.listdir(os.path.join(DATASET_FOLDER, \"train_thumbnails\")))\n\nprint()\nprint(f\"Number of images: {train_num_images}\")\nprint(f\"Number of thumbnails: {train_num_thumbnails}\")","metadata":{"execution":{"iopub.status.busy":"2023-11-19T17:47:14.565357Z","iopub.execute_input":"2023-11-19T17:47:14.565706Z","iopub.status.idle":"2023-11-19T17:47:14.907282Z","shell.execute_reply.started":"2023-11-19T17:47:14.565676Z","shell.execute_reply":"2023-11-19T17:47:14.905869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are only 538 training images, which is quite a small dataset. The images themselves take up a lot of space (up to 2.5 GB each!), presumably due to them having large dimensions and high resolution. There are no null values.","metadata":{}},{"cell_type":"code","source":"test_df = pd.read_csv(os.path.join(DATASET_FOLDER, \"test.csv\"))\ndisplay(test_df.head())\n\ntest_num_images = len(os.listdir(os.path.join(DATASET_FOLDER, \"test_images\")))\ntest_num_thumbnails = len(os.listdir(os.path.join(DATASET_FOLDER, \"test_thumbnails\")))\n\nprint()\nprint(f\"Number of images: {test_num_images}\")\nprint(f\"Number of thumbnails: {test_num_thumbnails}\")","metadata":{"execution":{"iopub.status.busy":"2023-11-19T17:47:14.908726Z","iopub.execute_input":"2023-11-19T17:47:14.909104Z","iopub.status.idle":"2023-11-19T17:47:14.931891Z","shell.execute_reply.started":"2023-11-19T17:47:14.909065Z","shell.execute_reply":"2023-11-19T17:47:14.930804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There is only one test image. It must just be an example to demonstrate what the real test data will look like.\n\nAccording to the [dataset description](https://www.kaggle.com/competitions/UBC-OCEAN/data), there will be about 2000 test images.","metadata":{}},{"cell_type":"markdown","source":"## What is updated_image_ids.json?","metadata":{}},{"cell_type":"markdown","source":"`updated_image_ids.json` lists the IDs of the images that had to be updated due to flawed masks. See https://www.kaggle.com/competitions/UBC-OCEAN/discussion/451892 for details.","metadata":{}},{"cell_type":"code","source":"with open(os.path.join(DATASET_FOLDER, \"updated_image_ids.json\")) as f:\n    updated_image_ids = json.load(f)\n\nprint(updated_image_ids)","metadata":{"execution":{"iopub.status.busy":"2023-11-19T17:47:14.936131Z","iopub.execute_input":"2023-11-19T17:47:14.936489Z","iopub.status.idle":"2023-11-19T17:47:14.947737Z","shell.execute_reply.started":"2023-11-19T17:47:14.93646Z","shell.execute_reply":"2023-11-19T17:47:14.946753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Labels","metadata":{}},{"cell_type":"code","source":"# Plot label distribution\n\ntrain_df[[\"label\"]].value_counts().plot.pie(ylabel=\"label\", autopct=\"%1.1f%%\", title=\"Labels\")","metadata":{"execution":{"iopub.status.busy":"2023-11-19T17:47:14.949406Z","iopub.execute_input":"2023-11-19T17:47:14.949728Z","iopub.status.idle":"2023-11-19T17:47:15.350926Z","shell.execute_reply.started":"2023-11-19T17:47:14.949693Z","shell.execute_reply":"2023-11-19T17:47:15.349457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The dataset is quite imbalanced, with HGSC dominating at 41% and MC and LGSC being less than 9% each.\n\nSince the evaluation metric for the competition is balanced accuracy (see [here](https://www.kaggle.com/competitions/UBC-OCEAN/overview/evaluation)), we need to pay attention to this class imbalance when training a model.","metadata":{}},{"cell_type":"code","source":"# Plot label distribution by is_tma\n\ntrain_labels_by_is_tma = train_df.groupby(\"is_tma\", as_index=False)[[\"label\"]].value_counts()\ntrain_labels_by_is_tma = train_labels_by_is_tma.pivot(index=\"label\", columns=\"is_tma\", values=\"count\")\ntrain_labels_by_is_tma = train_labels_by_is_tma.rename(columns={True: \"TMA\", False: \"WSI\"})\ndisplay(train_labels_by_is_tma.T)\n_ = train_labels_by_is_tma.plot.pie(\n    subplots=True,\n    figsize=(10, 10),\n    ylabel=\"\",\n    legend=False,\n    title=[f\"Labels for {x} images\" for x in list(train_labels_by_is_tma.columns)],\n    autopct=\"%1.1f%%\")","metadata":{"execution":{"iopub.status.busy":"2023-11-19T17:47:15.353522Z","iopub.execute_input":"2023-11-19T17:47:15.355188Z","iopub.status.idle":"2023-11-19T17:47:15.730422Z","shell.execute_reply.started":"2023-11-19T17:47:15.355137Z","shell.execute_reply":"2023-11-19T17:47:15.728495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The WSI images are the source of the imbalance, whereas the TMA images are perfectly balanced between the classes.","metadata":{}},{"cell_type":"markdown","source":"## TMA vs WSI images\n\nThe images come in two types. WSI images are larger, less magnified, and have thumbnails. TMA images are smaller, more magnified, and do not have thumbnails. There is a column `is_tma` in `train.csv` to distinguish the two, but this flag is not present in the test data. According to the [dataset description](https://www.kaggle.com/competitions/UBC-OCEAN/data), most of the test data will be TMAs.","metadata":{}},{"cell_type":"code","source":"# Plot the distribution of is_tma\n\ntrain_tma_counts = train_df[[\"is_tma\"]].value_counts().rename({True: \"TMA\", False: \"WSI\"})\ndisplay(train_tma_counts)\n\ntrain_tma_counts.plot.pie(ylabel=\"label\", autopct=\"%1.1f%%\")","metadata":{"execution":{"iopub.status.busy":"2023-11-19T17:47:15.732104Z","iopub.execute_input":"2023-11-19T17:47:15.732893Z","iopub.status.idle":"2023-11-19T17:47:15.874284Z","shell.execute_reply.started":"2023-11-19T17:47:15.732857Z","shell.execute_reply":"2023-11-19T17:47:15.873031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Even though most of the test data will be TMA images, the vast majority of the training data consists of WSI images. There are only 25 (4.6%) TMA images in the training set.\n\nThe \"textbook approach\" to creating a validation set would be to pull out mostly (or entirely) TMA images, so the validation set distribution matches the test set distribution as closely as possible. On the other hand, the model may perform better if it sees some of the TMA images during training. The decision of how to create a validation set may depend on how similar the WSI and TMA images look to each other, and how much variation there is within the TMA images.","metadata":{}},{"cell_type":"markdown","source":"## Image sizes","metadata":{}},{"cell_type":"code","source":"# Show image size by label\n\ndf_sizes = train_df[[\"image_width\", \"image_height\", \"label\"]].copy(deep=True)\nfor col in (\"image_width\", \"image_height\"):\n    df_sizes[col] = [round(i / 1000) * 1000 for i in df_sizes[col]]\n\ndf_sizes = df_sizes.groupby([\"image_width\", \"image_height\", \"label\"], as_index=False).size()\nfig = px.scatter(\n    df_sizes,\n    x=\"image_width\",\n    y=\"image_height\",\n    size=\"size\",\n    color=\"label\",\n    height=450,\n    width=900,\n    title=\"Image size by label\")\nfig.update_layout(title_x=0.5)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-19T17:47:15.876148Z","iopub.execute_input":"2023-11-19T17:47:15.876856Z","iopub.status.idle":"2023-11-19T17:47:17.980405Z","shell.execute_reply.started":"2023-11-19T17:47:15.876814Z","shell.execute_reply":"2023-11-19T17:47:17.979132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There is no clear difference between image dimensions based on label, but there is some difference, e.g. MC images tend to have greater height.","metadata":{}},{"cell_type":"code","source":"# Show image size by image type (WSI vs TMA)\n\ndf_sizes = train_df[[\"image_width\", \"image_height\", \"is_tma\"]].copy(deep=True)\nfor col in (\"image_width\", \"image_height\"):\n    df_sizes[col] = [round(i / 1000) * 1000 for i in df_sizes[col]]\n\ndf_sizes = df_sizes.groupby([\"image_width\", \"image_height\", \"is_tma\"], as_index=False).size()\nfig = px.scatter(\n    df_sizes,\n    x=\"image_width\",\n    y=\"image_height\",\n    size=\"size\",\n    color=\"is_tma\",\n    height=450,\n    width=900,\n    title=\"Image size by type (WSI / TMA)\")\nfig.update_layout(title_x=0.5)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-19T17:47:17.981674Z","iopub.execute_input":"2023-11-19T17:47:17.98226Z","iopub.status.idle":"2023-11-19T17:47:18.085568Z","shell.execute_reply.started":"2023-11-19T17:47:17.982227Z","shell.execute_reply":"2023-11-19T17:47:18.084262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There is a clear difference between WSI and TMA images. WSI images vary in size a lot and are much larger than TMA images.\n\nWSI images roughly range from (width x height) 17,000 x 15,000 to 80,000 x 45,000.","metadata":{}},{"cell_type":"code","source":"# TMA image sizes\n\ndf_sizes = train_df[train_df[\"is_tma\"] == True][[\"image_width\", \"image_height\"]].copy(deep=True)\n\ndf_sizes = df_sizes.groupby([\"image_width\", \"image_height\"], as_index=False).size()\nfig = px.scatter(\n    df_sizes,\n    x=\"image_width\",\n    y=\"image_height\",\n    size=\"size\",\n    height=450,\n    width=900,\n    title=\"TMA image sizes\")\nfig.update_layout(title_x=0.5)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-19T17:47:18.0868Z","iopub.execute_input":"2023-11-19T17:47:18.087154Z","iopub.status.idle":"2023-11-19T17:47:18.173415Z","shell.execute_reply.started":"2023-11-19T17:47:18.087124Z","shell.execute_reply":"2023-11-19T17:47:18.172156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There is much less variation in TMA image sizes: there is only 2964 x 2964 and 3388 x 3388.","metadata":{}},{"cell_type":"code","source":"# Show thumbnail sizes\n\ntrain_thumbnail_sizes = []\n\nfor path in os.listdir(os.path.join(DATASET_FOLDER, \"train_thumbnails\")):\n    id = regex.match(r\"^([0-9]+)_thumbnail\\.png$\", path).group(1)\n    path = os.path.join(DATASET_FOLDER, \"train_thumbnails\", path)\n    img = PIL.Image.open(path)\n    width, height = img.size\n    train_thumbnail_sizes.append({\"image_id\": id, \"image_width\": width, \"image_height\": height})\n\ndel img\n\ntrain_thumbnail_sizes = pd.DataFrame(train_thumbnail_sizes)\ndisplay(train_thumbnail_sizes.head())","metadata":{"execution":{"iopub.status.busy":"2023-11-19T17:47:18.175584Z","iopub.execute_input":"2023-11-19T17:47:18.176023Z","iopub.status.idle":"2023-11-19T17:47:26.297428Z","shell.execute_reply.started":"2023-11-19T17:47:18.175983Z","shell.execute_reply":"2023-11-19T17:47:26.296015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sizes = train_thumbnail_sizes[[\"image_width\", \"image_height\"]].copy(deep=True)\nfor col in (\"image_width\", \"image_height\"):\n    df_sizes[col] = [round(i / 10) * 10 for i in df_sizes[col]]\n\ndf_sizes = df_sizes.groupby([\"image_width\", \"image_height\"], as_index=False).size()\nfig = px.scatter(\n    df_sizes,\n    x=\"image_width\",\n    y=\"image_height\",\n    size=\"size\",\n    height=450,\n    width=900,\n    title=\"WSI thumbnail sizes\")\nfig.update_layout(title_x=0.5)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-19T17:47:26.299034Z","iopub.execute_input":"2023-11-19T17:47:26.299418Z","iopub.status.idle":"2023-11-19T17:47:26.383853Z","shell.execute_reply.started":"2023-11-19T17:47:26.299386Z","shell.execute_reply":"2023-11-19T17:47:26.382602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The thumbnails all seem to have width = 3000 while the heights vary from around 530 to 7840. Presumably all the WSI images were resized to have width 3000 while preserving the aspect ratio.\n\nThe size of these images explains why there are no thumbnails for TMA images. The TMA images are already about the same size as the WSI images, so there is no point making thumbnails of them.","metadata":{}},{"cell_type":"markdown","source":"## Viewing images","metadata":{}},{"cell_type":"code","source":"# Show WSI thumbnails\n\nfor lb, dfg in train_df[train_df[\"is_tma\"] == False].groupby(\"label\"):\n    fig, axes = plt.subplots(ncols=4, figsize=(16, 4))\n    fig.suptitle(f\"WSI thumbnails for label {lb}\")\n    for i, id in enumerate(dfg[\"image_id\"].sample(4)):\n        img_path = os.path.join(DATASET_FOLDER, \"train_thumbnails\", f\"{id}_thumbnail.png\")\n        axes[i].imshow(plt.imread(img_path))\n        axes[i].set_title(f\"Thumbnail {id}\")\n        axes[i].set_axis_off()\n    fig.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-19T17:47:26.38779Z","iopub.execute_input":"2023-11-19T17:47:26.388172Z","iopub.status.idle":"2023-11-19T17:48:06.026724Z","shell.execute_reply.started":"2023-11-19T17:47:26.38814Z","shell.execute_reply":"2023-11-19T17:48:06.025454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Show TMA images\n\nfor lb, dfg in train_df[train_df[\"is_tma\"] == True].groupby(\"label\"):\n    fig, axes = plt.subplots(ncols=4, figsize=(16, 4))\n    fig.suptitle(f\"TMA images for label {lb}\")\n    for i, id in enumerate(dfg[\"image_id\"].sample(4)):\n        img_path = os.path.join(DATASET_FOLDER, \"train_images\", f\"{id}.png\")\n        axes[i].imshow(plt.imread(img_path))\n        axes[i].set_title(f\"Image {id}\")\n        axes[i].set_axis_off()\n    fig.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-19T17:48:06.028316Z","iopub.execute_input":"2023-11-19T17:48:06.028657Z","iopub.status.idle":"2023-11-19T17:49:23.164549Z","shell.execute_reply.started":"2023-11-19T17:48:06.028628Z","shell.execute_reply":"2023-11-19T17:49:23.163242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In both the WSI and TMA images, there appear to be texture differences between the different cancer subtypes (labels). There are also larger-scale features like blobs, lines, curves and patches.\n\nIn the WSI images, there is a lot of variety in overall shape and it is unclear if this is meaningful for label prediction. Presumably not, because this depends on what tissue sample was taken, and this is done before the cancer subtype is known.\n\nThe key differences between the WSI and TMA images are:\n* The WSI images have a black background, but the TMA images have an off-white background. In the WSI images there is a large difference between the background and the tissue sample, but in the TMA images it is a small difference, so a model trained on primarily WSI images may be too sensitive to small changes in the background.\n* The WSI images seem perfectly cropped, whereas the TMA images are all under- or over-cropped to some extent.\n* The WSI images come in many shapes and sizes, and may even consist of multiple non-contiguous regions. In contrast, the TMA images are all one contiguous, roughly circular region.","metadata":{}},{"cell_type":"markdown","source":"Given that there are not many WSI images, but they are all quite large, we may consider taking random cropped sections of them to create a larger dataset of smaller images.\n\nThe TMA images show a lot of variation between them, and there are only 5 TMA images per class. This suggests that when doing a train-val split, putting even a single one of these into training as opposed to validation may significantly affect how good an estimate the validation set is of the data generating distribution of interest.\n\nIf we replace the TMA images by random crops, this addresses the small dataset size, but putting some of those crops into the training set might constitute data leakage if different parts of the same image are correlated with each other. In that case, the model could end up overfitting to the validation set.","metadata":{}},{"cell_type":"markdown","source":"## WSI vs TMA images\n\nHow similar are the TMA images to the WSI images, assuming we resize them to the same level of magnification?\n\nAccording to the [dataset description](https://www.kaggle.com/competitions/UBC-OCEAN/data), the WSI images are at 20x magnification and the TMA ones are at 40x. So, to faithfully compare the WSI and TMA images, we should downsample the TMAs by a factor of 2 in both dimensions.\n\nIn fact, we will compare the TMA images with the WSI thumbnails, because the TMA images are so large that my Kaggle notebook ran out of memory trying to open one of them 😅\n\nWe will need to crop the WSI thumbnails too, because they correspond to a much larger region of cell tissue than the TMA images.","metadata":{}},{"cell_type":"code","source":"WSI_MAGNIFICATION = 20.0\nTMA_MAGNIFICATION = 40.0\n\nNUM_IMGS_PER_CLASS_AND_TYPE = 2\n\ndef is_mostly_black(img):\n    color = PIL.ImageStat.Stat(img).mean\n    return np.linalg.norm(color) < 100\n\ndef random_crop(img, new_size):\n    width, height = img.size\n    new_width, new_height = new_size\n    new_width = min(width, new_width)\n    new_height = min(height, new_height)\n    crop_ok = False\n    while not crop_ok:\n        left = random.randrange(width - new_width + 1)\n        top = random.randrange(height - new_height + 1)\n        box = (left, top, left + new_width - 1, top + new_height - 1)\n        cropped = img.crop(box)\n        crop_ok = not is_mostly_black(cropped)\n        if not crop_ok:\n            print(\"Redoing crop\")\n    return cropped\n\nfor lb, dfg in train_df.groupby(\"label\"):\n    fig, axes = plt.subplots(ncols=2 * NUM_IMGS_PER_CLASS_AND_TYPE, figsize=(8 * NUM_IMGS_PER_CLASS_AND_TYPE, 4), sharex=True, sharey=True)\n    fig.suptitle(f\"WSI thumbnails and TMA images for label {lb}\")\n    \n    imgs = []\n    \n    # Sample WSI thumbnails and record magnifications.\n    #\n    # There is some discrepancy with aspect ratios that I don't fully understand, so I will skip\n    # samples where one of the aspect ratios doesn't match.\n    wsi_dfg = dfg[dfg[\"is_tma\"] == False]\n    wsi_sample = wsi_dfg.sample(NUM_IMGS_PER_CLASS_AND_TYPE)\n    wsi_sample_ok = False\n    while not wsi_sample_ok:\n        wsi_sample_ok = True\n        for _, wsi_row in wsi_sample.reset_index().iterrows():\n            wsi_id = wsi_row[\"image_id\"]\n            wsi_width = wsi_row[\"image_width\"]\n            wsi_height = wsi_row[\"image_height\"]\n            wsi_ratio = wsi_width / wsi_height\n            thumbnail_path = os.path.join(DATASET_FOLDER, \"train_thumbnails\", f\"{wsi_id}_thumbnail.png\")\n            thumbnail_img = PIL.Image.open(thumbnail_path)\n            thumbnail_width, thumbnail_height = thumbnail_img.size\n            thumbnail_ratio = thumbnail_width / thumbnail_height\n            if abs(thumbnail_ratio - wsi_ratio) > 1e-2:\n                print(f\"Skipping WSI image {wsi_id}: thumbnail aspect ratio {thumbnail_ratio:.3f} doesn't match train.csv aspect ratio {wsi_ratio:.3f}.\")\n                wsi_sample = wsi_dfg.sample(NUM_IMGS_PER_CLASS_AND_TYPE)\n                wsi_sample_ok = False\n                break\n    \n    for i, wsi_row in wsi_sample.reset_index().iterrows():\n        wsi_id = wsi_row[\"image_id\"]\n        wsi_width = wsi_row[\"image_width\"]\n        thumbnail_path = os.path.join(DATASET_FOLDER, \"train_thumbnails\", f\"{wsi_id}_thumbnail.png\")\n        thumbnail_img = PIL.Image.open(thumbnail_path)\n        thumbnail_width, _ = thumbnail_img.size\n        # Here we can assume that the two aspect ratios are the same\n        thumbnail_magnification = WSI_MAGNIFICATION * (thumbnail_width / wsi_width)\n        axes[i].set_title(f\"WSI thumbnail {wsi_id}\")\n        axes[i].set_axis_off()\n        imgs.append((thumbnail_img, thumbnail_magnification))\n    \n    # Sample TMA images and record magnifications.\n    tma_dfg = dfg[dfg[\"is_tma\"] == True]\n    for i, tma_row in tma_dfg.sample(NUM_IMGS_PER_CLASS_AND_TYPE).reset_index().iterrows():\n        tma_id = tma_row[\"image_id\"]\n        tma_path = os.path.join(DATASET_FOLDER, \"train_images\", f\"{tma_id}.png\")\n        tma_img = PIL.Image.open(tma_path)\n        tma_width, tma_height = tma_img.size\n        axes[i + NUM_IMGS_PER_CLASS_AND_TYPE].set_title(f\"TMA image {tma_id}\")\n        axes[i + NUM_IMGS_PER_CLASS_AND_TYPE].set_axis_off()\n        imgs.append((tma_img, TMA_MAGNIFICATION))\n    \n    # Scale all images to same magnification\n    min_mag = min(mag for _, mag in imgs)\n    for i, (img, mag) in enumerate(imgs):\n        img_width, img_height = img.size\n        scale_factor = min_mag / mag\n        img = img.resize((int(img_width * scale_factor), int(img_height * scale_factor)))\n        imgs[i] = (img, min_mag)\n        \n    # Crrop all images to the same size so matplotlib doesn't crop the larger ones\n    min_width = min(img.size[0] for img, _ in imgs)\n    min_height = min(img.size[1] for img, _ in imgs)\n    for i, (img, mag) in enumerate(imgs):\n        img = random_crop(img, (min_width, min_height))\n        axes[i].imshow(img)\n        \n    fig.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-19T17:49:23.166648Z","iopub.execute_input":"2023-11-19T17:49:23.167031Z","iopub.status.idle":"2023-11-19T17:49:35.375386Z","shell.execute_reply.started":"2023-11-19T17:49:23.166999Z","shell.execute_reply":"2023-11-19T17:49:35.374232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Next steps:\n* How to load the WSI images? Just calling `PIL.Image.open` causes an out of memory error.\n* Are there any mismatched image sizes, or mismatched image-thumbnail pairs, as a result of images in `updated_image_ids.json` having been updated?\n  * Do the WSI thumbnails match the full-size WSI images?\n  * Why are there mismatches between the aspect ratios in `train.csv` and those in `train_thumbnails`?\n* Make image_id the index.\n* Improve WSA vs TMI comparison.","metadata":{}}]}