{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"},{"sourceId":148683476,"sourceType":"kernelVersion"}],"dockerImageVersionId":30558,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## UBC Ovarian Cancer Subtype Classification and Outlier Detection (UBC-OCEAN)","metadata":{}},{"cell_type":"markdown","source":"## 1. Setup","metadata":{}},{"cell_type":"code","source":"import os\nfrom glob import glob\nfrom pathlib import Path\nfrom tqdm.notebook import tqdm\nimport numpy as np\nimport pandas as pd\nimport cv2\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2025-01-05T08:37:59.864677Z","iopub.execute_input":"2025-01-05T08:37:59.865206Z","iopub.status.idle":"2025-01-05T08:37:59.871067Z","shell.execute_reply.started":"2025-01-05T08:37:59.865167Z","shell.execute_reply":"2025-01-05T08:37:59.869873Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"competition_dataset_directory = Path('/kaggle/input/UBC-OCEAN')\njpeg_dataset_directory = Path('/kaggle/input/ubc-ocean-jpeg-dataset-pipeline')","metadata":{"execution":{"iopub.status.busy":"2025-01-05T08:37:59.873004Z","iopub.execute_input":"2025-01-05T08:37:59.873429Z","iopub.status.idle":"2025-01-05T08:37:59.885763Z","shell.execute_reply.started":"2025-01-05T08:37:59.8734Z","shell.execute_reply":"2025-01-05T08:37:59.884541Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def visualize_categorical_column_distribution(df, column, title, path=None):\n\n    \"\"\"\n    Visualize distribution of the given categorical column on the given dataframe\n\n    Parameters\n    ----------\n    df: pandas.DataFrame\n        Dataframe with given categorical column\n\n    column: str\n        Name of the categorical column\n\n    title: str\n        Title of the plot\n\n    path: path-like str or None\n        Path of the output file or None (if path is None, plot is displayed with selected backend)\n    \"\"\"\n\n    value_counts = df[column].value_counts()\n    n = value_counts.sum()\n\n    fig, ax = plt.subplots(figsize=(24, df[column].value_counts().shape[0] + 4), dpi=100)\n    ax.bar(\n        x=np.arange(len(value_counts)),\n        height=value_counts.values,\n    )\n    ax.set_xlabel('')\n    ax.set_ylabel('')\n    ax.set_xticks(\n        np.arange(len(value_counts)),\n        [\n            f'{value}\\n{count:,} ({(count / n * 100):.2f}%)' for value, count in value_counts.to_dict().items()\n        ]\n    )\n    ax.tick_params(axis='x', labelsize=15, pad=10)\n    ax.tick_params(axis='y', labelsize=15, pad=10)\n    ax.set_title(title, size=20, pad=15)\n\n    if path is None:\n        plt.show()\n    else:\n        plt.savefig(path)\n        plt.close(fig)\n\n\ndef visualize_continuous_column_distribution(df, column, title, path=None):\n\n    \"\"\"\n    Visualize distribution of the given continuous column on the given dataframe\n\n    Parameters\n    ----------\n    df: pandas.DataFrame\n        Dataframe with given continuous column\n\n    column: str\n        Name of the continuous column,\n\n    title: str\n        Title of the plot\n\n    path: path-like str or None\n        Path of the output file or None (if path is None, plot is displayed with selected backend)\n    \"\"\"\n\n    fig, ax = plt.subplots(figsize=(24, 6), dpi=100)\n    ax.hist(df[column], bins=16)\n    ax.tick_params(axis='x', labelsize=15)\n    ax.tick_params(axis='y', labelsize=15)\n    ax.set_xlabel('')\n    ax.set_ylabel('')\n    ax.set_title(\n        title + f'''\n        {column}\n        Mean: {np.mean(df[column]):.2f} Std: {np.std(df[column]):.2f}\n        Min: {np.min(df[column]):.2f} Max: {np.max(df[column]):.2f}\n        ''',\n        size=20,\n        pad=12.5,\n        loc='center',\n        wrap=True\n    )\n\n    if path is None:\n        plt.show()\n    else:\n        plt.savefig(path, bbox_inches='tight')\n        plt.close(fig)\n        \n        \ndef visualize_continuous_columns_relationship(df, column1, column2, group, title, path=None):\n\n    \"\"\"\n    Visualize relationship of two given continuous columns on the given dataframe\n\n    Parameters\n    ----------\n    df: pandas.DataFrame\n        Dataframe with given continuous columns\n\n    column1: str\n        Name of the first continuous column\n        \n    column2: str\n        Name of the second continuous column\n        \n    group: str or None\n        Name of the group column\n\n    title: str\n        Title of the plot\n\n    path: path-like str or None\n        Path of the output file or None (if path is None, plot is displayed with selected backend)\n    \"\"\"\n\n    fig, ax = plt.subplots(figsize=(16, 8), dpi=100)\n    sns.scatterplot(x=column1, y=column2, hue=group, data=df, ax=ax)\n    ax.tick_params(axis='x', labelsize=15)\n    ax.tick_params(axis='y', labelsize=15)\n    ax.set_xlabel(column1, size=20, labelpad=15)\n    ax.set_ylabel(column2, size=20, labelpad=15)\n    if group is not None:\n        legend = ax.legend(title=group, loc='lower right', prop={'size': 14})\n        legend.get_title().set_fontsize(15)\n    ax.set_title(title, size=20, pad=15)\n\n    if path is None:\n        plt.show()\n    else:\n        plt.savefig(path, bbox_inches='tight')\n        plt.close(fig)\n\n        \ndef visualize_image(image, title, path=None):\n\n    \"\"\"\n    Visualize the given image\n\n    Parameters\n    ----------\n    image: numpy.ndarray of shape (height, width, channel)\n        Image array\n        \n    title: str\n        Title of the plot\n\n    path: str or None\n        Path of the output file or None (if path is None, plot is displayed with selected backend)\n    \"\"\"\n\n    fig, ax = plt.subplots(figsize=(8, 8))\n    ax.imshow(image)\n    ax.set_xlabel('')\n    ax.set_ylabel('')\n    ax.tick_params(axis='x', labelsize=15, pad=10)\n    ax.tick_params(axis='y', labelsize=15, pad=10)\n    ax.set_title(title, size=15, pad=12.5, loc='center', wrap=True)\n\n    if path is None:\n        plt.show()\n    else:\n        plt.savefig(path)\n        plt.close(fig)\n","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2025-01-05T08:37:59.888069Z","iopub.execute_input":"2025-01-05T08:37:59.888774Z","iopub.status.idle":"2025-01-05T08:37:59.907906Z","shell.execute_reply.started":"2025-01-05T08:37:59.888725Z","shell.execute_reply":"2025-01-05T08:37:59.906797Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2. Introduction\n\nOvarian cancer is a type of cancer that begins in the ovaries, which are part of the female reproductive system. The ovaries are responsible for producing eggs and female sex hormones. Ovarian cancer can develop in different parts of the ovaries and can take on various forms, which makes it a complex disease to diagnose and treat. There are several types of ovarian cancer and the challenge in this competition is to classify those types using high resolution microscopy images of biopsy samples.","metadata":{}},{"cell_type":"markdown","source":"There are  **538** samples on training set and the fields are `image_id`, `label`, `image_width`, `image_height` and `is_tma`.","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv(competition_dataset_directory / 'train.csv')\ndf_train","metadata":{"execution":{"iopub.status.busy":"2025-01-05T08:37:59.909375Z","iopub.execute_input":"2025-01-05T08:37:59.90976Z","iopub.status.idle":"2025-01-05T08:37:59.97961Z","shell.execute_reply.started":"2025-01-05T08:37:59.909723Z","shell.execute_reply":"2025-01-05T08:37:59.978199Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3. Target\n\nTarget (`label`) is the ovarian cancer subtype. Subtypes on this dataset are:\n\n**HGSC**: High-grade serous carcinoma (HGSC) is the most common type of ovarian cancer, comprising around 75% of all cases within the category of epithelial ovarian cancers. Due to its prevalence, HGSC frequently serves as the standard reference when discussing ovarian cancer, unless there is a specific focus on a different subtype.\n\n**EC**: Endometrioid carcinoma (EC) is another subtype of epithelial ovarian cancer with the majority of cases being malignant and invasive in nature. When observed through medical imaging, these tumors typically present as complex, solid-cystic masses without specific distinguishing features and are often linked to the presence of endometriosis.\n\n**CC**: Clear cell carcinoma (CC) is an uncommon variant of epithelial ovarian cancer, representing approximately 5–11% of all cases in this category. Ovarian clear cell carcinoma exhibits distinctive clinical, histopathological, and molecular characteristics that set it apart from other subtypes of epithelial ovarian cancer.\n\n**LGSC**: Low-grade serous carcinoma (LGSC) is an infrequent subtype, making up just 2% of all cases of epithelial ovarian cancer and around 5% of serous carcinomas. Typically, women diagnosed with LGSC are comparatively younger, with an average age of 56, in contrast to those with high-grade serous carcinoma (HGSC), who have an average age of 63 at diagnosis.\n\n**MC**: Mucinous cancer (MC) is an uncommon form of ovarian cancer, presenting a variety of symptoms such as pelvic and abdominal discomfort, bloating, and fatigue. Generally, surgical intervention is necessary to address the tumor, potentially involving the removal of the ovaries, cervix, and uterus.\n\n**Other**: There is also another class named Other in the hidden test set that is not present in the training set. Other class consist of outlier subtypes that doesn't fit into classes listed above.","metadata":{}},{"cell_type":"code","source":"visualize_categorical_column_distribution(\n    df=df_train,\n    column='label',\n    title='Training Set Target Frequencies'\n)","metadata":{"execution":{"iopub.status.busy":"2025-01-05T08:37:59.982423Z","iopub.execute_input":"2025-01-05T08:37:59.982779Z","iopub.status.idle":"2025-01-05T08:38:00.333069Z","shell.execute_reply.started":"2025-01-05T08:37:59.98275Z","shell.execute_reply":"2025-01-05T08:38:00.331987Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 4. Image Types\n\nThe dataset comprises two image types: whole slide images (**WSI**) and tissue microarray (**TMA**). Whole slide images are at 20x magnification and and are relatively large in size. Tissue microarrays are smaller, measuring approximately 4000x4000 pixels but at 40x magnification.\n\nThere are **538** samples and images in total so there is only one image per sample in the dataset, however thumbnails of WSIs are also provided. Thumbnails are smaller resized versions of WSIs. Thumbnails are not given for TMAs since they are already relatively small compared to WSIs.","metadata":{}},{"cell_type":"code","source":"training_set_sample_count = df_train.shape[0]\ntraining_set_image_count = len(os.listdir(competition_dataset_directory / 'train_images'))\ntraining_set_thumbnail_count = len(os.listdir(competition_dataset_directory / 'train_thumbnails'))\n\nprint(f'Training Set Sample Count: {training_set_sample_count}')\nprint(f'Training Set Image Count: {training_set_image_count}')\nprint(f'Training Set Thumbnail Count: {training_set_thumbnail_count}')","metadata":{"execution":{"iopub.status.busy":"2025-01-05T08:38:00.334413Z","iopub.execute_input":"2025-01-05T08:38:00.334727Z","iopub.status.idle":"2025-01-05T08:38:00.359114Z","shell.execute_reply.started":"2025-01-05T08:38:00.3347Z","shell.execute_reply":"2025-01-05T08:38:00.357897Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"\n\nIt can be seen that **25** (4.65%) images are TMAs and **513** (95.35%) images are WSIs. The hidden test set contains roughly 2,000 images and the majority of which are TMAs.","metadata":{}},{"cell_type":"code","source":"visualize_categorical_column_distribution(\n    df=df_train,\n    column='is_tma',\n    title='Training Set Image Type Frequencies (is_tma)'\n)","metadata":{"execution":{"iopub.status.busy":"2025-01-05T08:38:00.360606Z","iopub.execute_input":"2025-01-05T08:38:00.361016Z","iopub.status.idle":"2025-01-05T08:38:00.633904Z","shell.execute_reply.started":"2025-01-05T08:38:00.360975Z","shell.execute_reply":"2025-01-05T08:38:00.632841Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Dimensions of thumbnails are not given in the training so they are extracted with the code below.","metadata":{}},{"cell_type":"code","source":"for idx, row in tqdm(df_train.loc[~df_train['is_tma']].iterrows(), total=df_train.loc[~df_train['is_tma']].shape[0]):\n    \n    image_path = str(competition_dataset_directory / 'train_thumbnails' / f'{row[\"image_id\"]}_thumbnail.png')\n    image = cv2.imread(image_path)\n    \n    df_train.loc[idx, 'thumbnail_height'] = image.shape[0]\n    df_train.loc[idx, 'thumbnail_width'] = image.shape[1]\n","metadata":{"execution":{"iopub.status.busy":"2025-01-05T08:38:00.63549Z","iopub.execute_input":"2025-01-05T08:38:00.635911Z","iopub.status.idle":"2025-01-05T08:39:49.58182Z","shell.execute_reply.started":"2025-01-05T08:38:00.635869Z","shell.execute_reply":"2025-01-05T08:39:49.580801Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Image dimensions (width and height) don't have a strong relationship between each other so some of the images can be highly asymmetrical.\n\nDimensions of TMAs can be seen from the visualization below. Two orange dots at the bottom left corner are them. ","metadata":{}},{"cell_type":"code","source":"visualize_continuous_columns_relationship(\n    df=df_train,\n    column1='image_width',\n    column2='image_height',\n    group='is_tma',\n    title='Image Height vs Width (WSIs and TMAs)'\n)","metadata":{"execution":{"iopub.status.busy":"2025-01-05T08:39:49.583253Z","iopub.execute_input":"2025-01-05T08:39:49.583551Z","iopub.status.idle":"2025-01-05T08:39:49.994765Z","shell.execute_reply.started":"2025-01-05T08:39:49.583524Z","shell.execute_reply":"2025-01-05T08:39:49.993291Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"There are 25 TMAs but there are only 2 unique height and width values in training set. All of the TMAs in training set are either 3388x3388 or 2964x2964 squares. Those width and height values may not be their raw sizes so they may be resized.","metadata":{}},{"cell_type":"code","source":"visualize_continuous_columns_relationship(\n    df=df_train.loc[df_train['is_tma']],\n    column1='image_width',\n    column2='image_height',\n    group=None,\n    title='Image Height vs Width (TMAs)'\n)","metadata":{"execution":{"iopub.status.busy":"2025-01-05T08:39:49.996417Z","iopub.execute_input":"2025-01-05T08:39:49.996815Z","iopub.status.idle":"2025-01-05T08:39:50.290845Z","shell.execute_reply.started":"2025-01-05T08:39:49.996778Z","shell.execute_reply":"2025-01-05T08:39:50.290004Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Thumbnails are very likely created by resizing width of WSIs to 3000 and some of them are highly asymmetrical due to that reason.","metadata":{}},{"cell_type":"code","source":"visualize_continuous_columns_relationship(\n    df=df_train,\n    column1='thumbnail_width',\n    column2='thumbnail_height',\n    group=None,\n    title='Thumbnail Height vs Width (WSIs)'\n)","metadata":{"execution":{"iopub.status.busy":"2025-01-05T08:39:50.294243Z","iopub.execute_input":"2025-01-05T08:39:50.29504Z","iopub.status.idle":"2025-01-05T08:39:50.626614Z","shell.execute_reply.started":"2025-01-05T08:39:50.294994Z","shell.execute_reply":"2025-01-05T08:39:50.625491Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 5. Dataset Pipeline\n\nDataset pipeline is implemented and run on [this](https://www.kaggle.com/code/gunesevitan/ubc-ocean-jpeg-dataset-pipeline) notebook. Images are exported as JPEGs with 80% compression quality and their longest edges are resized to 20000 pixels. Images that don't have any edge longer than 20000 pixels are not resized. It takes more than 10 hours to write 538 images as JPEGs with the current configurations. 12 hour limit on notebooks may not be enough to export images in higher settings so it should be done in multiple runs. However, 80% compression quality and max 20000 pixels on longest edge are more than enough for visualization purposes.\n\n**This notebook uses the updated data with fixed masks.**","metadata":{}},{"cell_type":"markdown","source":"## 6. Whole Slide Images (WSIs)\n\nWhole slide images (WSIs), also known as digital pathology slides or virtual slides, are high-resolution, digital representations of entire tissue samples or pathology slides typically used in the field of pathology and medical diagnostics. These images are created by scanning glass microscope slides at very high magnification.\n\nWSIs in this dataset are at 20x magnification and they are quite large. Their longest edges are resized to 20000 pixels but their raw sizes are larger than that. All of the WSIs in this dataset are visualized below.\n\nThe main problems with WSIs in this dataset are:\n* Almost identical duplicate regions\n* Inaccurate masks\n* Markers","metadata":{}},{"cell_type":"code","source":"training_wsis = df_train.loc[~df_train['is_tma'], ['image_id', 'is_tma', 'label']]\n\nfor idx, row in training_wsis.iterrows():\n    \n    image_path = str(jpeg_dataset_directory / 'train_compressed_images' / f'{row[\"image_id\"]}.jpg')\n    image = cv2.imread(image_path)\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    \n    visualize_image(\n        image=image,\n        title=f'Image ID: {row[\"image_id\"]} Type: WSI Label: {row[\"label\"]}\\nHeight: {image.shape[0]} Width: {image.shape[1]}\\nMean: {np.mean(image):.2f} Std: {np.std(image):.2f}\\nMin: {np.min(image):.2f} Max: {np.max(image):.2f}'\n    )\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 7. Tissue Microarrays (TMAs)\n\nTissue microarrays (TMAs) are used in pathology for the efficient and high-throughput analysis of multiple tissue samples simultaneously. They contain a smaller single patch of tissue unlike WSIs.\n\nTMAs in this dataset are at 40x magnification and they are relatively smaller in size compared to WSIs so they didn't need resizing. All of the TMAs in this dataset are visualized below.\n\nTMAs in this dataset don't have any problems at first glance.","metadata":{}},{"cell_type":"code","source":"training_tmas = df_train.loc[df_train['is_tma'], ['image_id', 'is_tma', 'label']]\n\nfor idx, row in training_tmas.iterrows():\n    \n    image_path = str(jpeg_dataset_directory / 'train_compressed_images' / f'{row[\"image_id\"]}.jpg')\n    image = cv2.imread(image_path)\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    \n    visualize_image(\n        image=image,\n        title=f'Image ID: {row[\"image_id\"]} Type: TMA Label: {row[\"label\"]}\\nHeight: {image.shape[0]} Width: {image.shape[1]}\\nMean: {np.mean(image):.2f} Std: {np.std(image):.2f}\\nMin: {np.min(image):.2f} Max: {np.max(image):.2f}'\n    )\n","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}