{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"},{"sourceId":7296715,"sourceType":"datasetVersion","datasetId":4232470}],"dockerImageVersionId":30626,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-12-28T11:37:12.56099Z","iopub.execute_input":"2023-12-28T11:37:12.561328Z","iopub.status.idle":"2023-12-28T11:37:13.099999Z","shell.execute_reply.started":"2023-12-28T11:37:12.561303Z","shell.execute_reply":"2023-12-28T11:37:13.098563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# IMPORTS","metadata":{}},{"cell_type":"markdown","source":"Data Handling and Manipulation","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd","metadata":{"execution":{"iopub.status.busy":"2023-12-28T11:37:49.162574Z","iopub.execute_input":"2023-12-28T11:37:49.162962Z","iopub.status.idle":"2023-12-28T11:37:49.168177Z","shell.execute_reply.started":"2023-12-28T11:37:49.162934Z","shell.execute_reply":"2023-12-28T11:37:49.166689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Image Processing:","metadata":{}},{"cell_type":"code","source":"import os\nfrom PIL import Image\nimport cv2","metadata":{"execution":{"iopub.status.busy":"2023-12-28T11:37:53.141675Z","iopub.execute_input":"2023-12-28T11:37:53.142037Z","iopub.status.idle":"2023-12-28T11:37:53.436035Z","shell.execute_reply.started":"2023-12-28T11:37:53.142005Z","shell.execute_reply":"2023-12-28T11:37:53.434057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Machine Learning Libraries:","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nimport keras_cv\nimport keras_core as keras\nfrom keras_core import ops\nfrom tensorflow.keras import layers, models\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report, confusion_matrix","metadata":{"execution":{"iopub.status.busy":"2023-12-28T11:37:56.850844Z","iopub.execute_input":"2023-12-28T11:37:56.851251Z","iopub.status.idle":"2023-12-28T11:38:16.277027Z","shell.execute_reply.started":"2023-12-28T11:37:56.85122Z","shell.execute_reply":"2023-12-28T11:38:16.275713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Data Augmentation:","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.image import ImageDataGenerator","metadata":{"execution":{"iopub.status.busy":"2023-12-28T11:40:12.174516Z","iopub.execute_input":"2023-12-28T11:40:12.174932Z","iopub.status.idle":"2023-12-28T11:40:12.179874Z","shell.execute_reply.started":"2023-12-28T11:40:12.174899Z","shell.execute_reply":"2023-12-28T11:40:12.178391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Model Evaluation and Visualization:","metadata":{}},{"cell_type":"code","source":"import glob\nfrom matplotlib import pyplot as plt\nfrom matplotlib.image import imread\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Set the style for the plot\nsns.set(style=\"whitegrid\")","metadata":{"execution":{"iopub.status.busy":"2023-12-28T11:40:19.056838Z","iopub.execute_input":"2023-12-28T11:40:19.057242Z","iopub.status.idle":"2023-12-28T11:40:19.065441Z","shell.execute_reply.started":"2023-12-28T11:40:19.057201Z","shell.execute_reply":"2023-12-28T11:40:19.063889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Utilities for Model Saving and Loading:","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.models import load_model","metadata":{"execution":{"iopub.status.busy":"2023-12-28T11:38:34.164782Z","iopub.execute_input":"2023-12-28T11:38:34.165201Z","iopub.status.idle":"2023-12-28T11:38:34.17188Z","shell.execute_reply.started":"2023-12-28T11:38:34.16515Z","shell.execute_reply":"2023-12-28T11:38:34.170123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# LOAD","metadata":{}},{"cell_type":"code","source":" # Inference\ntest_csv_path = \"/kaggle/input/UBC-OCEAN/test.csv\"\ntest_thumbnail_paths = \"/kaggle/input/UBC-OCEAN/test_thumbnails\"\n\n# SEED\nkeras.utils.set_random_seed(seed=42)","metadata":{"execution":{"iopub.status.busy":"2023-12-28T11:38:43.090174Z","iopub.execute_input":"2023-12-28T11:38:43.090615Z","iopub.status.idle":"2023-12-28T11:38:43.097212Z","shell.execute_reply.started":"2023-12-28T11:38:43.090583Z","shell.execute_reply":"2023-12-28T11:38:43.095334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train a simple image classification model that will classify the scans of biopsy samples to their respective subtypes","metadata":{}},{"cell_type":"markdown","source":"/kaggle/working/The dataset comprises images related to ovarian carcinoma, categorized into two main types: whole slide images (WSI) and tissue microarray (TMA). WSI images are captured at a 20x magnification, potentially yielding large file sizes. TMAs, on the other hand, are smaller in dimensions (approximately 4,000x4,000 pixels) but at a higher 40x magnification.\n\n is_tma: A binary value indicating whether the slide is a tissue microarray. This information is only available for the train set.\n\nthe dataset includes a folder named [train/test]_thumbnails containing smaller .png versions of the whole slide images. Thumbnails, however, are not provided for TMAs.","metadata":{}},{"cell_type":"code","source":"df_test = pd.read_csv(test_csv_path)\ndf_test = df_test.drop(columns=['image_height','image_width'])","metadata":{"execution":{"iopub.status.busy":"2023-12-28T11:39:36.995105Z","iopub.execute_input":"2023-12-28T11:39:36.995528Z","iopub.status.idle":"2023-12-28T11:39:37.005251Z","shell.execute_reply.started":"2023-12-28T11:39:36.995498Z","shell.execute_reply":"2023-12-28T11:39:37.003756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-28T11:39:40.004769Z","iopub.execute_input":"2023-12-28T11:39:40.005153Z","iopub.status.idle":"2023-12-28T11:39:40.020434Z","shell.execute_reply.started":"2023-12-28T11:39:40.005119Z","shell.execute_reply":"2023-12-28T11:39:40.018922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from skimage.io import imread\nfrom skimage.transform import resize\nfrom tqdm import tqdm\n# Initialize X_train as an empty list to store the loaded images\nX_test_final = []\ny_test_final = []\n\n# Loop through each row in the DataFrame\nfor index, row in tqdm(df_test.iterrows(), desc=\"Loading Test Images\"):\n    # Get the image_id and label for each row\n    image_id = row['image_id']\n\n    # Construct the full file path\n    image_path = os.path.join(test_thumbnail_paths, f\"{image_id}_thumbnail.png\")\n\n    # Read and resize the image\n    img = imread(image_path)\n    target_size = (256, 256)\n    img = resize(img, target_size, anti_aliasing=True)\n\n    # Append the image to X_train and the corresponding label to y_train\n    X_test_final.append(img)\n\n# Convert the lists to numpy arrays\nX_test_final = np.array(X_test_final)\n\n# Normalize pixel values to the range [0, 1]\nX_test_final = X_test_final / 255.0\n\n# Print the shapes of X_train and y_train to verify\nprint(\"Shape of X_test:\", X_test_final.shape)\n","metadata":{"execution":{"iopub.status.busy":"2023-12-28T12:10:01.580091Z","iopub.execute_input":"2023-12-28T12:10:01.580557Z","iopub.status.idle":"2023-12-28T12:10:03.114528Z","shell.execute_reply.started":"2023-12-28T12:10:01.580525Z","shell.execute_reply":"2023-12-28T12:10:03.112877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.models import Sequential, load_model \nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, BatchNormalization, Dropout\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import EarlyStopping, ModelCheckpoint\nfrom tensorflow.keras.utils import to_categorical\n","metadata":{"execution":{"iopub.status.busy":"2023-12-28T12:10:41.159439Z","iopub.execute_input":"2023-12-28T12:10:41.159873Z","iopub.status.idle":"2023-12-28T12:10:41.167804Z","shell.execute_reply.started":"2023-12-28T12:10:41.159838Z","shell.execute_reply":"2023-12-28T12:10:41.165914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = load_model('/kaggle/input/models/model_UBC_OCSCO.h5')","metadata":{"execution":{"iopub.status.busy":"2023-12-28T12:11:59.962698Z","iopub.execute_input":"2023-12-28T12:11:59.963078Z","iopub.status.idle":"2023-12-28T12:12:04.306066Z","shell.execute_reply.started":"2023-12-28T12:11:59.963049Z","shell.execute_reply":"2023-12-28T12:12:04.304379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = model.predict(X_test_final)","metadata":{"execution":{"iopub.status.busy":"2023-12-28T12:14:21.450286Z","iopub.execute_input":"2023-12-28T12:14:21.451071Z","iopub.status.idle":"2023-12-28T12:14:21.800231Z","shell.execute_reply.started":"2023-12-28T12:14:21.451038Z","shell.execute_reply":"2023-12-28T12:14:21.798462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_mapping = {\n    \"HGSC\": 0 ,\n    \"EC\": 1,\n    \"CC\": 2,\n    \"LGSC\": 3,\n    \"MC\": 4  # Add more labels as needed\n}","metadata":{"execution":{"iopub.status.busy":"2023-12-28T12:15:35.789475Z","iopub.execute_input":"2023-12-28T12:15:35.790043Z","iopub.status.idle":"2023-12-28T12:15:35.795138Z","shell.execute_reply.started":"2023-12-28T12:15:35.790012Z","shell.execute_reply":"2023-12-28T12:15:35.793836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions_final = model.predict(X_test_final)\npredictions_final","metadata":{"execution":{"iopub.status.busy":"2023-12-28T12:16:01.838061Z","iopub.execute_input":"2023-12-28T12:16:01.839541Z","iopub.status.idle":"2023-12-28T12:16:01.938094Z","shell.execute_reply.started":"2023-12-28T12:16:01.839506Z","shell.execute_reply":"2023-12-28T12:16:01.936285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"binary_predictions_final = np.zeros_like(predictions_final)\nbinary_predictions_final[np.arange(len(predictions_final)), predictions_final.argmax(axis=1)] = 1\nprint(binary_predictions_final)","metadata":{"execution":{"iopub.status.busy":"2023-12-28T12:16:11.437326Z","iopub.execute_input":"2023-12-28T12:16:11.437729Z","iopub.status.idle":"2023-12-28T12:16:11.445041Z","shell.execute_reply.started":"2023-12-28T12:16:11.437699Z","shell.execute_reply":"2023-12-28T12:16:11.443818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predicted_labels_indices_final = np.argmax(binary_predictions_final, axis=1)\n\n# Map numerical values to labels using the label_mapping dictionary\npredicted_labels_final = [list(label_mapping.keys())[list(label_mapping.values()).index(idx)] for idx in predicted_labels_indices_final]\n\nprint(predicted_labels_final)","metadata":{"execution":{"iopub.status.busy":"2023-12-28T12:16:20.657019Z","iopub.execute_input":"2023-12-28T12:16:20.657436Z","iopub.status.idle":"2023-12-28T12:16:20.664062Z","shell.execute_reply.started":"2023-12-28T12:16:20.657406Z","shell.execute_reply":"2023-12-28T12:16:20.662716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Add the labels to the DataFrame\ndf_test['label'] = predicted_labels_final\n\n# Display the resulting DataFrame\nprint(df_test)","metadata":{"execution":{"iopub.status.busy":"2023-12-28T12:16:33.417737Z","iopub.execute_input":"2023-12-28T12:16:33.418115Z","iopub.status.idle":"2023-12-28T12:16:33.42564Z","shell.execute_reply.started":"2023-12-28T12:16:33.418087Z","shell.execute_reply":"2023-12-28T12:16:33.424773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = df_test\nsubmission.to_csv('/kaggle/working/submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-12-28T12:16:37.647134Z","iopub.execute_input":"2023-12-28T12:16:37.64753Z","iopub.status.idle":"2023-12-28T12:16:37.656948Z","shell.execute_reply.started":"2023-12-28T12:16:37.647502Z","shell.execute_reply":"2023-12-28T12:16:37.655539Z"},"trusted":true},"execution_count":null,"outputs":[]}]}