{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"}],"dockerImageVersionId":30626,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-12-28T09:25:59.443355Z","iopub.execute_input":"2023-12-28T09:25:59.443724Z","iopub.status.idle":"2023-12-28T09:26:00.013917Z","shell.execute_reply.started":"2023-12-28T09:25:59.443695Z","shell.execute_reply":"2023-12-28T09:26:00.011363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# IMPORTS","metadata":{}},{"cell_type":"markdown","source":"Data Handling and Manipulation","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd","metadata":{"execution":{"iopub.status.busy":"2023-12-28T09:26:18.461417Z","iopub.execute_input":"2023-12-28T09:26:18.462045Z","iopub.status.idle":"2023-12-28T09:26:18.468409Z","shell.execute_reply.started":"2023-12-28T09:26:18.462008Z","shell.execute_reply":"2023-12-28T09:26:18.46636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Image Processing:","metadata":{}},{"cell_type":"code","source":"import os\nfrom PIL import Image\nimport cv2","metadata":{"execution":{"iopub.status.busy":"2023-12-28T09:26:22.950819Z","iopub.execute_input":"2023-12-28T09:26:22.95125Z","iopub.status.idle":"2023-12-28T09:26:23.168856Z","shell.execute_reply.started":"2023-12-28T09:26:22.951217Z","shell.execute_reply":"2023-12-28T09:26:23.16759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Machine Learning Libraries:","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nimport keras_cv\nimport keras_core as keras\nfrom keras_core import ops\nfrom tensorflow.keras import layers, models\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report, confusion_matrix","metadata":{"execution":{"iopub.status.busy":"2023-12-28T09:26:26.520478Z","iopub.execute_input":"2023-12-28T09:26:26.520894Z","iopub.status.idle":"2023-12-28T09:26:44.775971Z","shell.execute_reply.started":"2023-12-28T09:26:26.520863Z","shell.execute_reply":"2023-12-28T09:26:44.774794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Data Augmentation:","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.image import ImageDataGenerator","metadata":{"execution":{"iopub.status.busy":"2023-12-28T09:26:47.878584Z","iopub.execute_input":"2023-12-28T09:26:47.879017Z","iopub.status.idle":"2023-12-28T09:26:47.88493Z","shell.execute_reply.started":"2023-12-28T09:26:47.878986Z","shell.execute_reply":"2023-12-28T09:26:47.883572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Model Evaluation and Visualization:","metadata":{}},{"cell_type":"code","source":"import glob\nfrom matplotlib import pyplot as plt\nfrom matplotlib.image import imread\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Set the style for the plot\nsns.set(style=\"whitegrid\")","metadata":{"execution":{"iopub.status.busy":"2023-12-28T09:26:53.360469Z","iopub.execute_input":"2023-12-28T09:26:53.360885Z","iopub.status.idle":"2023-12-28T09:26:53.458957Z","shell.execute_reply.started":"2023-12-28T09:26:53.360852Z","shell.execute_reply":"2023-12-28T09:26:53.457815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Utilities for Model Saving and Loading:","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.models import load_model","metadata":{"execution":{"iopub.status.busy":"2023-12-28T09:26:58.249191Z","iopub.execute_input":"2023-12-28T09:26:58.249571Z","iopub.status.idle":"2023-12-28T09:26:58.255354Z","shell.execute_reply.started":"2023-12-28T09:26:58.249543Z","shell.execute_reply":"2023-12-28T09:26:58.254316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# LOAD","metadata":{}},{"cell_type":"code","source":"# Training\ntrain_csv_path = \"/kaggle/input/UBC-OCEAN/train.csv\"\ntrain_thumbnail_paths = \"/kaggle/input/UBC-OCEAN/train_thumbnails\"\n    \n # Inference\ntest_csv_path = \"/kaggle/input/UBC-OCEAN/test.csv\"\ntest_thumbnail_paths = \"/kaggle/input/UBC-OCEAN/test_thumbnails\"\n\n# SEED\nkeras.utils.set_random_seed(seed=42)","metadata":{"execution":{"iopub.status.busy":"2023-12-28T09:27:02.700096Z","iopub.execute_input":"2023-12-28T09:27:02.700485Z","iopub.status.idle":"2023-12-28T09:27:02.707065Z","shell.execute_reply.started":"2023-12-28T09:27:02.700456Z","shell.execute_reply":"2023-12-28T09:27:02.705747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train a simple image classification model that will classify the scans of biopsy samples to their respective subtypes","metadata":{}},{"cell_type":"markdown","source":"The dataset comprises images related to ovarian carcinoma, categorized into two main types: whole slide images (WSI) and tissue microarray (TMA). WSI images are captured at a 20x magnification, potentially yielding large file sizes. TMAs, on the other hand, are smaller in dimensions (approximately 4,000x4,000 pixels) but at a higher 40x magnification.\n\n is_tma: A binary value indicating whether the slide is a tissue microarray. This information is only available for the train set.\n\nthe dataset includes a folder named [train/test]_thumbnails containing smaller .png versions of the whole slide images. Thumbnails, however, are not provided for TMAs.","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(train_csv_path)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-28T09:27:27.330542Z","iopub.execute_input":"2023-12-28T09:27:27.331793Z","iopub.status.idle":"2023-12-28T09:27:27.364214Z","shell.execute_reply.started":"2023-12-28T09:27:27.331739Z","shell.execute_reply":"2023-12-28T09:27:27.363011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create the thumbnail df where is_tma == False(filtering)\ndf = df[df[\"is_tma\"] == False]\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-28T09:27:31.280014Z","iopub.execute_input":"2023-12-28T09:27:31.280532Z","iopub.status.idle":"2023-12-28T09:27:31.298289Z","shell.execute_reply.started":"2023-12-28T09:27:31.280494Z","shell.execute_reply":"2023-12-28T09:27:31.297138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the distribution of the target classes\nplt.figure(figsize=(10, 6))\nsns.countplot(data=df, x='label', order=df['label'].value_counts().index)\nplt.title('Distribution of Target Classes')\nplt.xlabel('Label')\nplt.ylabel('Count')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-28T09:27:34.450388Z","iopub.execute_input":"2023-12-28T09:27:34.450826Z","iopub.status.idle":"2023-12-28T09:27:34.765784Z","shell.execute_reply.started":"2023-12-28T09:27:34.450794Z","shell.execute_reply":"2023-12-28T09:27:34.764303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.nunique()","metadata":{"execution":{"iopub.status.busy":"2023-12-27T17:48:24.316406Z","iopub.execute_input":"2023-12-27T17:48:24.316906Z","iopub.status.idle":"2023-12-27T17:48:24.332383Z","shell.execute_reply.started":"2023-12-27T17:48:24.316864Z","shell.execute_reply":"2023-12-27T17:48:24.331153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from PIL import Image\nimport random\n# Define the paths to the image directories thumbnails\ntrain_data = glob.glob('/kaggle/input/UBC-OCEAN/train_thumbnails/*.png')\ntest_data = glob.glob('/kaggle/input/UBC-OCEAN/test_thumbnails/*.png')\n\n# Display a few sample images from the training set\nnum_samples = 5\n\n# Randomly select sample images from the training set\nsample_images = random.sample(train_data, num_samples)\n\n# Display the sample images\nplt.figure(figsize=(15, 8))\nfor i, image_path in enumerate(sample_images, 1):\n    image = Image.open(image_path)\n    plt.subplot(1, num_samples, i)\n    plt.imshow(image)\n    plt.title(f'Sample {i}')\n    plt.axis('off')\n\nplt.tight_layout()\nplt.savefig('samples.png')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-28T09:27:39.907919Z","iopub.execute_input":"2023-12-28T09:27:39.908329Z","iopub.status.idle":"2023-12-28T09:27:54.003196Z","shell.execute_reply.started":"2023-12-28T09:27:39.908257Z","shell.execute_reply":"2023-12-28T09:27:54.002363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nfrom skimage.io import imread\nfrom skimage.transform import resize\nfrom tqdm import tqdm\n\n# Initialize X_train as an empty list to store the loaded images\nX_train = []\ny_train = []\n\nX_val = []\ny_val = []\n\n# Define a mapping of labels to numerical values\nlabel_mapping = {\n    \"HGSC\": 0 ,\n    \"EC\": 1,\n    \"CC\": 2,\n    \"LGSC\": 3,\n    \"MC\": 4  # Add more labels as needed\n}\n\ndatagen = ImageDataGenerator(\n    rotation_range=20,\n    width_shift_range=0.2,\n    height_shift_range=0.2,\n    shear_range=0.2,\n    zoom_range=0.2,\n    horizontal_flip=True,\n    fill_mode='nearest'\n)\n\ni=0\n# Loop through each row in the DataFrame\nfor index, row in tqdm(df.iterrows(), desc=\"Loading Train Images\"):\n    # Get the image_id and label for each row\n    image_id = row['image_id']\n    label = row['label']\n\n    # Construct the full file path\n    image_path = os.path.join(train_thumbnail_paths, f\"{image_id}_thumbnail.png\")\n\n    # Read and resize the image\n    img = imread(image_path)\n    target_size = (256, 256)\n    img = resize(img, target_size, anti_aliasing=True)\n    numerical_label = label_mapping[label]\n    \n    if (numerical_label==1):\n        # Augment the image with label \"EC\" and append both original and augmented images\n        X_train.append(img)\n        y_train.append(numerical_label)\n        \n    if (numerical_label==2):\n        # Augment the image with label \"CC\" and append both original and augmented images\n        X_train.append(img)\n        y_train.append(numerical_label)\n    \n    if (numerical_label==3):\n        # Augment the image with label \"LGSC\" and append both original and augmented images\n        X_train.append(img)\n        y_train.append(numerical_label)\n        X_train.append(img)\n        y_train.append(numerical_label)\n        X_train.append(img)\n        y_train.append(numerical_label)\n        \n    if (numerical_label==4):\n        # Augment the image with label \"MC\" and append both original and augmented images\n        X_train.append(img)\n        y_train.append(numerical_label)\n        X_train.append(img)\n        y_train.append(numerical_label)\n        X_train.append(img)\n        y_train.append(numerical_label)\n        \n    # Append the original image to X_train and the corresponding label to y_train\n    X_train.append(img)\n    y_train.append(numerical_label)\n    \n\n# Convert the lists to numpy arrays\nX_train = np.array(X_train)\ny_train = np.array(y_train)\n\n# Normalize pixel values to the range [0, 1]\nX_train = X_train / 255.0\n\n# Print the shapes of X_train and y_train to verify\nprint(\"Shape of X_train:\", X_train.shape)\nprint(\"Shape of y_train:\", y_train.shape)\n","metadata":{"execution":{"iopub.status.busy":"2023-12-28T09:27:59.162538Z","iopub.execute_input":"2023-12-28T09:27:59.163707Z","iopub.status.idle":"2023-12-28T09:41:16.836235Z","shell.execute_reply.started":"2023-12-28T09:27:59.163674Z","shell.execute_reply":"2023-12-28T09:41:16.834236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = pd.read_csv(test_csv_path)\ndf_test = df_test.drop(columns=['image_height','image_width'])","metadata":{"execution":{"iopub.status.busy":"2023-12-28T09:42:21.920912Z","iopub.execute_input":"2023-12-28T09:42:21.921437Z","iopub.status.idle":"2023-12-28T09:42:21.940608Z","shell.execute_reply.started":"2023-12-28T09:42:21.921406Z","shell.execute_reply":"2023-12-28T09:42:21.939511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-27T07:15:57.345917Z","iopub.execute_input":"2023-12-27T07:15:57.346827Z","iopub.status.idle":"2023-12-27T07:15:57.372916Z","shell.execute_reply.started":"2023-12-27T07:15:57.346779Z","shell.execute_reply":"2023-12-27T07:15:57.368567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize X_train as an empty list to store the loaded images\nX_test_final = []\ny_test_final = []\n\n# Loop through each row in the DataFrame\nfor index, row in tqdm(df_test.iterrows(), desc=\"Loading Test Images\"):\n    # Get the image_id and label for each row\n    image_id = row['image_id']\n\n    # Construct the full file path\n    image_path = os.path.join(test_thumbnail_paths, f\"{image_id}_thumbnail.png\")\n\n    # Read and resize the image\n    img = imread(image_path)\n    target_size = (256, 256)\n    img = resize(img, target_size, anti_aliasing=True)\n\n    # Append the image to X_train and the corresponding label to y_train\n    X_test_final.append(img)\n\n# Convert the lists to numpy arrays\nX_test_final = np.array(X_test_final)\n\n# Normalize pixel values to the range [0, 1]\nX_test_final = X_test_final / 255.0\n\n# Print the shapes of X_train and y_train to verify\nprint(\"Shape of X_test:\", X_test_final.shape)\n","metadata":{"execution":{"iopub.status.busy":"2023-12-28T09:42:32.430803Z","iopub.execute_input":"2023-12-28T09:42:32.432023Z","iopub.status.idle":"2023-12-28T09:42:33.584153Z","shell.execute_reply.started":"2023-12-28T09:42:32.431975Z","shell.execute_reply":"2023-12-28T09:42:33.583092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.models import Sequential, load_model \nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, BatchNormalization, Dropout\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import EarlyStopping, ModelCheckpoint\nfrom tensorflow.keras.utils import to_categorical\n\nimage_height=256\nimage_width=256\nnum_channels=3\nnum_classes=5\n\n# One-hot encode \ny_train_encoded = to_categorical(y_train)\n#y_test_encoded = to_categorical(y_test)\n\nx_train, x_val, Y_train_encoded, Y_val_encoded = train_test_split(X_train, y_train_encoded, test_size=0.1, random_state=42)\nx_train, x_test, Y_train_encoded, Y_test_encoded = train_test_split(x_train, Y_train_encoded, test_size=0.1, random_state=42)\n\n\n# CNN \nmodel = Sequential([\n    Conv2D(32, (3, 3), activation='relu', input_shape=(image_height, image_width, num_channels)),\n    MaxPooling2D((2, 2)),\n    Conv2D(64, (3, 3), activation='relu'),\n    MaxPooling2D((2, 2)),\n    Conv2D(128, (3, 3), activation='relu'),\n    MaxPooling2D((2, 2)),\n    Flatten(),\n    Dense(256, activation='relu'),\n    Dense(128, activation='relu'),\n    Dropout(0.4),\n    Dense(num_classes, activation='sigmoid')\n])\n\noptimizer = Adam(learning_rate=0.001)\nmodel.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])\n\n# callbacks for early stopping and model checkpoint\nearly_stopping = EarlyStopping(monitor='val_loss', patience=6, restore_best_weights=True)\nmodel_checkpoint = ModelCheckpoint('best_model.h5', monitor='val_loss', save_best_only=True)\n\n# Train model with callbacks\nhistory = model.fit(x_train, Y_train_encoded, validation_data=(x_val, Y_val_encoded), epochs=40, batch_size=16, callbacks=[ model_checkpoint])\n#history = model.fit(x_train, Y_train_encoded, validation_data=(x_val, Y_val_encoded), epochs=40, batch_size=16, callbacks=[model_checkpoint])\n\n# test_loss, test_acc = model.evaluate(X_test, Y_test_encoded)\n# print(f'Test accuracy: {test_acc}')\n\n# best model from checkpoint\n# best_model = load_model('best_model.h5')\n\n# predictions = best_model.predict(x_test)\n\n\nplt.plot(history.history['accuracy'], label='Training Accuracy')\nplt.plot(history.history['val_accuracy'], label='Validation Accuracy')\nplt.xlabel('Epochs')\nplt.ylabel('Accuracy')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-28T09:45:38.763422Z","iopub.execute_input":"2023-12-28T09:45:38.763941Z","iopub.status.idle":"2023-12-28T10:32:06.040021Z","shell.execute_reply.started":"2023-12-28T09:45:38.763907Z","shell.execute_reply":"2023-12-28T10:32:06.038635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_loss, test_acc = model.evaluate(x_test, Y_test_encoded)\nprint(f'Test accuracy: {test_acc}')","metadata":{"execution":{"iopub.status.busy":"2023-12-28T10:32:08.197563Z","iopub.execute_input":"2023-12-28T10:32:08.198381Z","iopub.status.idle":"2023-12-28T10:32:11.035333Z","shell.execute_reply.started":"2023-12-28T10:32:08.198343Z","shell.execute_reply":"2023-12-28T10:32:11.033603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_model = load_model('best_model.h5')\nbest_model.save('/kaggle/working/UBC')","metadata":{"execution":{"iopub.status.busy":"2023-12-28T10:36:36.081939Z","iopub.execute_input":"2023-12-28T10:36:36.082472Z","iopub.status.idle":"2023-12-28T10:36:40.841446Z","shell.execute_reply.started":"2023-12-28T10:36:36.082431Z","shell.execute_reply":"2023-12-28T10:36:40.840207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y_test_encoded\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predicted_labels_indices = np.argmax(Y_test_encoded, axis=1)\n\n# Map numerical values to labels using the label_mapping dictionary\npredicted_labels = [list(label_mapping.keys())[list(label_mapping.values()).index(idx)] for idx in predicted_labels_indices]\n\nprint(predicted_labels)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"binary_predictions = np.zeros_like(predictions)\nbinary_predictions[np.arange(len(predictions)), predictions.argmax(axis=1)] = 1\nprint(binary_predictions)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predicted_labels_indices_ = np.argmax(binary_predictions, axis=1)\n\n# Map numerical values to labels using the label_mapping dictionary\npredicted_labels_ = [list(label_mapping.keys())[list(label_mapping.values()).index(idx)] for idx in predicted_labels_indices_]\n\nprint(predicted_labels_)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions_final = best_model.predict(X_test_final)\npredictions_final","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"binary_predictions_final = np.zeros_like(predictions_final)\nbinary_predictions_final[np.arange(len(predictions_final)), predictions_final.argmax(axis=1)] = 1\nprint(binary_predictions_final)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predicted_labels_indices_final = np.argmax(binary_predictions_final, axis=1)\n\n# Map numerical values to labels using the label_mapping dictionary\npredicted_labels_final = [list(label_mapping.keys())[list(label_mapping.values()).index(idx)] for idx in predicted_labels_indices_final]\n\nprint(predicted_labels_final)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Add the labels to the DataFrame\ndf_test['label'] = predicted_labels_final\n\n# Display the resulting DataFrame\nprint(df_test) ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = df_test\nsubmission.to_csv('/kaggle/working/submission.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{},"execution_count":null,"outputs":[]}]}