{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"}],"dockerImageVersionId":30626,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-12-25T13:32:48.691483Z","iopub.execute_input":"2023-12-25T13:32:48.692045Z","iopub.status.idle":"2023-12-25T13:32:48.791057Z","shell.execute_reply.started":"2023-12-25T13:32:48.692Z","shell.execute_reply":"2023-12-25T13:32:48.789793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# IMPORTS","metadata":{}},{"cell_type":"markdown","source":"Data Handling and Manipulation","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd","metadata":{"execution":{"iopub.status.busy":"2023-12-25T13:32:48.793482Z","iopub.execute_input":"2023-12-25T13:32:48.794251Z","iopub.status.idle":"2023-12-25T13:32:48.799786Z","shell.execute_reply.started":"2023-12-25T13:32:48.794207Z","shell.execute_reply":"2023-12-25T13:32:48.798561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Image Processing:","metadata":{}},{"cell_type":"code","source":"import os\nfrom PIL import Image\nimport cv2","metadata":{"execution":{"iopub.status.busy":"2023-12-25T13:32:48.80098Z","iopub.execute_input":"2023-12-25T13:32:48.802029Z","iopub.status.idle":"2023-12-25T13:32:49.046971Z","shell.execute_reply.started":"2023-12-25T13:32:48.801984Z","shell.execute_reply":"2023-12-25T13:32:49.045488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Machine Learning Libraries:","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nimport keras_cv\nimport keras_core as keras\nfrom keras_core import ops\nfrom tensorflow.keras import layers, models\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report, confusion_matrix","metadata":{"execution":{"iopub.status.busy":"2023-12-25T13:32:49.049541Z","iopub.execute_input":"2023-12-25T13:32:49.049924Z","iopub.status.idle":"2023-12-25T13:32:55.074555Z","shell.execute_reply.started":"2023-12-25T13:32:49.049891Z","shell.execute_reply":"2023-12-25T13:32:55.073121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Data Augmentation:","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.image import ImageDataGenerator","metadata":{"execution":{"iopub.status.busy":"2023-12-25T13:32:55.076164Z","iopub.execute_input":"2023-12-25T13:32:55.076828Z","iopub.status.idle":"2023-12-25T13:32:55.084814Z","shell.execute_reply.started":"2023-12-25T13:32:55.076793Z","shell.execute_reply":"2023-12-25T13:32:55.083658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Model Evaluation and Visualization:","metadata":{}},{"cell_type":"code","source":"import glob\nfrom matplotlib import pyplot as plt\nfrom matplotlib.image import imread\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Set the style for the plot\nsns.set(style=\"whitegrid\")","metadata":{"execution":{"iopub.status.busy":"2023-12-25T13:32:55.085771Z","iopub.execute_input":"2023-12-25T13:32:55.086088Z","iopub.status.idle":"2023-12-25T13:32:55.22901Z","shell.execute_reply.started":"2023-12-25T13:32:55.086061Z","shell.execute_reply":"2023-12-25T13:32:55.227771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Utilities for Model Saving and Loading:","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.models import load_model","metadata":{"execution":{"iopub.status.busy":"2023-12-25T13:32:55.231994Z","iopub.execute_input":"2023-12-25T13:32:55.232933Z","iopub.status.idle":"2023-12-25T13:32:55.238357Z","shell.execute_reply.started":"2023-12-25T13:32:55.232888Z","shell.execute_reply":"2023-12-25T13:32:55.23697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# LOAD","metadata":{}},{"cell_type":"code","source":"# Training\ntrain_csv_path = \"/kaggle/input/UBC-OCEAN/train.csv\"\ntrain_thumbnail_paths = \"/kaggle/input/UBC-OCEAN/train_thumbnails\"\n    \n # Inference\ntest_csv_path = \"/kaggle/input/UBC-OCEAN/test.csv\"\ntest_thumbnail_paths = \"/kaggle/input/UBC-OCEAN/test_thumbnails\"\n\n# SEED\nkeras.utils.set_random_seed(seed=42)","metadata":{"execution":{"iopub.status.busy":"2023-12-25T13:32:55.24033Z","iopub.execute_input":"2023-12-25T13:32:55.240862Z","iopub.status.idle":"2023-12-25T13:32:55.252357Z","shell.execute_reply.started":"2023-12-25T13:32:55.240801Z","shell.execute_reply":"2023-12-25T13:32:55.251049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train a simple image classification model that will classify the scans of biopsy samples to their respective subtypes","metadata":{}},{"cell_type":"markdown","source":"The dataset comprises images related to ovarian carcinoma, categorized into two main types: whole slide images (WSI) and tissue microarray (TMA). WSI images are captured at a 20x magnification, potentially yielding large file sizes. TMAs, on the other hand, are smaller in dimensions (approximately 4,000x4,000 pixels) but at a higher 40x magnification.\n\n is_tma: A binary value indicating whether the slide is a tissue microarray. This information is only available for the train set.\n\nthe dataset includes a folder named [train/test]_thumbnails containing smaller .png versions of the whole slide images. Thumbnails, however, are not provided for TMAs.","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(train_csv_path)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-25T13:32:55.254136Z","iopub.execute_input":"2023-12-25T13:32:55.25502Z","iopub.status.idle":"2023-12-25T13:32:55.302952Z","shell.execute_reply.started":"2023-12-25T13:32:55.254898Z","shell.execute_reply":"2023-12-25T13:32:55.301544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create the thumbnail df where is_tma == False(filtering)\ndf = df[df[\"is_tma\"] == False]\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-25T13:32:55.308174Z","iopub.execute_input":"2023-12-25T13:32:55.308572Z","iopub.status.idle":"2023-12-25T13:32:55.327777Z","shell.execute_reply.started":"2023-12-25T13:32:55.308539Z","shell.execute_reply":"2023-12-25T13:32:55.326659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the distribution of the target classes\nplt.figure(figsize=(10, 6))\nsns.countplot(data=df, x='label', order=df['label'].value_counts().index)\nplt.title('Distribution of Target Classes')\nplt.xlabel('Label')\nplt.ylabel('Count')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-25T13:32:55.329026Z","iopub.execute_input":"2023-12-25T13:32:55.329341Z","iopub.status.idle":"2023-12-25T13:32:55.695256Z","shell.execute_reply.started":"2023-12-25T13:32:55.329314Z","shell.execute_reply":"2023-12-25T13:32:55.694164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.nunique()","metadata":{"execution":{"iopub.status.busy":"2023-12-25T13:32:55.697722Z","iopub.execute_input":"2023-12-25T13:32:55.698209Z","iopub.status.idle":"2023-12-25T13:32:55.713887Z","shell.execute_reply.started":"2023-12-25T13:32:55.698167Z","shell.execute_reply":"2023-12-25T13:32:55.712384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from PIL import Image\nimport random\n# Define the paths to the image directories thumbnails\ntrain_data = glob.glob('/kaggle/input/UBC-OCEAN/train_thumbnails/*.png')\ntest_data = glob.glob('/kaggle/input/UBC-OCEAN/test_thumbnails/*.png')\n\n# Display a few sample images from the training set\nnum_samples = 5\n\n# Randomly select sample images from the training set\nsample_images = random.sample(train_data, num_samples)\n\n# Display the sample images\nplt.figure(figsize=(15, 8))\nfor i, image_path in enumerate(sample_images, 1):\n    image = Image.open(image_path)\n    plt.subplot(1, num_samples, i)\n    plt.imshow(image)\n    plt.title(f'Sample {i}')\n    plt.axis('off')\n\nplt.tight_layout()\nplt.savefig('samples.png')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-25T13:32:55.715285Z","iopub.execute_input":"2023-12-25T13:32:55.715857Z","iopub.status.idle":"2023-12-25T13:33:09.81724Z","shell.execute_reply.started":"2023-12-25T13:32:55.715823Z","shell.execute_reply":"2023-12-25T13:33:09.816163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"import os\nimport pandas as pd\nimport numpy as np\nfrom skimage.io import imread\nfrom skimage.transform import resize\nfrom tqdm import tqdm\n\n# Get the list of file names in the directory\nimage_files = os.listdir(train_thumbnail_paths)\n\n# Initialize X_train as an empty list to store the loaded images\nX_train = []\n\n# Loop through each image file and load the image\nfor image_file in tqdm(image_files, desc=\"Loading Train Images\"):\n    # Construct the full file path\n    image_path = os.path.join(train_thumbnail_paths, image_file)\n    \n    # Read and resize the image\n    img = imread(image_path)\n    target_size = (256, 256)\n    img = resize(img, target_size, anti_aliasing=True)\n    \n    # Append the image to X_train\n    X_train.append(img)\n\n# Convert the list of images to a numpy array\nX_train = np.array(X_train)\n\n# Normalize pixel values to the range [0, 1]\nX_train = X_train / 255.0\n\n# Print the shape of X_train to verify the loaded images\nprint(\"Shape of X_train:\", X_train.shape)\"\"\"\n","metadata":{"execution":{"iopub.status.busy":"2023-12-25T13:33:09.818766Z","iopub.execute_input":"2023-12-25T13:33:09.819633Z","iopub.status.idle":"2023-12-25T13:33:09.829656Z","shell.execute_reply.started":"2023-12-25T13:33:09.81959Z","shell.execute_reply":"2023-12-25T13:33:09.828502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"from sklearn.model_selection import train_test_split\nfrom tensorflow.keras.models import Sequential, load_model \nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, BatchNormalization, Dropout\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import EarlyStopping, ModelCheckpoint\nfrom tensorflow.keras.utils import to_categorical\n\n# One-hot encode \ny_train_encoded = to_categorical(y_train)\ny_test_encoded = to_categorical(y_test)\"\"\"","metadata":{"execution":{"iopub.status.busy":"2023-12-25T13:33:09.831759Z","iopub.execute_input":"2023-12-25T13:33:09.832335Z","iopub.status.idle":"2023-12-25T13:33:09.848483Z","shell.execute_reply.started":"2023-12-25T13:33:09.832292Z","shell.execute_reply":"2023-12-25T13:33:09.847391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nfrom skimage.io import imread\nfrom skimage.transform import resize\nfrom tqdm import tqdm\n\n# Initialize X_train as an empty list to store the loaded images\nX_train = []\ny_train = []\n\nX_val = []\ny_val = []\n\n# Define a mapping of labels to numerical values\nlabel_mapping = {\n    \"HGSC\": 0 ,\n    \"EC\": 1,\n    \"CC\": 2,\n    \"LGSC\": 3,\n    \"MC\": 4  # Add more labels as needed\n}\n\ndatagen = ImageDataGenerator(\n    rotation_range=20,\n    width_shift_range=0.2,\n    height_shift_range=0.2,\n    shear_range=0.2,\n    zoom_range=0.2,\n    horizontal_flip=True,\n    fill_mode='nearest'\n)\n\ni=0\n# Loop through each row in the DataFrame\nfor index, row in tqdm(df.iterrows(), desc=\"Loading Train Images\"):\n    # Get the image_id and label for each row\n    image_id = row['image_id']\n    label = row['label']\n\n    # Construct the full file path\n    image_path = os.path.join(train_thumbnail_paths, f\"{image_id}_thumbnail.png\")\n\n    # Read and resize the image\n    img = imread(image_path)\n    target_size = (256, 256)\n    img = resize(img, target_size, anti_aliasing=True)\n    numerical_label = label_mapping[label]\n    \n    if (numerical_label==1):\n        # Augment the image with label \"EC\" and append both original and augmented images\n        X_train.append(img)\n        y_train.append(numerical_label)\n        \n    if (numerical_label==2):\n        # Augment the image with label \"CC\" and append both original and augmented images\n        X_train.append(img)\n        y_train.append(numerical_label)\n    \n    if (numerical_label==3):\n        # Augment the image with label \"LGSC\" and append both original and augmented images\n        X_train.append(img)\n        y_train.append(numerical_label)\n        X_train.append(img)\n        y_train.append(numerical_label)\n        X_train.append(img)\n        y_train.append(numerical_label)\n        \n    if (numerical_label==4):\n        # Augment the image with label \"MC\" and append both original and augmented images\n        X_train.append(img)\n        y_train.append(numerical_label)\n        X_train.append(img)\n        y_train.append(numerical_label)\n        X_train.append(img)\n        y_train.append(numerical_label)\n        \n    # Append the original image to X_train and the corresponding label to y_train\n    X_train.append(img)\n    y_train.append(numerical_label)\n    \n\n# Convert the lists to numpy arrays\nX_train = np.array(X_train)\ny_train = np.array(y_train)\n\n# Normalize pixel values to the range [0, 1]\nX_train = X_train / 255.0\n\n# Print the shapes of X_train and y_train to verify\nprint(\"Shape of X_train:\", X_train.shape)\nprint(\"Shape of y_train:\", y_train.shape)\n","metadata":{"execution":{"iopub.status.busy":"2023-12-25T13:33:09.850411Z","iopub.execute_input":"2023-12-25T13:33:09.850894Z","iopub.status.idle":"2023-12-25T13:45:44.252227Z","shell.execute_reply.started":"2023-12-25T13:33:09.85085Z","shell.execute_reply":"2023-12-25T13:45:44.249634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = pd.read_csv(test_csv_path)\ndf_test.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-25T13:45:44.255022Z","iopub.execute_input":"2023-12-25T13:45:44.2554Z","iopub.status.idle":"2023-12-25T13:45:44.288425Z","shell.execute_reply.started":"2023-12-25T13:45:44.255368Z","shell.execute_reply":"2023-12-25T13:45:44.285526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize X_train as an empty list to store the loaded images\nX_test = []\ny_test = []\n\n# Loop through each row in the DataFrame\nfor index, row in tqdm(df_test.iterrows(), desc=\"Loading Test Images\"):\n    # Get the image_id and label for each row\n    image_id = row['image_id']\n\n    # Construct the full file path\n    image_path = os.path.join(test_thumbnail_paths, f\"{image_id}_thumbnail.png\")\n\n    # Read and resize the image\n    img = imread(image_path)\n    target_size = (256, 256)\n    img = resize(img, target_size, anti_aliasing=True)\n\n    # Append the image to X_train and the corresponding label to y_train\n    X_test.append(img)\n\n# Convert the lists to numpy arrays\nX_test = np.array(X_test)\n\n# Normalize pixel values to the range [0, 1]\nX_test = X_test / 255.0\n\n# Print the shapes of X_train and y_train to verify\nprint(\"Shape of X_test:\", X_test.shape)\n","metadata":{"execution":{"iopub.status.busy":"2023-12-25T13:45:44.293413Z","iopub.execute_input":"2023-12-25T13:45:44.294953Z","iopub.status.idle":"2023-12-25T13:45:45.38438Z","shell.execute_reply.started":"2023-12-25T13:45:44.294761Z","shell.execute_reply":"2023-12-25T13:45:45.383526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.models import Sequential, load_model \nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, BatchNormalization, Dropout\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import EarlyStopping, ModelCheckpoint\nfrom tensorflow.keras.utils import to_categorical\n\nimage_height=256\nimage_width=256\nnum_channels=3\nnum_classes=5\n\n# One-hot encode \ny_train_encoded = to_categorical(y_train)\n#y_test_encoded = to_categorical(y_test)\n\nx_train, x_val, Y_train_encoded, Y_val_encoded = train_test_split(X_train, y_train_encoded, test_size=0.1, random_state=42)\nx_train, x_test, Y_train_encoded, Y_test_encoded = train_test_split(x_train, Y_train_encoded, test_size=0.1, random_state=42)\n\n\n# CNN \nmodel = Sequential([\n    Conv2D(32, (3, 3), activation='relu', input_shape=(image_height, image_width, num_channels)),\n    MaxPooling2D((2, 2)),\n    Conv2D(64, (3, 3), activation='relu'),\n    MaxPooling2D((2, 2)),\n    Conv2D(64, (3, 3), activation='relu'),\n    MaxPooling2D((2, 2)),\n    Flatten(),\n    Dense(256, activation='relu'),\n    Dense(64, activation='relu'),\n    Dropout(0.4),\n    Dense(num_classes, activation='sigmoid')\n])\n\noptimizer = Adam(learning_rate=0.001)\nmodel.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])\n\n# callbacks for early stopping and model checkpoint\nearly_stopping = EarlyStopping(monitor='val_loss', patience=6, restore_best_weights=True)\nmodel_checkpoint = ModelCheckpoint('best_model.h5', monitor='val_loss', save_best_only=True)\n\n# Train model with callbacks\nhistory = model.fit(x_train, Y_train_encoded, validation_data=(x_val, Y_val_encoded), epochs=50, batch_size=16, callbacks=[ model_checkpoint])\n#history = model.fit(x_train, Y_train_encoded, validation_data=(x_val, Y_val_encoded), epochs=40, batch_size=16, callbacks=[model_checkpoint])\n\n# test_loss, test_acc = model.evaluate(X_test, Y_test_encoded)\n# print(f'Test accuracy: {test_acc}')\n\n# best model from checkpoint\nbest_model = load_model('best_model.h5')\n\npredictions = best_model.predict(x_test)\n\n\nplt.plot(history.history['accuracy'], label='Training Accuracy')\nplt.plot(history.history['val_accuracy'], label='Validation Accuracy')\nplt.xlabel('Epochs')\nplt.ylabel('Accuracy')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-25T13:54:52.753836Z","iopub.execute_input":"2023-12-25T13:54:52.75429Z","iopub.status.idle":"2023-12-25T14:43:23.408632Z","shell.execute_reply.started":"2023-12-25T13:54:52.754252Z","shell.execute_reply":"2023-12-25T14:43:23.406763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_loss, test_acc = model.evaluate(x_test, Y_test_encoded)\nprint(f'Test accuracy: {test_acc}')","metadata":{"execution":{"iopub.status.busy":"2023-12-25T14:46:01.653368Z","iopub.execute_input":"2023-12-25T14:46:01.65384Z","iopub.status.idle":"2023-12-25T14:46:03.610102Z","shell.execute_reply.started":"2023-12-25T14:46:01.653805Z","shell.execute_reply":"2023-12-25T14:46:03.60886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions","metadata":{"execution":{"iopub.status.busy":"2023-12-25T14:46:04.551288Z","iopub.execute_input":"2023-12-25T14:46:04.55205Z","iopub.status.idle":"2023-12-25T14:46:04.568143Z","shell.execute_reply.started":"2023-12-25T14:46:04.552012Z","shell.execute_reply":"2023-12-25T14:46:04.567152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y_test_encoded","metadata":{"execution":{"iopub.status.busy":"2023-12-25T14:46:09.357939Z","iopub.execute_input":"2023-12-25T14:46:09.358865Z","iopub.status.idle":"2023-12-25T14:46:09.374571Z","shell.execute_reply.started":"2023-12-25T14:46:09.358829Z","shell.execute_reply":"2023-12-25T14:46:09.373174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\n\n# Assuming 'predictions' is your original array of probabilities\npredictions_binary = (predictions > 0.5).astype(int)\n\nprint(predictions_binary)","metadata":{"execution":{"iopub.status.busy":"2023-12-25T14:46:12.671073Z","iopub.execute_input":"2023-12-25T14:46:12.67178Z","iopub.status.idle":"2023-12-25T14:46:12.681156Z","shell.execute_reply.started":"2023-12-25T14:46:12.671742Z","shell.execute_reply":"2023-12-25T14:46:12.679406Z"},"trusted":true},"execution_count":null,"outputs":[]}]}