{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":13865318,"sourceType":"datasetVersion","datasetId":8833743}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Ovarian Cancer Subtype Classification\n\nOvarian carcinoma is the most deadly cancer of the female reproductive system and includes five common subtypes, along with a few rare ones.\nEach subtype has different biological and clinical characteristics, so identifying the correct subtype is important for proper treatment.\nData science techniques can help improve and automate this subtype identification process.\n\n## Dataset\n\nThis dataset focuses on classifying ovarian cancer subtypes from microscopy biopsy images. It includes large whole slide images and smaller tissue microarray images with varying sizes and magnifications. The target classes include CC, EC, HGSC, LGSC, MC","metadata":{}},{"cell_type":"code","source":"import os\nimport cv2\nimport pandas as pd\nfrom PIL import Image\nimport numpy as np","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T19:55:53.510004Z","iopub.execute_input":"2025-12-15T19:55:53.510288Z","iopub.status.idle":"2025-12-15T19:55:53.514091Z","shell.execute_reply.started":"2025-12-15T19:55:53.510266Z","shell.execute_reply":"2025-12-15T19:55:53.513354Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"root_dir = \"/kaggle/input/ovarian-cancer\"","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-15T19:55:53.515016Z","iopub.execute_input":"2025-12-15T19:55:53.515191Z","iopub.status.idle":"2025-12-15T19:55:53.526953Z","shell.execute_reply.started":"2025-12-15T19:55:53.515176Z","shell.execute_reply":"2025-12-15T19:55:53.526362Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv(f'{root_dir}/train.csv')\ndf = df[df[\"is_tma\"] == False]\nprint(df.shape)\n\ndf[\"image_thumbnail_path\"] = df[\"image_id\"].apply(\n    lambda x: f\"{root_dir}/train_thumbnails/{x}_thumbnail.png\"\n)\ndf.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T19:55:53.528008Z","iopub.execute_input":"2025-12-15T19:55:53.528194Z","iopub.status.idle":"2025-12-15T19:55:53.544832Z","shell.execute_reply.started":"2025-12-15T19:55:53.528176Z","shell.execute_reply":"2025-12-15T19:55:53.544246Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Visualize\n\n\n## Here we visualize the distributed of target classes","metadata":{}},{"cell_type":"code","source":"from matplotlib import pyplot as plt\nimport seaborn as sns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T19:55:53.545374Z","iopub.execute_input":"2025-12-15T19:55:53.54557Z","iopub.status.idle":"2025-12-15T19:55:53.549022Z","shell.execute_reply.started":"2025-12-15T19:55:53.545548Z","shell.execute_reply":"2025-12-15T19:55:53.54835Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nsns.countplot(data=df, x='label', order=df['label'].value_counts().index)\nplt.title('Distribution of Target Classes')\nplt.xlabel('Label')\nplt.ylabel('Count')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T19:55:53.550385Z","iopub.execute_input":"2025-12-15T19:55:53.550637Z","iopub.status.idle":"2025-12-15T19:55:53.679602Z","shell.execute_reply.started":"2025-12-15T19:55:53.550614Z","shell.execute_reply":"2025-12-15T19:55:53.678872Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"label_map = {\n    'HGSC': 0,\n    'EC': 1,\n    'CC': 2,\n    'LGSC': 3,\n    'MC': 4\n}\n\ndf['label_int'] = df['label'].map(label_map)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T19:55:53.680194Z","iopub.execute_input":"2025-12-15T19:55:53.680398Z","iopub.status.idle":"2025-12-15T19:55:53.686455Z","shell.execute_reply.started":"2025-12-15T19:55:53.680383Z","shell.execute_reply":"2025-12-15T19:55:53.685718Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['label_int'] = df['label'].map(label_map)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T19:55:53.687104Z","iopub.execute_input":"2025-12-15T19:55:53.687306Z","iopub.status.idle":"2025-12-15T19:55:53.698389Z","shell.execute_reply.started":"2025-12-15T19:55:53.687291Z","shell.execute_reply":"2025-12-15T19:55:53.697654Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T19:55:53.699095Z","iopub.execute_input":"2025-12-15T19:55:53.699457Z","iopub.status.idle":"2025-12-15T19:55:53.71481Z","shell.execute_reply.started":"2025-12-15T19:55:53.69944Z","shell.execute_reply":"2025-12-15T19:55:53.714208Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"HGSC_DF = df[df['label'] == 'HGSC']\nEC_DF = df[df['label'] == 'EC']\nCC_DF = df[df['label'] == 'CC']\nLGSC_DF = df[df['label'] == 'LGSC']\nMC_DF = df[df['label'] == 'MC']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T19:55:53.716712Z","iopub.execute_input":"2025-12-15T19:55:53.716913Z","iopub.status.idle":"2025-12-15T19:55:53.728102Z","shell.execute_reply.started":"2025-12-15T19:55:53.716899Z","shell.execute_reply":"2025-12-15T19:55:53.727526Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"HGSC_DF.shape, EC_DF.shape, CC_DF.shape, LGSC_DF.shape, MC_DF.shape\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T19:55:53.728762Z","iopub.execute_input":"2025-12-15T19:55:53.728933Z","iopub.status.idle":"2025-12-15T19:55:53.739652Z","shell.execute_reply.started":"2025-12-15T19:55:53.728919Z","shell.execute_reply":"2025-12-15T19:55:53.739087Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"MC_DF.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T19:55:53.740315Z","iopub.execute_input":"2025-12-15T19:55:53.740621Z","iopub.status.idle":"2025-12-15T19:55:53.755238Z","shell.execute_reply.started":"2025-12-15T19:55:53.740599Z","shell.execute_reply":"2025-12-15T19:55:53.754678Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Image Processing\n\nUsing\n- Clahe\n- Denoising\n- Resizing ","metadata":{}},{"cell_type":"code","source":"def smart_resize(image_path, target_size=(224, 224)):\n    img = Image.open(image_path).convert(\"RGB\")\n    img.thumbnail(target_size, Image.LANCZOS)\n\n    new_img = Image.new(\"RGB\", target_size, (0, 0, 0))\n    x_center = (target_size[0] - img.width) // 2\n    y_center = (target_size[1] - img.height) // 2\n    new_img.paste(img, (x_center, y_center))\n    return np.array(new_img)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T19:55:53.75597Z","iopub.execute_input":"2025-12-15T19:55:53.756307Z","iopub.status.idle":"2025-12-15T19:55:53.765962Z","shell.execute_reply.started":"2025-12-15T19:55:53.756286Z","shell.execute_reply":"2025-12-15T19:55:53.765359Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_random_resized_sample(df, target_size=(224, 224)):\n    row = df.sample(1).iloc[0]\n    img_path = row['image_thumbnail_path']\n    label = row['label']\n    resized_img = smart_resize(img_path, target_size)\n    return resized_img, label\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T19:55:53.766611Z","iopub.execute_input":"2025-12-15T19:55:53.766827Z","iopub.status.idle":"2025-12-15T19:55:53.779881Z","shell.execute_reply.started":"2025-12-15T19:55:53.766805Z","shell.execute_reply":"2025-12-15T19:55:53.779289Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Clahe**","metadata":{}},{"cell_type":"code","source":"def apply_clahe(image, clip_limit=2.0, tile_grid_size=(8, 8)):\n    # Convert to LAB color space\n    lab = cv2.cvtColor(image, cv2.COLOR_RGB2LAB)\n    l, a, b = cv2.split(lab)\n\n    # Apply CLAHE only on L channel\n    clahe = cv2.createCLAHE(clipLimit=clip_limit, tileGridSize=tile_grid_size)\n    cl = clahe.apply(l)\n\n    # Merge channels back\n    merged = cv2.merge((cl, a, b))\n    enhanced = cv2.cvtColor(merged, cv2.COLOR_LAB2RGB)\n\n    return enhanced","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T19:55:53.780589Z","iopub.execute_input":"2025-12-15T19:55:53.78088Z","iopub.status.idle":"2025-12-15T19:55:53.790633Z","shell.execute_reply.started":"2025-12-15T19:55:53.780865Z","shell.execute_reply":"2025-12-15T19:55:53.789941Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Denoising**","metadata":{}},{"cell_type":"code","source":"def apply_denoise(image, ksize=(5, 5)):\n    return cv2.GaussianBlur(image, ksize, 0.5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T19:55:53.791311Z","iopub.execute_input":"2025-12-15T19:55:53.791511Z","iopub.status.idle":"2025-12-15T19:55:53.801756Z","shell.execute_reply.started":"2025-12-15T19:55:53.791497Z","shell.execute_reply":"2025-12-15T19:55:53.801219Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_image_row(images_dict, figsize=(15, 5)):\n\n    n = len(images_dict)\n\n    plt.figure(figsize=figsize)\n    \n    for idx, (title, image) in enumerate(images_dict.items(), 1):\n        plt.subplot(1, n, idx)\n        plt.imshow(image)\n        plt.title(title)\n        plt.axis(\"off\")\n\n    plt.tight_layout()\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T19:55:53.802563Z","iopub.execute_input":"2025-12-15T19:55:53.803043Z","iopub.status.idle":"2025-12-15T19:55:53.813832Z","shell.execute_reply.started":"2025-12-15T19:55:53.803025Z","shell.execute_reply":"2025-12-15T19:55:53.813146Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## MC: Mucinous Carcinoma ","metadata":{}},{"cell_type":"code","source":"img, lbl = get_random_resized_sample(MC_DF)\nimg_clahe = apply_clahe(img)\nimg_denoise = apply_denoise(img)\nimages_to_show = {\n    \"Original\": img,\n    \"CLAHE\": img_clahe,\n    \"Denoise\": img_denoise\n}\n\nplot_image_row(images_to_show)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T19:55:53.814505Z","iopub.execute_input":"2025-12-15T19:55:53.814716Z","iopub.status.idle":"2025-12-15T19:55:54.541906Z","shell.execute_reply.started":"2025-12-15T19:55:53.814692Z","shell.execute_reply":"2025-12-15T19:55:54.540959Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## LGSC: Low-Grade Serous Carcinoma","metadata":{}},{"cell_type":"code","source":"img, lbl = get_random_resized_sample(LGSC_DF)\nimg_clahe = apply_clahe(img)\nimg_denoise = apply_denoise(img)\nimages_to_show = {\n    \"Original\": img,\n    \"CLAHE\": img_clahe,\n    \"Denoise\": img_denoise\n}\n\nplot_image_row(images_to_show)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T19:55:54.542761Z","iopub.execute_input":"2025-12-15T19:55:54.542991Z","iopub.status.idle":"2025-12-15T19:55:55.20312Z","shell.execute_reply.started":"2025-12-15T19:55:54.542974Z","shell.execute_reply":"2025-12-15T19:55:55.202157Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## CC: Clear Cell Carcinoma","metadata":{}},{"cell_type":"code","source":"img, lbl = get_random_resized_sample(CC_DF)\nimg_clahe = apply_clahe(img)\nimg_denoise = apply_denoise(img)\nimages_to_show = {\n    \"Original\": img,\n    \"CLAHE\": img_clahe,\n    \"Denoise\": img_denoise\n}\n\nplot_image_row(images_to_show)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T19:55:55.204031Z","iopub.execute_input":"2025-12-15T19:55:55.204263Z","iopub.status.idle":"2025-12-15T19:55:55.911873Z","shell.execute_reply.started":"2025-12-15T19:55:55.204237Z","shell.execute_reply":"2025-12-15T19:55:55.910947Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## EC: Endometrioid Carcinoma","metadata":{}},{"cell_type":"code","source":"img, lbl = get_random_resized_sample(EC_DF)\nimg_clahe = apply_clahe(img)\nimg_denoise = apply_denoise(img)\nimages_to_show = {\n    \"Original\": img,\n    \"CLAHE\": img_clahe,\n    \"Denoise\": img_denoise\n}\n\nplot_image_row(images_to_show)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T19:55:55.914805Z","iopub.execute_input":"2025-12-15T19:55:55.915017Z","iopub.status.idle":"2025-12-15T19:55:56.79755Z","shell.execute_reply.started":"2025-12-15T19:55:55.915Z","shell.execute_reply":"2025-12-15T19:55:56.796526Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## HGSC: High-Grade Serous Carcinoma","metadata":{}},{"cell_type":"code","source":"img, lbl = get_random_resized_sample(HGSC_DF)\nimg_clahe = apply_clahe(img)\nimg_denoise = apply_denoise(img)\nimages_to_show = {\n    \"Original\": img,\n    \"CLAHE\": img_clahe,\n    \"Denoise\": img_denoise\n}\n\nplot_image_row(images_to_show)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T19:55:56.798402Z","iopub.execute_input":"2025-12-15T19:55:56.798627Z","iopub.status.idle":"2025-12-15T19:55:57.248127Z","shell.execute_reply.started":"2025-12-15T19:55:56.79861Z","shell.execute_reply":"2025-12-15T19:55:57.247349Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model Training\n\nHere we use three type of model and also a custom model.\n\n - **EfficientNet**\n - **DenseNet**\n - **ResNet**","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\n\nimport keras\nfrom keras import layers, models\nfrom keras.applications import EfficientNetB5, DenseNet121, ResNet50\nfrom keras.callbacks import EarlyStopping, ReduceLROnPlateau\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.utils import class_weight\nfrom sklearn.metrics import classification_report, confusion_matrix\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport numpy as np\n\n\ndef evaluate_model(model, train_ds, val_ds, test_ds, history, class_names):\n    \n    # Accuracy & Loss\n    train_loss, train_acc = model.evaluate(train_ds, verbose=1)\n    val_loss, val_acc = model.evaluate(val_ds, verbose=1)\n    test_loss, test_acc = model.evaluate(test_ds, verbose=1)\n\n    print(\"Training Accuracy:\", train_acc)\n    print(\"Validation Accuracy:\", val_acc)\n    print(\"Testing Accuracy:\", test_acc)\n    print(\"Training Loss:\", train_loss)\n    print(\"Validation Loss:\", val_loss)\n    print(\"Testing Loss:\", test_loss)\n\n    # Predictions\n    y_true, y_pred = [], []\n    for x, y in test_ds:\n        preds = model.predict(x, verbose=0)\n        y_true.extend(y.numpy())\n        y_pred.extend(np.argmax(preds, axis=1))\n\n    \n    print(\"\\nClassification Report:\\n\")\n    print(classification_report(y_true, y_pred, target_names=class_names))\n\n    # Confusion matrix\n    cm = confusion_matrix(y_true, y_pred)\n    plt.figure(figsize=(6,5))\n    sns.heatmap(cm, annot=True, fmt=\"d\",\n                xticklabels=class_names,\n                yticklabels=class_names)\n    plt.xlabel(\"Predicted\")\n    plt.ylabel(\"True\")\n    plt.title(\"Confusion Matrix\")\n    plt.show()\n\n    # Accuracy graph\n    plt.figure()\n    plt.plot(history.history['acc'], label='Train Acc')\n    plt.plot(history.history['val_acc'], label='Val Acc')\n    plt.legend()\n    plt.title(\"Accuracy Curve\")\n    plt.show()\n\n    # Loss graph\n    plt.figure()\n    plt.plot(history.history['loss'], label='Train Loss')\n    plt.plot(history.history['val_loss'], label='Val Loss')\n    plt.legend()\n    plt.title(\"Loss Curve\")\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T19:55:57.249107Z","iopub.execute_input":"2025-12-15T19:55:57.249628Z","iopub.status.idle":"2025-12-15T19:55:57.258066Z","shell.execute_reply.started":"2025-12-15T19:55:57.249602Z","shell.execute_reply":"2025-12-15T19:55:57.257366Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Train Test split","metadata":{}},{"cell_type":"code","source":"train_df, temp_df = train_test_split(\n    df,\n    test_size=0.2,\n    stratify=df['label_int'],\n    random_state=42\n)\n\nval_df, test_df = train_test_split(\n    temp_df,\n    test_size=0.5,\n    stratify=temp_df['label_int'],\n    random_state=42\n)\n\nprint(\"Train:\", train_df.shape)\nprint(\"Validation:\", val_df.shape)\nprint(\"Test:\", test_df.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T19:55:57.258783Z","iopub.execute_input":"2025-12-15T19:55:57.259542Z","iopub.status.idle":"2025-12-15T19:55:57.275196Z","shell.execute_reply.started":"2025-12-15T19:55:57.259517Z","shell.execute_reply":"2025-12-15T19:55:57.274597Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Augmentation","metadata":{}},{"cell_type":"code","source":"IMG_SIZE = (224, 224)\nBATCH_SIZE = 16\n\ndata_augmentation = keras.Sequential([\n    layers.RandomFlip(\"horizontal\"),\n    layers.RandomRotation(0.2),\n    layers.RandomZoom(0.2),\n    layers.RandomContrast(0.3),\n])\n\ndef load_image(path, label, augment=False):\n    image = tf.io.read_file(path)\n    image = tf.image.decode_png(image, channels=3)\n    image = tf.image.resize(image, IMG_SIZE)\n\n    if augment:\n        image = data_augmentation(image)\n\n    return image, label\n\ndef df_to_dataset(df, augment=False, shuffle=True):\n    paths = df[\"image_thumbnail_path\"].values\n    labels = df[\"label_int\"].values\n\n    ds = tf.data.Dataset.from_tensor_slices((paths, labels))\n    ds = ds.map(\n        lambda x, y: load_image(x, y, augment),\n        num_parallel_calls=tf.data.AUTOTUNE\n    )\n\n    if shuffle:\n        ds = ds.shuffle(1000)\n\n    ds = ds.batch(BATCH_SIZE).prefetch(tf.data.AUTOTUNE)\n    return ds\n\ntrain_ds = df_to_dataset(train_df, augment=True)\nval_ds = df_to_dataset(val_df, augment=False, shuffle=False)\ntest_ds = df_to_dataset(test_df, augment=False, shuffle=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T19:55:57.276128Z","iopub.execute_input":"2025-12-15T19:55:57.276893Z","iopub.status.idle":"2025-12-15T19:55:57.528326Z","shell.execute_reply.started":"2025-12-15T19:55:57.276871Z","shell.execute_reply":"2025-12-15T19:55:57.527781Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Callbacks**","metadata":{}},{"cell_type":"code","source":"early_stop = EarlyStopping(\n    monitor='val_loss',\n    mode='min',\n    patience=6,\n    restore_best_weights=True\n)\n\nreduce_lr = ReduceLROnPlateau(\n    monitor='val_loss',\n    mode='min',\n    factor=0.3,\n    patience=3,\n    min_lr=1e-6\n)\n\ncheckpoint = keras.callbacks.ModelCheckpoint(\n    'effnet.weights.h5',\n    monitor=\"val_loss\",\n    verbose=0,\n    save_best_only=True,\n    save_weights_only=True,\n    mode=\"auto\",\n    save_freq=\"epoch\",\n    initial_value_threshold=None,\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T19:55:57.533151Z","iopub.execute_input":"2025-12-15T19:55:57.53345Z","iopub.status.idle":"2025-12-15T19:55:57.544934Z","shell.execute_reply.started":"2025-12-15T19:55:57.533429Z","shell.execute_reply":"2025-12-15T19:55:57.544372Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# EfficientNet","metadata":{}},{"cell_type":"code","source":"def build_effnetb_model():\n    base = EfficientNetB5(\n        weights='imagenet', \n        include_top=False, \n        input_shape=(224,224,3),\n    )\n\n    x = layers.GlobalAveragePooling2D()(base.output)\n    outputs = layers.Dense(5, activation='softmax')(x)\n    model = models.Model(inputs=base.input, outputs=outputs)\n    \n    model.compile(\n        optimizer=keras.optimizers.AdamW(\n            learning_rate=1e-4,\n            weight_decay=1e-5,\n        ),\n        loss=keras.losses.SparseCategoricalCrossentropy(\n            from_logits=False\n        ),\n        metrics=[\n            keras.metrics.SparseCategoricalAccuracy(name='acc')\n        ]\n    )\n    return model\n\neff_model = build_effnetb_model()\neff_model.count_params() / 1e6\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T19:55:57.54549Z","iopub.execute_input":"2025-12-15T19:55:57.545696Z","iopub.status.idle":"2025-12-15T19:55:59.940896Z","shell.execute_reply.started":"2025-12-15T19:55:57.545682Z","shell.execute_reply":"2025-12-15T19:55:59.940295Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history_eff_warmup = eff_model.fit(\n    train_ds,\n    validation_data=val_ds,\n    epochs=20,\n    callbacks=[\n        early_stop, \n        reduce_lr,\n        checkpoint\n    ]\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T19:55:59.941569Z","iopub.execute_input":"2025-12-15T19:55:59.94209Z","iopub.status.idle":"2025-12-15T20:07:02.276942Z","shell.execute_reply.started":"2025-12-15T19:55:59.94207Z","shell.execute_reply":"2025-12-15T20:07:02.276369Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Evaluation of EfficientNet","metadata":{}},{"cell_type":"code","source":"class_names = ['HGSC', 'EC', 'CC', 'LGSC', 'MC']\n\neff_model.load_weights('effnet.weights.h5')\nevaluate_model(\n    model=eff_model,\n    train_ds=train_ds,\n    val_ds=val_ds,\n    test_ds=test_ds,\n    history=history_eff_warmup,\n    class_names=class_names\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T20:07:02.277808Z","iopub.execute_input":"2025-12-15T20:07:02.278056Z","iopub.status.idle":"2025-12-15T20:08:05.183802Z","shell.execute_reply.started":"2025-12-15T20:07:02.278029Z","shell.execute_reply":"2025-12-15T20:08:05.183201Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# DenseNet","metadata":{}},{"cell_type":"code","source":"def build_densenet_model():\n    inputs = keras.Input(\n        shape=(224, 224, 3)\n    )\n    x = layers.Rescaling(1./255)(inputs)\n\n    base = DenseNet121(\n        weights='imagenet', include_top=False, input_tensor=x\n    )\n    x = base.output\n\n    x = layers.GlobalAveragePooling2D()(x)\n    outputs = layers.Dense(5, activation='softmax')(x)\n    model = models.Model(inputs=base.input, outputs=outputs)\n\n    model.compile(\n        optimizer=keras.optimizers.AdamW(\n            learning_rate=1e-4,\n            weight_decay=1e-5,\n        ),\n        loss=keras.losses.SparseCategoricalCrossentropy(\n            from_logits=False\n        ),\n        metrics=[\n            keras.metrics.SparseCategoricalAccuracy(name='acc')\n        ]\n    )\n    return model\n\ndense_model = build_densenet_model()\ndense_model.count_params() / 1e6\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T20:08:05.184576Z","iopub.execute_input":"2025-12-15T20:08:05.184915Z","iopub.status.idle":"2025-12-15T20:08:07.664173Z","shell.execute_reply.started":"2025-12-15T20:08:05.18489Z","shell.execute_reply":"2025-12-15T20:08:07.663573Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"checkpoint = keras.callbacks.ModelCheckpoint(\n    'dense121.weights.h5',\n    monitor=\"val_loss\",\n    verbose=0,\n    save_best_only=True,\n    save_weights_only=True,\n    mode=\"auto\",\n    save_freq=\"epoch\",\n    initial_value_threshold=None,\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T20:08:07.664887Z","iopub.execute_input":"2025-12-15T20:08:07.665148Z","iopub.status.idle":"2025-12-15T20:08:07.66884Z","shell.execute_reply.started":"2025-12-15T20:08:07.665124Z","shell.execute_reply":"2025-12-15T20:08:07.668161Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history_dense_warmup = dense_model.fit(\n    train_ds,\n    validation_data=val_ds,\n    epochs=20,\n    callbacks=[\n        checkpoint,\n        early_stop, \n        reduce_lr\n    ]\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T20:08:07.669579Z","iopub.execute_input":"2025-12-15T20:08:07.669826Z","iopub.status.idle":"2025-12-15T20:16:33.201868Z","shell.execute_reply.started":"2025-12-15T20:08:07.66981Z","shell.execute_reply":"2025-12-15T20:16:33.20126Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dense_model.load_weights('dense121.weights.h5')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T20:16:33.20262Z","iopub.execute_input":"2025-12-15T20:16:33.202869Z","iopub.status.idle":"2025-12-15T20:16:34.635113Z","shell.execute_reply.started":"2025-12-15T20:16:33.202841Z","shell.execute_reply":"2025-12-15T20:16:34.634564Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"evaluate_model(\n    model=dense_model,\n    train_ds=train_ds,\n    val_ds=val_ds,\n    test_ds=test_ds,\n    history=history_dense_warmup,\n    class_names=class_names\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T20:16:34.636607Z","iopub.execute_input":"2025-12-15T20:16:34.636942Z","iopub.status.idle":"2025-12-15T20:17:35.859855Z","shell.execute_reply.started":"2025-12-15T20:16:34.636916Z","shell.execute_reply":"2025-12-15T20:17:35.859109Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ResNet 50","metadata":{}},{"cell_type":"code","source":"from keras.applications import ResNet50\n\ndef build_resnet50_model():\n    inputs = keras.Input(\n        shape=(224, 224, 3)\n    )\n    x = layers.Rescaling(1./255)(inputs)\n\n    \n    base = ResNet50(\n        weights='imagenet', include_top=False, input_tensor=x\n    )\n    x = base.output\n\n    x = layers.GlobalAveragePooling2D()(x)\n    outputs = layers.Dense(5, activation='softmax')(x)\n\n    model = models.Model(inputs=base.input, outputs=outputs)\n    model.compile(\n        optimizer=keras.optimizers.AdamW(\n            learning_rate=1e-4,\n            weight_decay=1e-5,\n        ),\n        loss=keras.losses.SparseCategoricalCrossentropy(\n            from_logits=False\n        ),\n        metrics=[\n            keras.metrics.SparseCategoricalAccuracy(name='acc')\n        ]\n    )\n    return model\n\nresnet_model = build_resnet50_model()\nresnet_model.count_params() / 1e6","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T20:17:35.860739Z","iopub.execute_input":"2025-12-15T20:17:35.861002Z","iopub.status.idle":"2025-12-15T20:17:37.495885Z","shell.execute_reply.started":"2025-12-15T20:17:35.860979Z","shell.execute_reply":"2025-12-15T20:17:37.495115Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"checkpoint = keras.callbacks.ModelCheckpoint(\n    'res50.weights.h5',\n    monitor=\"val_loss\",\n    verbose=0,\n    save_best_only=True,\n    save_weights_only=True,\n    mode=\"auto\",\n    save_freq=\"epoch\",\n    initial_value_threshold=None,\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T20:17:37.496687Z","iopub.execute_input":"2025-12-15T20:17:37.497224Z","iopub.status.idle":"2025-12-15T20:17:37.500822Z","shell.execute_reply.started":"2025-12-15T20:17:37.497204Z","shell.execute_reply":"2025-12-15T20:17:37.500063Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history_resnet_warmup = resnet_model.fit(\n    train_ds,\n    validation_data=val_ds,\n    epochs=20,\n    # class_weight=class_weights,\n    callbacks=[\n        checkpoint\n        # early_stop, reduce_lr\n    ]\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T20:17:37.501548Z","iopub.execute_input":"2025-12-15T20:17:37.502089Z","iopub.status.idle":"2025-12-15T20:27:55.939824Z","shell.execute_reply.started":"2025-12-15T20:17:37.502067Z","shell.execute_reply":"2025-12-15T20:27:55.939056Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"resnet_model.load_weights('res50.weights.h5')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T20:27:55.940891Z","iopub.execute_input":"2025-12-15T20:27:55.941159Z","iopub.status.idle":"2025-12-15T20:27:56.827346Z","shell.execute_reply.started":"2025-12-15T20:27:55.941133Z","shell.execute_reply":"2025-12-15T20:27:56.826758Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"evaluate_model(\n    model=resnet_model,\n    train_ds=train_ds,\n    val_ds=val_ds,\n    test_ds=test_ds,\n    history=history_resnet_warmup,\n    class_names=class_names\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T20:27:56.828403Z","iopub.execute_input":"2025-12-15T20:27:56.82867Z","iopub.status.idle":"2025-12-15T20:28:35.586984Z","shell.execute_reply.started":"2025-12-15T20:27:56.828647Z","shell.execute_reply":"2025-12-15T20:28:35.586299Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Custom CNN (Optimized for Thumbnail)","metadata":{}},{"cell_type":"code","source":"label_map = {\n    'HGSC': 0,\n    'EC': 1,\n    'CC': 2,\n    'LGSC': 3,\n    'MC': 4\n}\ndf['label_int'] = df['label'].map(label_map)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T20:36:56.434565Z","iopub.execute_input":"2025-12-15T20:36:56.434978Z","iopub.status.idle":"2025-12-15T20:36:56.439525Z","shell.execute_reply.started":"2025-12-15T20:36:56.434961Z","shell.execute_reply":"2025-12-15T20:36:56.438694Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from keras.callbacks import EarlyStopping, ReduceLROnPlateau\n\nearly_stop = EarlyStopping(\n    monitor='val_accuracy',\n    patience=6,\n    restore_best_weights=True\n)\n\nreduce_lr = ReduceLROnPlateau(\n    monitor='val_loss',\n    factor=0.3,\n    patience=3,\n    min_lr=1e-6\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T20:36:56.440165Z","iopub.execute_input":"2025-12-15T20:36:56.440373Z","iopub.status.idle":"2025-12-15T20:36:56.453888Z","shell.execute_reply.started":"2025-12-15T20:36:56.440358Z","shell.execute_reply":"2025-12-15T20:36:56.453309Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras import models, layers\n\ndef build_custom_cnn():\n    inputs = layers.Input(shape=(224,224,3))\n\n    x = layers.Conv2D(32, (3,3), padding='same', activation='relu')(inputs)\n    x = layers.BatchNormalization()(x)\n    x = layers.MaxPooling2D((2,2))(x)\n\n    x = layers.Conv2D(64, (3,3), padding='same', activation='relu')(x)\n    x = layers.BatchNormalization()(x)\n    x = layers.MaxPooling2D((2,2))(x)\n\n    x = layers.Conv2D(128, (3,3), padding='same', activation='relu')(x)\n    x = layers.BatchNormalization()(x)\n    x = layers.MaxPooling2D((2,2))(x)\n\n    x = layers.Conv2D(256, (3,3), padding='same', activation='relu')(x)\n    x = layers.BatchNormalization()(x)\n    x = layers.MaxPooling2D((2,2))(x)\n\n    x = layers.GlobalAveragePooling2D()(x)\n    x = layers.Dense(256, activation='relu')(x)\n    x = layers.Dropout(0.5)(x)\n    outputs = layers.Dense(5, activation='softmax')(x)\n\n    model = models.Model(inputs, outputs)\n    model.compile(\n        optimizer=keras.optimizers.AdamW(\n            learning_rate=1e-4,\n            weight_decay=1e-5,\n        ),\n        loss=keras.losses.SparseCategoricalCrossentropy(\n            from_logits=False\n        ),\n        metrics=[\n            keras.metrics.SparseCategoricalAccuracy(name='acc')\n        ]\n    )\n    return model\n\ncustom_model = build_custom_cnn()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T20:28:35.661565Z","iopub.execute_input":"2025-12-15T20:28:35.661821Z","iopub.status.idle":"2025-12-15T20:28:35.749604Z","shell.execute_reply.started":"2025-12-15T20:28:35.661799Z","shell.execute_reply":"2025-12-15T20:28:35.748931Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"checkpoint = keras.callbacks.ModelCheckpoint(\n    'custom.weights.h5',\n    monitor=\"val_loss\",\n    verbose=0,\n    save_best_only=True,\n    save_weights_only=True,\n    mode=\"auto\",\n    save_freq=\"epoch\",\n    initial_value_threshold=None,\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T20:28:35.750365Z","iopub.execute_input":"2025-12-15T20:28:35.750662Z","iopub.status.idle":"2025-12-15T20:28:35.75399Z","shell.execute_reply.started":"2025-12-15T20:28:35.750645Z","shell.execute_reply":"2025-12-15T20:28:35.753276Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history_custom = custom_model.fit(\n    train_ds,\n    validation_data=val_ds,\n    epochs=20,\n    # class_weight=class_weights,\n    callbacks=[checkpoint, \n               # early_stop, reduce_lr\n              ]\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T20:28:35.754631Z","iopub.execute_input":"2025-12-15T20:28:35.755027Z","iopub.status.idle":"2025-12-15T20:36:27.239304Z","shell.execute_reply.started":"2025-12-15T20:28:35.755006Z","shell.execute_reply":"2025-12-15T20:36:27.238719Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"custom_model.load_weights('custom.weights.h5')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T20:36:27.240095Z","iopub.execute_input":"2025-12-15T20:36:27.240431Z","iopub.status.idle":"2025-12-15T20:36:27.324682Z","shell.execute_reply.started":"2025-12-15T20:36:27.240411Z","shell.execute_reply":"2025-12-15T20:36:27.324141Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"evaluate_model(\n    model=custom_model,\n    train_ds=train_ds,\n    val_ds=val_ds,\n    test_ds=test_ds,\n    history=history_custom,\n    class_names=class_names\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T20:36:27.325291Z","iopub.execute_input":"2025-12-15T20:36:27.325523Z","iopub.status.idle":"2025-12-15T20:36:56.431163Z","shell.execute_reply.started":"2025-12-15T20:36:27.325507Z","shell.execute_reply":"2025-12-15T20:36:56.430573Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}