{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"}],"dockerImageVersionId":30627,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport matplotlib.pyplot as plt # data visualization\nimport seaborn as sns # data visualization\nimport cv2 as cv\nfrom PIL import Image\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    print(os.path.join(dirname))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_image_dir = '../input/UBC-OCEAN/train_images'\ntest_image_dir = '../input/UBC-OCEAN/test_images'\ntrain_thumbnails_folder_path = '../input/UBC-OCEAN/train_thumbnails'\ntest_thumbnails_folder_path = '../input/UBC-OCEAN/test_thumbnails'","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading training label file","metadata":{}},{"cell_type":"code","source":"train_data = pd.read_csv(\"../input/UBC-OCEAN/train.csv\")\ntrain_data.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Since the 'train_thumbnails' folder contains smaller-sized images, we can use them for model processing. However, not all images in the training set are located in 'train_thumbnails,' so we need to identify the missing ones and process them accordingly.","metadata":{}},{"cell_type":"code","source":"train_files = [filename.split('.')[0] for filename in os.listdir(train_image_dir)]\nthumbnails_files = [filename.replace('_thumbnail', '').split('.')[0] for filename in os.listdir(train_thumbnails_folder_path)]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_set = set(train_files)\nthumbnails_set = set(thumbnails_files)\n\nmissing_image_ids = train_set - thumbnails_set\nmissing_image_ids_list = list(missing_image_ids)\n\nprint(\"Image IDs in train_files but not in thumbnails_files:\")\nprint(missing_image_ids_list)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Missing Image id list\nmissing_image_ids_list = [int(image_id) for image_id in missing_image_ids_list]\nfiltered_train_data = train_data[train_data['image_id'].isin(missing_image_ids_list)]\n\nfiltered_train_data.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Since all the missing images in thumbnails are TMA images that are small sized, we can use them from the train_images folder instead.","metadata":{}},{"cell_type":"code","source":"train_data['full_path'] = ''\nfor index, row in train_data.iterrows():\n    image_id = row['image_id']\n    if image_id in missing_image_ids_list:\n        train_data.at[index, 'full_path'] = os.path.join(train_image_dir, str(image_id) + '.png')\n    else:\n        train_data.at[index, 'full_path'] = os.path.join(train_thumbnails_folder_path, str(image_id) + '_thumbnail.png')\n\ntrain_data.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Exploratory Data Analysis(EDA)","metadata":{}},{"cell_type":"markdown","source":"#### Label and Image Type Distribution","metadata":{}},{"cell_type":"code","source":"colors = ['#FFE4D6', '#FACBEA', '#D988B9', '#B0578D', '#EF9595']\n\n# Extracting label distribution\nlabels = train_data['label'].value_counts().index\nsizes = train_data['label'].value_counts().values\n\n# Extracting is_tma distribution\nis_tma_counts = train_data['is_tma'].value_counts()\n\n# Plotting side-by-side pie charts\nfig, (ax1, ax2) = plt.subplots(1, 2, figsize=(10, 10))\n\n# Plotting the pie chart for label distribution\nax1.pie(sizes, labels=labels, autopct='%1.1f%%', startangle=140, colors=colors)\nax1.set_title('Distribution of Labels')\n\n# Plotting the pie chart for is_tma distribution\nax2.pie(is_tma_counts, labels=is_tma_counts.index, autopct='%1.1f%%', startangle=140, colors=['#A0D8B3', '#E5F9DB'])\nax2.set_title('Distribution of is_tma')\n\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Statistice of images for each label","metadata":{}},{"cell_type":"code","source":"columns_to_exclude = ['image_id']\nstyled_summaries = {}\n\nfor label in train_data['label'].unique():\n    filtered_data = train_data[train_data['label'] == label].drop(columns=columns_to_exclude)\n    sta_summary = filtered_data.describe(include=['float64', 'int64', 'float', 'int']).round(2)\n    styled_summary = sta_summary.T.style.background_gradient(cmap='magma', low=0.2, high=0.1).set_caption(f'<h2 style=\"text-align:center;font-size:15px\">{label} Summary Table')\n    styled_summaries[label] = styled_summary\n\n# Display the styled summaries for each label\nfor label, styled_summary in styled_summaries.items():\n    display(styled_summary)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Boxplot of image_width and image_height for each label","metadata":{}},{"cell_type":"code","source":"# Set the style of seaborn\nsns.set(style=\"whitegrid\")\n\n# Create subplots for 'image_width' and 'image_height'\nfig, axes = plt.subplots(nrows=1, ncols=2, figsize=(15, 6))\n\n# Boxplot for image_width\nsns.boxplot(x='label', y='image_width', data=train_data, palette='viridis', ax=axes[0])\naxes[0].set_title('Boxplot for image_width')\n\n# Boxplot for image_height\nsns.boxplot(x='label', y='image_height', data=train_data, palette='magma', ax=axes[1])\naxes[1].set_title('Boxplot for image_height')\n\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Thumnails and train image distribution","metadata":{}},{"cell_type":"code","source":"def plot_distribution(image_dir, thumbnail_dir, data_type, ax):\n    image_count = len(os.listdir(image_dir))\n    thumbnail_count = len(os.listdir(thumbnail_dir))\n\n    labels = ['Images', 'Thumbnails']\n    counts = [image_count, thumbnail_count]\n\n    bars = ax.bar([data_type + ' ' + label for label in labels], counts, color=['#76448A', '#CCCCFF'], width=0.4)  # Adjusted bar width\n\n    for bar, count in zip(bars, counts):\n        yval = bar.get_height()\n        text_ypos = max(yval, 0.5)\n        ax.text(bar.get_x() + bar.get_width()/2, text_ypos, round(count, 2), ha='center', va='bottom', color='black')\n\n    ax.grid(False)\n    ax.set_title(f'Distribution of {data_type} Images and {data_type} Thumbnails')\n    ax.set_xlabel('Dataset Type')\n    ax.set_ylabel('Number of Files')\n\n\nfig, axes = plt.subplots(1, 2, figsize=(12, 5))\nplot_distribution(train_image_dir, train_thumbnails_folder_path, 'Train', axes[0])\nplot_distribution(test_image_dir, test_thumbnails_folder_path, 'Test', axes[1])\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Randomly plot images from training set","metadata":{}},{"cell_type":"code","source":"label_column = 'label'\npath_column = 'full_path'\nimage_id_column = 'image_id'\n\nunique_labels = train_data[label_column].unique()\n\nimages_per_label = 5\n\n# Randomly select and plot 5 images for each label\nfor label in unique_labels:\n    label_data = train_data[train_data[label_column] == label]\n    sample_images = label_data.sample(min(images_per_label, len(label_data)))\n\n    plt.figure(figsize=(10, 3))\n    plt.suptitle(f'Images for Label {label}', y=1.1, fontsize=12)  # Adjusted font size\n\n    for i, (_, row) in enumerate(sample_images.iterrows()):\n        image_path = row[path_column]\n        image = Image.open(image_path)\n\n        plt.subplot(1, images_per_label, i + 1)\n        plt.imshow(image)\n        plt.title(f'Image ID: {row[image_id_column]}', fontsize=10)\n        plt.axis('off')\n\n    plt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Plot images from test set","metadata":{}},{"cell_type":"code","source":"test_data = pd.read_csv('../input/UBC-OCEAN/test.csv')\ntest_data.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_thumbnails_folder_path = '../input/UBC-OCEAN/test_thumbnails'\ntest_data['full_path'] = test_data['image_id'].apply(lambda x: os.path.join(test_thumbnails_folder_path, f\"{x}_thumbnail.png\"))\n\ntest_data","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_id_column = 'image_id'\nwidth_column = 'image_width'\nheight_column = 'image_height'\npath_column = 'full_path'\n\nnum_images_to_plot = len(test_data)\n\nplt.figure(figsize=(10, 5 * num_images_to_plot))\nplt.suptitle(f'Images from Test Data', y=1.02, fontsize=16)\n\nfor i, (_, row) in enumerate(test_data.iterrows()):\n    image_path = row[path_column]\n    image = Image.open(image_path)\n\n    plt.subplot(num_images_to_plot, 1, i + 1)\n    plt.imshow(image)\n    plt.title(f'Image ID: {row[image_id_column]}')\n    plt.axis('off')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Preprocessing","metadata":{}},{"cell_type":"code","source":"# Convert text labels to numerical labels\nlabel_mapping = {'CC': 0, 'EC': 1, 'HGSC': 2, 'LGSC': 3, 'MC': 4}\ntrain_data['numerical_label'] = train_data['label'].map(label_mapping)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_counts = train_data['label'].value_counts()\nlabel_counts","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* As indicated above, the classes are very imbalanced. Therefore, we increase the number of images for each class by upsampling (The number increase can vary by models)","metadata":{}},{"cell_type":"code","source":"from sklearn.utils import resample\n\n# Define the target number of samples for each class\ntarget_samples = 222\n\n# Resample each class to have the target number of samples\nresampled_data = []\nfor label in train_data['numerical_label'].unique():\n    class_data = train_data[train_data['numerical_label'] == label]\n    resampled_class = resample(class_data, replace=True, n_samples=target_samples, random_state=42)\n    resampled_data.append(resampled_class)\n\nbalanced_train_data = pd.concat(resampled_data)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Transformation","metadata":{}},{"cell_type":"markdown","source":"#### Data Normalization and convert image data to numpy array","metadata":{}},{"cell_type":"code","source":"# Resize images to 512x512\nimage_size = (512, 512)\n\nx_balanced = np.empty(shape=(len(balanced_train_data), *image_size, 3), dtype=np.uint8)\ny_balanced = np.empty(shape=len(balanced_train_data), dtype=np.uint8)\n\nfor index, full_path in enumerate(balanced_train_data['full_path']):\n    image_array = Image.open(full_path).resize(image_size).convert('RGB')\n    x_balanced[index] = np.array(image_array)\n    y_balanced[index] = balanced_train_data.iloc[index]['numerical_label']\n\nprint(x_balanced.shape)\nprint(y_balanced.shape)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Set target to one-hot labels for classification problem","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder\n\ny_targets_balanced = y_balanced.reshape(len(y_balanced), -1)\nenc = OneHotEncoder()\nenc.fit(y_targets_balanced)\ny_balanced = enc.transform(y_targets_balanced).toarray()\nprint(y_balanced.shape)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(y_balanced[:5])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Data Augumentation for training data to increase the variation of images","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\n# Split the balanced data into 80% training and 20% validation\nx_train, x_val, y_train, y_val  = train_test_split(\n    x_balanced, y_balanced, test_size=0.2, random_state=1, stratify=y_balanced\n)\nx_train.shape, x_val.shape, y_train.shape, y_val.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.preprocessing.image import ImageDataGenerator\n\ndatagen = ImageDataGenerator(\n    rotation_range=20,\n    width_shift_range=0.2,\n    height_shift_range=0.2,\n    shear_range=0.2,\n    zoom_range=0.2,\n    horizontal_flip=True,\n    vertical_flip=True,\n    fill_mode='nearest'\n)\n\ndatagen.fit(x_train)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Image after agumentation","metadata":{}},{"cell_type":"code","source":"# Choose a sample image from x_train\nsample_image = x_train[2]\nsample_image = np.expand_dims(sample_image, axis=0)\n\n# Generate augmented images\naugmented_images = datagen.flow(sample_image, batch_size=5)\n\n\nplt.figure(figsize=(20, 10))\n\n# Plot the original image\nplt.subplot(1, 6, 1)\nplt.imshow(sample_image[0].astype('uint8'))\nplt.axis('off')\nplt.title('Original')\n\n# Plot the augmented images\nfor i, augmented_image in enumerate(augmented_images):\n    plt.subplot(1, 6, i + 2)\n    plt.imshow(augmented_image[0].astype('uint8'))\n    plt.axis('off')\n    plt.title(f'Augmented {i + 1}')\n\n    if i == 3:\n        break  # Display only 4 augmented images\n\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Engineering","metadata":{}},{"cell_type":"code","source":"# Examine one sample image from each label\nfeature_extraction_df = train_data.groupby('label').apply(lambda x: x.sample(1)).reset_index(drop=True)\nfeature_extraction_df = feature_extraction_df[['image_id', 'label', 'full_path']]\n\nfeature_extraction_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Color Histograms","metadata":{}},{"cell_type":"code","source":"import math\n\ndef create_color_histogram(image_path, image_id, label):\n    image = Image.open(image_path)\n    np_image = np.array(image)\n    flattened_array = np_image.reshape((-1, 3))\n\n    # Plot the color histogram using seaborn\n    sns.histplot(flattened_array, bins=256, kde=False)\n    plt.title(f'Image ID: {image_id}\\nLabel: {label}')\n    plt.xlabel('Pixel Value')\n    plt.ylabel('Frequency')\n\n\nnum_images = len(feature_extraction_df)\nnum_rows = math.ceil(num_images / 3)\n\n\nplt.figure(figsize=(15, 5 * num_rows))\n\n\nfor idx, row in feature_extraction_df.iterrows():\n    subplot_position = (num_rows, 3, idx % (3 * num_rows) + 1)\n\n    plt.subplot(*subplot_position)\n    create_color_histogram(row['full_path'], row['image_id'], row['label'])\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Grayscale Features","metadata":{}},{"cell_type":"code","source":"def apply_grayscale_and_plot(image_id, label, image_path):\n    image = Image.open(image_path)\n    grayscale_image = image.convert('L')\n\n    flattened_array = np.array(grayscale_image).flatten()\n    print(f\"Flattened Array for Image ID {image_id} and Label {label}:\\n{flattened_array}\\n\")\n\n    plt.figure(figsize=(15, 5))\n\n    # Create a subplot for the original image\n    plt.subplot(1, 2, 1)\n    plt.imshow(image)\n    plt.title(f'Original Image (ID: {image_id}, Label: {label})')\n\n    # Create a subplot for the grayscale image\n    plt.subplot(1, 2, 2)\n    plt.imshow(grayscale_image, cmap='gray')\n    plt.title(f'Grayscale Image (ID: {image_id}, Label: {label})')\n\n    plt.tight_layout()\n    plt.show()\n\n    return flattened_array\n\nflattened_arrays = []\n\nfor idx, row in feature_extraction_df.iterrows():\n    grayscale_features = apply_grayscale_and_plot(idx, row['label'], row['full_path'])\n    flattened_arrays.append(grayscale_features)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Conturing","metadata":{}},{"cell_type":"code","source":"# Function to apply contour and histogram plots to an image\ndef apply_contour_and_histogram(image_path, image_id, label):\n    # Open an image\n    image = Image.open(image_path)\n\n    # Convert the image to grayscale\n    im_array = np.array(image.convert('L'))\n\n    # Create a new figure for contour plot\n    plt.figure()\n    plt.gray()\n    plt.contour(im_array, origin='image')\n    plt.axis('equal')\n    plt.axis('off')\n    plt.title(f'Contour Plot - Image ID: {image_id}, Label: {label}')\n\n\n# Iterate over rows in feature_extraction_df\nfor idx, row in feature_extraction_df.iterrows():\n    apply_contour_and_histogram(row['full_path'], row['image_id'], row['label'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model - RestNet50","metadata":{}},{"cell_type":"code","source":"import keras.applications\nprint(dir( keras.applications))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras import layers\nfrom keras.layers import Input, Dense, Conv2D, MaxPooling2D, ZeroPadding2D, AveragePooling2D, BatchNormalization, Activation, Flatten, Dropout\nfrom keras.models import Model\nfrom keras.regularizers import l1, l2, l1_l2","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def identity_block(input_tensor, kernel_size, filters, stage, block, use_bias=True, train_bn=True):\n    filters1, filters2, filters3 = filters\n    conv_name_base = 'res' + str(stage) + block + '_branch'\n    bn_name_base = 'bn' + str(stage) + block + '_branch'\n\n    x = Conv2D(filters1, (1, 1), use_bias=use_bias, name=conv_name_base + '2a')(input_tensor)\n    x = BatchNormalization(name=bn_name_base + '2a')(x, training=train_bn)\n    x = Activation('relu')(x)\n\n    x = Conv2D(filters2, kernel_size, padding='same', use_bias=use_bias, name=conv_name_base + '2b')(x)\n    x = BatchNormalization(name=bn_name_base + '2b')(x, training=train_bn)\n    x = Activation('relu')(x)\n\n    x = Conv2D(filters3, (1, 1), use_bias=use_bias, name=conv_name_base + '2c')(x)\n    x = BatchNormalization(name=bn_name_base + '2c')(x, training=train_bn)\n\n    x = layers.add([x, input_tensor])\n    x = Activation('relu')(x)\n    return x\n\ndef conv_block(input_tensor, kernel_size, filters, stage, block, strides=(2, 2), use_bias=True, train_bn=True):\n    filters1, filters2, filters3 = filters\n    conv_name_base = 'res' + str(stage) + block + '_branch'\n    bn_name_base = 'bn' + str(stage) + block + '_branch'\n\n    x = Conv2D(filters1, (1, 1), strides=strides, use_bias=use_bias, name=conv_name_base + '2a')(input_tensor)\n    x = BatchNormalization(name=bn_name_base + '2a')(x, training=train_bn)\n    x = Activation('relu')(x)\n\n    x = Conv2D(filters2, kernel_size, padding='same', use_bias=use_bias, name=conv_name_base + '2b')(x)\n    x = BatchNormalization(name=bn_name_base + '2b')(x, training=train_bn)\n    x = Activation('relu')(x)\n    x = Conv2D(filters3, (1, 1), use_bias=use_bias, name=conv_name_base + '2c')(x)\n    x = BatchNormalization(name=bn_name_base + '2c')(x, training=train_bn)\n\n    shortcut = Conv2D(filters3, (1, 1), strides=strides, use_bias=use_bias, name=conv_name_base + '1')(input_tensor)\n    shortcut = BatchNormalization(name=bn_name_base + '1')(shortcut, training=train_bn)\n\n    x = layers.add([x, shortcut])\n    x = Activation('relu')(x)\n    return x\n\ndef ResNet50(input_shape=(512, 512, 3), classes=5):\n    img_input = Input(shape=input_shape)\n    x = ZeroPadding2D((3, 3))(img_input)\n    \n    x = Conv2D(64, (7, 7), strides=(2, 2), use_bias=True, name='conv1')(x)\n    x = BatchNormalization(name='bn_conv1')(x, training=True)\n    x = Activation('relu')(x)\n    x = MaxPooling2D((3, 3), strides=(2, 2))(x)\n\n    x = conv_block(x, 3, [64, 64, 256], stage=2, block='a', strides=(1, 1), use_bias=True, train_bn=True)\n    x = identity_block(x, 3, [64, 64, 256], stage=2, block='b', use_bias=True, train_bn=True)\n    x = identity_block(x, 3, [64, 64, 256], stage=2, block='c', use_bias=True, train_bn=True)\n\n    x = conv_block(x, 3, [128, 128, 512], stage=3, block='a', use_bias=True, train_bn=True)\n    x = identity_block(x, 3, [128, 128, 512], stage=3, block='b', use_bias=True, train_bn=True)\n    x = identity_block(x, 3, [128, 128, 512], stage=3, block='c', use_bias=True, train_bn=True)\n    x = identity_block(x, 3, [128, 128, 512], stage=3, block='d', use_bias=True, train_bn=True)\n\n    x = conv_block(x, 3, [256, 256, 1024], stage=4, block='a', use_bias=True, train_bn=True)\n    x = identity_block(x, 3, [256, 256, 1024], stage=4, block='b', use_bias=True, train_bn=True)\n    x = identity_block(x, 3, [256, 256, 1024], stage=4, block='c', use_bias=True, train_bn=True)\n    x = identity_block(x, 3, [256, 256, 1024], stage=4, block='d', use_bias=True, train_bn=True)\n    x = identity_block(x, 3, [256, 256, 1024], stage=4, block='e', use_bias=True, train_bn=True)\n    x = identity_block(x, 3, [256, 256, 1024], stage=4, block='f', use_bias=True, train_bn=True)\n\n    x = conv_block(x, 3, [512, 512, 2048], stage=5, block='a', use_bias=True, train_bn=True)\n    x = identity_block(x, 3, [512, 512, 2048], stage=5, block='b', use_bias=True, train_bn=True)\n    x = identity_block(x, 3, [512, 512, 2048], stage=5, block='c', use_bias=True, train_bn=True)\n    x = AveragePooling2D((7, 7), name='avg_pool')(x)\n    \n    x = Flatten()(x)\n    x = Dense(classes, activation='softmax')(x)\n\n\n    model = Model(img_input, x, name='resnet50')\n    return model","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = ResNet50()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Evaluation","metadata":{}},{"cell_type":"code","source":"from keras.callbacks import ReduceLROnPlateau, EarlyStopping\n\n# Define other callbacks\nreduce_lr = ReduceLROnPlateau(monitor='val_loss', factor=0.1, patience=2, min_lr=0.0001)\nearly_stop = EarlyStopping(monitor=\"val_loss\", mode=\"min\", patience=100)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.optimizers import Adam\n\n\n# Compile the model\nmodel.compile(optimizer='adam', loss='categorical_crossentropy', metrics=['accuracy'])\n\n# Train the model\nepochs = 100\nbatch_size = 8\n\nhistory = model.fit(datagen.flow(x_train, y_train, batch_size=batch_size),\n                          steps_per_epoch=len(x_train) / batch_size,\n                          validation_data=(x_val, y_val),\n                          epochs=epochs,\n                          callbacks=[reduce_lr, early_stop])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Training Accuracy:\", history.history['accuracy'][-1])\nprint(\"Validation Accuracy:\", history.history['val_accuracy'][-1])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Classification report","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import classification_report\n\ny_true = np.argmax(y_val, axis=1)\ny_pred = np.argmax(model.predict(x_val), axis=1)\n\n# Generate the classification report\nclass_report = classification_report(y_true, y_pred)\n\n\nprint(\"\\n Validation Classification Report:\\n\", class_report)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Testing Model","metadata":{}},{"cell_type":"code","source":"test_data = pd.read_csv('../input/UBC-OCEAN/test.csv')\ntest_data","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_thumbnails_folder_path = '../input/UBC-OCEAN/test_thumbnails'\ntest_data['full_path'] = test_data['image_id'].apply(lambda x: os.path.join(test_thumbnails_folder_path, f\"{x}_thumbnail.png\"))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test = np.empty(shape=(len(test_data), 512, 512, 3), dtype=np.uint8)\n\nfor index, full_path in enumerate(test_data['full_path']):\n    image_array = Image.open(full_path).resize((512, 512)).convert('RGB')\n    x_test[index] = image_array\n\nprint(x_test.shape)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = model.predict(x_test)\npredictions","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data['label'] = predictions.argmax(axis=1)  # Assuming one-hot encoding, get the index of the max value\nsubmission_df = test_data[['image_id', 'label']]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"reverse_label_mapping = {v: k for k, v in label_mapping.items()}\n\n# Map numerical labels to actual labels\nsubmission_df.loc[:, 'label'] = submission_df['label'].map(reverse_label_mapping)\n\nsubmission_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save to CSV\nsubmission_df.to_csv('submission.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}