{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"},{"sourceId":7330943,"sourceType":"datasetVersion","datasetId":4255536}],"dockerImageVersionId":30626,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nfrom tensorflow import keras\nimport tensorflow as tf\nfrom keras import layers\n#from tensorflow.python.keras.backend import cond\n#from keras import ops\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport sklearn\nfrom sklearn.model_selection import train_test_split\nfrom keras import regularizers\nfrom tensorflow.keras import applications as app\nimport logging\nfrom tensorflow.keras.models import load_model\nfrom sklearn.metrics import confusion_matrix\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.metrics import confusion_matrix\nfrom tensorflow.image import extract_patches\nimport numpy as np","metadata":{"execution":{"iopub.status.busy":"2024-01-03T13:34:04.773915Z","iopub.execute_input":"2024-01-03T13:34:04.774368Z","iopub.status.idle":"2024-01-03T13:34:21.440321Z","shell.execute_reply.started":"2024-01-03T13:34:04.774317Z","shell.execute_reply":"2024-01-03T13:34:21.438836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.image import ImageDataGenerator\nimport os\nimport time\nfrom tensorflow import keras\nimport tensorflow as tf\nfrom keras import layers\n#from tensorflow.python.keras.backend import cond\n#from keras import ops\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport sklearn\nfrom sklearn.model_selection import train_test_split\nfrom keras import regularizers\nfrom tensorflow.keras import applications as app\nimport logging\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.metrics import confusion_matrix\nimport numpy as np\nfrom tensorflow.image import extract_patches\nimport PIL.Image as Image\nImage.MAX_IMAGE_PIXELS = None\nos.environ[\"OMP_NUM_THREADS\"] = \"1\"\nfrom keras.models import load_model\nfrom keras.models import model_from_json\n","metadata":{"execution":{"iopub.status.busy":"2024-01-03T13:34:31.697729Z","iopub.execute_input":"2024-01-03T13:34:31.698486Z","iopub.status.idle":"2024-01-03T13:34:31.708873Z","shell.execute_reply.started":"2024-01-03T13:34:31.698431Z","shell.execute_reply":"2024-01-03T13:34:31.707434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Set Arguments** \nFor a new prediction, just change training_data_full_path and train_csv. we considered the same names for both training and prediction modes.","metadata":{}},{"cell_type":"code","source":"################1- Mandatory Args ################\nTraining_mode = False # enters the prediction mode (for competition testing)\ntraining_data_full_path = \"/kaggle/input/UBC-OCEAN/test_images\"  # path to training data, For the test predictions, set training mode to False and set this as the test file path\ntrain_csv = \"/kaggle/input/UBC-OCEAN/test.csv\" #label dataframe file that contains columns: [image_id, label,  image_width,  image_height,  is_tma]\ntrained_model = \"/kaggle/input/ubc-final-model/checkpoint.weights.h5\" # weights for trained model\nmodel_architecture = \"/kaggle/input/ubc-final-model/model_before_training.h5\"\n################2- Optional Args ################\nimage_id_column = 'image_id' # column that contains image ids in train_csv\ndata_type = \".png\" # to read image files, also if there is any suffix in addition to image ids, you should add them here\nbatch_size = 2\nseed = 43\nuse_attention = True\ntarget_size=(2000,2000)\n # if false, the model will be set to predict mode, and the best model parameters are used for predictions\nvalidation_split = 0.1\n\n# Augmentation Params:\nrotation_range = 40\nwidth_shift_range = .2\nheight_shift_range = .2\nshear_range = 0.2\nzoom_range = 0.2\nhorizontal_flip = True\nvertical_flip = True\n\n# Transfer Learning Parameters\ntransfer_learning_trainable = False  # Bool, weather update transfer learning model weights or not\npooling=\"avg\" \t\t\t # - pooling (str): pooling method (avg, max, None); default(\"max\")\ninclude_top=False \t\t #   - include_top (bool): weather use classifier or not; default(False)\nweights=\"imagenet\" \t\t #    - weights (str): which weights to use; default(\"imagenet\")\ntrainable = False\n# Transformer Model Parameters\nnum_layers = 1 # num transformer blocks\ndropout_rate_attention = 0.1\nnum_heads = 8\ndropout_rate_mlp = 0.2\nprojection_dim = 64\nuse_attention=False\npatch_size=512\n# Classifier\nmlp_units_classifier = [128,64]\ndropout_rate_classifier = 0.3\nregularization_rate = 0.01\nactivation_function  = \"elu\"\nregularization_type = None\nnum_classes = 5\n\n\n# Training params:\nnum_epochs =150\nlearning_rate = 0.03\n\n# prediction mode\npretrained_model_dir = \"/kaggle/input/ubc-final-model/checkpoint.weights.h5\"","metadata":{"execution":{"iopub.status.busy":"2024-01-03T13:48:24.19057Z","iopub.execute_input":"2024-01-03T13:48:24.191085Z","iopub.status.idle":"2024-01-03T13:48:24.202843Z","shell.execute_reply.started":"2024-01-03T13:48:24.191042Z","shell.execute_reply":"2024-01-03T13:48:24.20172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" # **Model utils and functions**","metadata":{}},{"cell_type":"code","source":"\n# MLP builder\ndef mlp_builder(x, hidden_units, dropout_rate, seed,\n        regularization_type=None, regularization_rate=0.01, activation_function='gelu'):\n    \"\"\"\n    Multi-Layer Perceptron (MLP) function with options for regularization, activation, etc.\n\n    Parameters:\n    - x: Input tensor\n    - hidden_units: List of integers, representing the number of units in each hidden layer\n    - dropout_rate: Float, dropout rate for dropout layers\n    - seed: Integer, seed for random number generation\n    - regularization_type: String, type of regularization ('l1', 'l2', None)\n    - regularization_rate: Float, regularization rate (if applicable)\n    - activation_function: String, activation function for hidden layers\n\n    Returns:\n    - Output tensor\n    \"\"\"\n    if seed != None:\n      initializer = keras.initializers.GlorotUniform(seed=seed)\n    for units in hidden_units:\n        # Activation Function\n        activation_func = getattr(tf.keras.activations, activation_function, None)\n        if activation_func is None:\n            raise ValueError(f\"Invalid activation function: {activation_function}\")\n\n        # Regularization\n        if regularization_type is not None:\n            if regularization_type not in ['l1', 'l2']:\n                raise ValueError(f\"Invalid regularization type: {regularization_type}\")\n            regularizer = getattr(regularizers, regularization_type)(l=regularization_rate)\n        else:\n            regularizer = None\n\n        x = layers.Dense(units,\n                         activation=activation_func,\n                         kernel_regularizer=regularizer,\n                         kernel_initializer=initializer)(x)\n        x = layers.Dropout(dropout_rate, seed=seed)(x)\n    return x\n\n\n\nclass Patches(layers.Layer):\n    def __init__(self, patch_size, **kwargs):\n        super(Patches, self).__init__(**kwargs)\n        self.patch_size = patch_size\n\n    def call(self, images):\n        # Your existing implementation remains unchanged\n        input_shape = tf.shape(images)\n        batch_size = input_shape[0]\n        height = input_shape[1]\n        width = input_shape[2]\n        channels = input_shape[3]\n\n        num_patches_h = height // self.patch_size\n        num_patches_w = width // self.patch_size\n\n        patches = tf.image.extract_patches(\n            images,\n            sizes=[1, self.patch_size, self.patch_size, 1],\n            strides=[1, self.patch_size, self.patch_size, 1],\n            rates=[1, 1, 1, 1],\n            padding='VALID'\n        )\n\n        patches = tf.reshape(\n            patches,\n            (batch_size, num_patches_h * num_patches_w, self.patch_size * self.patch_size * channels)\n        )\n\n        return patches\n\n    def get_config(self):\n        config = super(Patches, self).get_config()\n        config.update({\"patch_size\": self.patch_size})\n        return config\n\nclass PatchEncoder(layers.Layer):\n    def __init__(self, num_patches, projection_dim, **kwargs):\n        super().__init__(**kwargs)\n        self.num_patches = num_patches\n        self.projection = layers.Dense(units=projection_dim)\n        self.position_embedding = layers.Embedding(\n            input_dim=num_patches, output_dim=projection_dim\n        )\n\n    def call(self, patch):\n        positions = tf.expand_dims(\n            tf.range(start=0, limit=self.num_patches, delta=1), axis=0\n        )\n        projected_patches = self.projection(patch)\n        encoded = projected_patches + self.position_embedding(positions)\n        return encoded\n\n    def get_config(self):\n        config = super().get_config()\n        config.update({\"num_patches\": self.num_patches})\n        return config\n\n\n#Transfer Learning builder\ndef transfer_learning_builder(base_model, input_shape,\n                              trainable=False, pooling=None):\n  \"\"\"\n  builds the transfer learning base model\n  params:\n   - base_model (keras.applications.object)\n   - input_shape (3 dims): eg. (150,150,3)\n   - trainable(Bool): freeze transfered model or not\n\n  \"\"\"\n  inputs = keras.Input(shape=input_shape)\n  base_model.trainable=trainable\n  x = base_model(inputs, training=trainable)\n  if pooling == None:\n    x = keras.layers.GlobalAveragePooling2D()(x)\n  model = keras.Model(inputs, x)\n  return model\n\n\n\n# for loop transfer leaning\ndef For_loop_TR(model, input_dirs, output, test=False):\n    \"\"\"\n    Gets model and input directory that contains all image tiles.\n    \"\"\"\n    if not os.path.exists(output):\n        os.mkdir(output)\n\n    file_names = [dir for dir in os.listdir(input_dirs) if \".npz\" in dir]\n\n    if test:\n        file_names = file_names[0:5]\n\n    total_iterations = len(file_names)\n\n    for i, file in enumerate(file_names, 1):\n        output_file_path = os.path.join(output, file)\n        print(\"output_file path::::\",output_file_path)\n\n        # Check if the output file already exists\n        if os.path.exists(f\"{output_file_path}.npy\"):\n            print(f\"Iteration {i}/{total_iterations}: Output file {output_file_path} already exists. Skipping.\")\n            continue\n\n        input_dir = os.path.join(input_dirs, file)\n        print(f\"Iteration {i}/{total_iterations}: Processing file {file}\")\n        print(input_dir)\n\n        img_tiles = np.load(input_dir)\n        img_tiles = img_tiles['dataset1']\n        print(img_tiles.shape)\n\n        preds = model.predict(img_tiles)\n        print(preds.shape)\n\n        preds = np.array(preds)\n        print(preds.shape)\n\n        np.save(output_file_path, preds)\n\n\n#Patch builder\ndef patch_builder(list_files, input_dir, padding_limit, feature_size):\n  \"\"\"\n  gets the input directory, which is the output of Transfer learning model\n  with a provided padding limit, which is the size of tiles for all images\n  and the feature size which is the number of dim[-1] of input_dir\n  \"\"\"\n  #inputs = os.listdir(input_dir)\n  inputs = list_files\n  zeros = [0] * feature_size\n  L = []\n  names = []\n  for input in inputs:\n    names.append(input.split(\".\")[0])\n    file = os.path.join(input_dir, input)\n    arr = np.load(file)\n    if arr.shape[0] <= padding_limit:\n      padding_size = padding_limit - arr.shape[0]\n      Z = zeros * padding_size\n      Z = np.array(Z).reshape((padding_size, feature_size))\n      arr = np.concatenate((arr,Z), axis=0)\n    L.append(arr)\n\n  return np.stack(L, axis=0), names\n\n\n\n\n\n\n\n# Transformer Builder\ndef transformer_builder(input_npy, num_layers=2, epsilon=1e-6,\n                        dropout_rate_attention=0.1,num_heads=8,\n                        dropout_rate_mlp=0.2, projection_dim=256,\n                        seed=43):\n  \"\"\"\n  Model Description:\n  -------------------\n  The transformer_builder function builds a simplified Transformer model with a specified number of layers.\n  Each layer consists of a pre-norm MultiHead Attention block followed by a Multi-Layer Perceptron (MLP) block.\n  The final output is a flattened representation tensor.\n\n  Parameters:\n  -----------\n  - input_npy: Input tensor representing the input data.\n  - num_layers: Number of transformer layers to stack. Default is 2.\n  - epsilon: A small positive constant added to the variance in LayerNormalization. Default is 1e-6.\n  - dropout_rate_attention: Dropout rate applied to attention output. Default is 0.1.\n  - num_heads: Number of attention heads in the MultiHead Attention block. Default is 8.\n  - dropout_rate_mlp: Dropout rate applied to the output of the MLP block. Default is 0.2.\n  - projection_dim: Dimensionality of the projected key space in MultiHead Attention. Default is 256.\n  - seed: Seed for random number generation. Default is 43.\n\n  Returns:\n  --------\n  - representation: The final flattened representation tensor output by the transformer model.\n  \"\"\"\n\n\n  transformer_units = [projection_dim*2 ,projection_dim]\n  for _ in range(num_layers):\n    #x1 = layers.LayerNormalization(epsilon=epsilon)(input_npy)\n    attention_output = layers.MultiHeadAttention(num_heads=num_heads,\n                                      key_dim=projection_dim,\n                                      dropout=dropout_rate_attention)(input_npy,input_npy)\n    x2 = layers.Add()([attention_output, input_npy])\n    x3 = layers.LayerNormalization(epsilon=epsilon)(x2)\n    x3 = mlp_builder(x=x3, hidden_units=transformer_units,\n                     dropout_rate=dropout_rate_mlp, seed=43,\n                     regularization_type=None, regularization_rate=0.01,\n                     activation_function='elu')\n    input_npy = layers.Add()([x3, x2])\n\n    # Create a [batch_size, projection_dim] tensor.\n  representation = layers.LayerNormalization(epsilon=epsilon)(input_npy)\n  representation = layers.Flatten()(representation)\n\n  return representation\n\n\n\n# Classifier Builder\ndef classifier_builder(representation_1, representation_2,\n                       mlp_units, dropout_rate, regularization_rate=0.01,\n                       activation_function=\"gelu\", regularization_type=None,\n                       num_classes=6, use_two_heads=True):\n  \"\"\"\n    # Classifier Builder Description:\n  # ------------------------------\n  # The classifier_builder function constructs a classification model by concatenating two input representations\n  # and applying a multi-layer perceptron (MLP) block. The final layer is a Dense layer with softmax activation\n  # for multi-class classification.\n\n  # Parameters:\n  # -----------\n  # - representation_1: The first input representation tensor.\n  # - representation_2: The second input representation tensor.\n  # - mlp_units: List of integers representing the number of units in each hidden layer of the MLP block.\n  # - dropout_rate: Dropout rate applied to the concatenated tensor before the MLP block.\n  # - regularization_rate: Rate for L1 or L2 regularization (if applicable). Default is 0.01.\n  # - activation_function: Activation function for the MLP block. Default is \"gelu\".\n  # - regularization_type: Type of regularization (\"l1\", \"l2\", or None). Default is None.\n  # - num_classes: Number of classes in the classification task.\n\n  # Returns:\n  # --------\n  # - classifier: The final Dense layer representing the classifier for the given representations.\n        it can be used as the outpuf for the final model\n  \"\"\"\n\n  if use_two_heads:\n      concat = layers.Concatenate()([representation_1, representation_2])\n  else:\n      concat = representation_2\n  concat = layers.Dropout(dropout_rate)(concat)\n  features =  mlp_builder(x=concat, hidden_units=mlp_units,\n                     dropout_rate=dropout_rate, seed=43,\n                     regularization_type=regularization_type,\n                          regularization_rate=regularization_rate,\n                      activation_function=activation_function)\n  classifier = layers.Dense(num_classes, activation=\"softmax\")(features)\n  return classifier\n\n\n# Define model\ndef model_builder(input_npy_1,  num_layers=2, epsilon=1e-6,\n                        dropout_rate_attention=0.1,num_heads=8,\n                        dropout_rate_mlp=0.2, projection_dim=256,\n                        seed=43,\n                        mlp_units_classifier=[1024, 512, 256, 128, 64, 32, 16],\n                        dropout_rate_classifier=0.1, regularization_rate=0.01,\n                        activation_function=\"gelu\", regularization_type=None,\n                        num_classes=6, use_attention=True, patch_size=256,\n                        transfer_lr_model=None, trainable=False, pooling=None):\n\n\n  input_npy_1 = keras.layers.Input(shape=(input_npy_1))\n\n  #Build Transformer Blocks and get representations per each model\n  print(\"----------------------------\")\n  \n  if use_attention:\n      print(\"4-  Build Transformer Blocks and get representations per each model\")\n      # Create patches.\n      patches = Patches(patch_size)(input_npy_1)\n      num_patches = patches.shape[1]\n      # Encode patches.\n      encoded_patches = PatchEncoder(num_patches, projection_dim)(patches)\n    \n      representation_1 = transformer_builder(encoded_patches, num_layers, epsilon,\n                            dropout_rate_attention,num_heads,\n                            dropout_rate_mlp, projection_dim,\n                            seed=43)\n  else:\n      representation_1 = layers.Flatten()(input_npy_1)\n\n  transfer_lr_model.trainable=trainable\n  representation_2 = transfer_lr_model(input_npy_1, training=trainable)\n  if pooling == None:\n      representation_2 = keras.layers.GlobalAveragePooling2D()(representation_2)\n  # Ensemble models and build classifier\n  print(\"----------------------------\")\n  print(\"5-  Ensemble models and build classifier\")\n  classifier = classifier_builder(representation_1, representation_2,\n                       mlp_units=mlp_units_classifier,\n                       dropout_rate=dropout_rate_classifier,\n                       regularization_rate=regularization_rate,\n                       activation_function=activation_function,\n                       regularization_type=regularization_type,\n                       num_classes=num_classes, use_two_heads=use_attention)\n  #Build model\n  print(\"---------------------------\")\n  print(\"6- building model\")\n  model = keras.Model(inputs=input_npy_1, outputs=classifier)\n  model.save(\"model_before_training.h5\")\n  print(model.summary())\n  return model\n\n\n\n# Add outlier to data\ndef outlier_adder(padded_feature_arrays, labels, num_outliers=30, label_num=6):\n\n\n  outlier_data = np.random.rand(num_outliers, *padded_feature_arrays.shape[1:])\n  combined_data = np.concatenate([padded_feature_arrays, outlier_data], axis=0)\n  outlier_label = np.array([label_num] * num_outliers)\n  combined_label = np.concatenate([labels, outlier_label], axis=0)\n\n  return combined_data, combined_label\n\n\n\n\n# split training and testing data\ndef split_train_test_data(datashape, combined_labels,\n                          test_portion=0.2,\n                          seed = 42):\n\n\n    # Create an array of indices corresponding to the batches\n  batch_indices = np.arange(datashape[0])\n  \n\n  # Perform a train-test split on the batch indices\n  batch_indices_train, batch_indices_test, y_train, y_test = train_test_split(\n      batch_indices, combined_labels, test_size=test_portion, random_state=42,\n      stratify=combined_labels)\n  return batch_indices_train, batch_indices_test, y_train, y_test\n\n\n\n\n\n\n\n\n\n\n#compile and train model\ndef compile_fit_model(model, learning_rate, weight_decay,\n                      x_train_1, x_train_2, y_train,\n                      x_test_1, x_test_2, y_test, \n                      num_epochs, validation_split=0.1, \n                      batch_size=64):\n    \"\"\"\n    Compile and fit a Keras model.\n\n    Parameters:\n    - model (keras.Model): The Keras model to compile and fit.\n    - learning_rate (float): The learning rate for the optimizer.\n    - weight_decay (float): The weight decay for the AdamW optimizer.\n    - x_train_1 (numpy.ndarray): Training data for input 1.\n    - x_train_2 (numpy.ndarray): Training data for input 2.\n    - y_train (numpy.ndarray): Training labels.\n    - x_test_1 and 2 (numpy.ndarray): Test data for both inputs.\n    - y_test (numpy.ndarray): Test labels.\n    - num_epochs (int): Number of training epochs.\n    - validation_split (float): Fraction of the training data to be used as validation data.\n    - batch_size (int): Batch size for training.\n\n    Returns:\n    - history (keras.callbacks.History): Training history.\n    \"\"\"\n    if not os.path.exists(\"tmp\"):\n        os.mkdir(\"tmp\")\n    optimizer = keras.optimizers.AdamW(\n        learning_rate=learning_rate, weight_decay=weight_decay\n    )\n\n    model.compile(\n        optimizer=optimizer,\n        loss=keras.losses.SparseCategoricalCrossentropy(from_logits=False),\n        metrics=[\n            keras.metrics.SparseCategoricalAccuracy(name=\"accuracy\"),\n            keras.metrics.SparseTopKCategoricalAccuracy(5, name=\"top-5-accuracy\"),\n        ],\n    )\n\n    checkpoint_filepath = \"tmp/checkpoint.weights.h5\"\n    checkpoint_callback = keras.callbacks.ModelCheckpoint(\n        checkpoint_filepath,\n        monitor=\"val_accuracy\",\n        save_best_only=True,\n        save_weights_only=True,\n    )\n\n    history = model.fit(\n        x=[x_train_1, x_train_2],\n        y=y_train,\n        batch_size=batch_size,\n        epochs=num_epochs,\n        validation_split=validation_split,  # or use validation_data=(x_test, y_test)\n        callbacks=[checkpoint_callback],\n    )\n\n    model.load_weights(checkpoint_filepath)\n    model.save(\"tmp/final_model.h5\")\n    _, accuracy, top_5_accuracy = model.evaluate([x_test_1, x_test_2], y_test)  # pass the inputs as a list\n    print(f\"Test accuracy: {round(accuracy * 100, 2)}%\")\n    print(f\"Test top 5 accuracy: {round(top_5_accuracy * 100, 2)}%\")\n\n    return history, model\n\n\n\ndef compile_fit_model_with_generator(model, learning_rate,\n                      train_gen, val_gen,\n                      num_epochs):\n    \"\"\"\n    Compile and fit a Keras model.\n\n    Parameters:\n    - model (keras.Model): The Keras model to compile and fit.\n    - learning_rate (float): The learning rate for the optimizer.\n    - weight_decay (float): The weight decay for the AdamW optimizer.\n    - x_train_1 (numpy.ndarray): Training data for input 1.\n    - x_train_2 (numpy.ndarray): Training data for input 2.\n    - y_train (numpy.ndarray): Training labels.\n    - x_test_1 and 2 (numpy.ndarray): Test data for both inputs.\n    - y_test (numpy.ndarray): Test labels.\n    - num_epochs (int): Number of training epochs.\n    - validation_split (float): Fraction of the training data to be used as validation data.\n    - batch_size (int): Batch size for training.\n\n    Returns:\n    - history (keras.callbacks.History): Training history.\n    \"\"\"\n    if not os.path.exists(\"tmp\"):\n        os.mkdir(\"tmp\")\n\n\n    model.compile(\n        optimizer=\"adam\",\n        loss='categorical_crossentropy',\n        metrics=[\"accuracy\"],\n    )\n\n    checkpoint_filepath = \"tmp/checkpoint.weights.h5\"\n    checkpoint_callback = keras.callbacks.ModelCheckpoint(\n        checkpoint_filepath,\n        monitor=\"val_accuracy\",\n        save_best_only=True,\n        save_weights_only=True,\n    )\n\n    history = model.fit(\n    train_gen,\n    steps_per_epoch=len(train_gen),\n    epochs=num_epochs,\n    validation_data=val_gen,\n    validation_steps=len(val_gen),\n    callbacks=[checkpoint_callback])\n\n    model.load_weights(checkpoint_filepath)\n    model.save(\"tmp/final_model.h5\")\n\n    return history, model\n\n\n\n#Plot results\ndef plot_history(history, item):\n    plt.plot(history.history[item], label=item)\n    plt.plot(history.history[\"val_\" + item], label=\"val_\" + item)\n    plt.xlabel(\"Epochs\")\n    plt.ylabel(item)\n    plt.title(\"Train and Validation {} Over Epochs\".format(item), fontsize=14)\n    plt.legend()\n    plt.grid()\n    plt.show()\n\n\n\ndef save_history_plot(history, item, save_path):\n    plt.plot(history.history[item], label=item)\n    plt.plot(history.history[\"val_\" + item], label=\"val_\" + item)\n    plt.xlabel(\"Epochs\")\n    plt.ylabel(item)\n    plt.title(\"Train and Validation {} Over Epochs\".format(item), fontsize=14)\n    plt.legend()\n    plt.grid()\n    plt.savefig(save_path)\n    plt.close()\n    \n\n\n\ndef log_and_print(message):\n    logging.info(message)\n    print(message)\ndef plot_confusion_matrix(y_true, x_test_1, x_test_2, model, class_label_dict, save_path=\"confusion.png\"):\n    \"\"\"\n    Calculate, plot, and optionally save the confusion matrix.\n\n    Parameters:\n    - y_true: True labels\n    - x_test_1: Test data input 1\n    - x_test_2: Test data input 2\n    - model: Trained model\n    - class_label_dict: Dictionary mapping class names to numeric labels\n    - save_path: Optional path to save the plot (e.g., 'confusion_matrix.png')\n\n    Returns:\n    - cm: Confusion matrix\n    \"\"\"\n\n    # Calculate the confusion matrix\n    y_pred_prob = model.predict([x_test_1, x_test_2])\n    \n    # Convert predicted probabilities to class labels\n    y_pred_labels = np.argmax(y_pred_prob, axis=1)\n\n    # Map numeric labels to class names using the provided dictionary\n    class_labels = [key for key, value in sorted(class_label_dict.items(), key=lambda x: x[1])]\n    \n    # Create a confusion matrix using scikit-learn\n    cm = confusion_matrix(y_true, y_pred_labels)\n\n    # Plot the confusion matrix using seaborn\n    plt.figure(figsize=(8, 6))\n    sns.heatmap(cm, annot=True, fmt='d', cmap='Blues', xticklabels=class_labels, yticklabels=class_labels)\n    plt.title('Confusion Matrix')\n    plt.xlabel('Predicted')\n    plt.ylabel('True')\n\n    # Save the plot\n    plt.savefig(save_path, bbox_inches='tight')\n\n    return cm\n","metadata":{"execution":{"iopub.status.busy":"2024-01-03T13:48:30.392612Z","iopub.execute_input":"2024-01-03T13:48:30.393034Z","iopub.status.idle":"2024-01-03T13:48:30.473691Z","shell.execute_reply.started":"2024-01-03T13:48:30.392997Z","shell.execute_reply":"2024-01-03T13:48:30.472427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Prediction mode**","metadata":{}},{"cell_type":"code","source":"################## ################## ################## ################## ###################\n    ################3- Build and Train the Model, DO NOT CHANGE THIS ###################\n################## ################## ################## ################## ###################\n# Step 1: Read the CSV file\nuse_two_heads = use_attention\ntarget_size_inp=(target_size[0],target_size[1],3)\ndf = pd.read_csv(train_csv)\ndf['full_path'] = df[image_id_column].apply(lambda x: f'{x}{data_type}')  # Adjust column names\n\n\n\n\n\n\n\n\n\n###########TRAINING MODE:\nif Training_mode:\n    \n    # Step 2: Create ImageDataGenerator\n    datagen = ImageDataGenerator(rescale=1./255,\n                                 validation_split=validation_split,\n                                 rotation_range=rotation_range,\n                                 width_shift_range=width_shift_range,\n                                 height_shift_range=height_shift_range,\n                                 shear_range=shear_range,\n                                 zoom_range=zoom_range,\n                                 horizontal_flip=horizontal_flip,\n                                 vertical_flip=vertical_flip,\n                                 fill_mode='nearest'\n                                 )  # You can add other augmentations as needed\n    \n    log_and_print(\"#### Entering Training mode ####\")\n    log_and_print(\"1- Image Data Generator\")\n    # Step 4: Create data generators\n    train_generator = datagen.flow_from_dataframe(\n        dataframe=df,\n        directory=training_data_full_path,\n        x_col='full_path',\n        y_col='label',\n        target_size=target_size,\n        batch_size=batch_size,\n        class_mode='categorical',  # Use 'binary' for binary classification\n        shuffle=True,\n        subset='training',\n        seed=seed\n    )\n    validation_generator = datagen.flow_from_dataframe(\n        dataframe=df,\n        directory=training_data_full_path,\n        x_col='full_path',\n        y_col='label',\n        target_size=target_size,\n        batch_size=batch_size,\n        class_mode='categorical',  # Use 'binary' for binary classification\n        shuffle=True,\n        subset='validation',\n        seed=seed\n    )\n    \n    print(validation_generator.samples)\n\n\n    # Transfer Learning model\n    log_and_print(\"-------------------------\")\n    log_and_print(\"2- Transfer Learning model\")\n    base_model = app.EfficientNetV2B0(include_top=include_top,\n                                weights=weights,\n                                input_shape=target_size_inp,\n                                pooling=pooling)\n\n    # Build model\n    model = model_builder(input_npy_1=target_size_inp, \n                  num_layers=num_layers, epsilon=1e-6,\n                  dropout_rate_attention=dropout_rate_attention, \n                  num_heads=num_heads,\n                  dropout_rate_mlp=dropout_rate_mlp, \n                  projection_dim=projection_dim,\n                  seed=seed,\n                  mlp_units_classifier=mlp_units_classifier,\n                  dropout_rate_classifier=dropout_rate_classifier, \n                  regularization_rate=regularization_rate,\n                  activation_function=activation_function, \n                  regularization_type=regularization_type,\n                  num_classes=num_classes, \n                  use_attention=use_attention, patch_size=patch_size,\n                  transfer_lr_model=base_model, trainable=trainable, pooling=pooling)\n    \n    # compile and Fit model\n    log_and_print(\"-----------\")\n    log_and_print(\"6- Compile and Fit the model\")\n    history, model = compile_fit_model_with_generator(model=model,\n                                                      learning_rate=learning_rate,\n                                                      train_gen=train_generator,\n                                                      val_gen=validation_generator,\n                                                      num_epochs=num_epochs)\n    model_json = model.to_json()\n    with open(\"model_architecture.json\", \"w\") as json_file:\n        json_file.write(model_json)\n\nelse:\n    pred_datagen = ImageDataGenerator(rescale=1./255,\n                             fill_mode='nearest')\n\n    prediction_generator = pred_datagen.flow_from_dataframe(\n        dataframe=df,\n        directory=training_data_full_path,\n        x_col='full_path',\n        target_size=target_size,\n        batch_size=batch_size,\n        class_mode=None,\n        shuffle=False  # Use 'binary' for binary classification\n    )\n    \n    loaded_model = keras.models.load_model(model_architecture)\n    loaded_model.load_weights(trained_model)\n    \n    preds = loaded_model.predict(prediction_generator)\n    # You can use preds to compare with y_pred data\n\n","metadata":{"execution":{"iopub.status.busy":"2024-01-03T14:39:39.026814Z","iopub.execute_input":"2024-01-03T14:39:39.027235Z","iopub.status.idle":"2024-01-03T14:40:54.769647Z","shell.execute_reply.started":"2024-01-03T14:39:39.027205Z","shell.execute_reply":"2024-01-03T14:40:54.76862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"out_label={0: 'HGSC',1: 'LGSC', 2: 'EC', 3: 'CC', 4: 'MC', 5: 'Outliers'}\ndf_prediction=pd.DataFrame(df[\"image_id\"])\ndf_prediction[\"label\"]=\"\"\nfor i in range(len(df_prediction)):\n    df_prediction.iloc[i,1]=out_label[np.argmax(preds[i])]\ndf_prediction.to_csv(\"/kaggle/working/submission.csv\",index=False)","metadata":{"execution":{"iopub.status.busy":"2024-01-03T15:49:48.503031Z","iopub.execute_input":"2024-01-03T15:49:48.503527Z","iopub.status.idle":"2024-01-03T15:49:48.516577Z","shell.execute_reply.started":"2024-01-03T15:49:48.503489Z","shell.execute_reply":"2024-01-03T15:49:48.515004Z"},"trusted":true},"execution_count":null,"outputs":[]}]}