{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"},{"sourceId":233,"sourceType":"modelInstanceVersion","modelInstanceId":162}],"dockerImageVersionId":30588,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings(\"ignore\")\n\nimport os\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport tqdm\nimport cv2\nfrom PIL import Image\n\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.utils import Sequence\nfrom tensorflow.keras.callbacks import ReduceLROnPlateau\nfrom tensorflow.keras.models import load_model, save_model\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\n\nimport tensorflow_hub as hub","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-11-28T13:26:53.613839Z","iopub.execute_input":"2023-11-28T13:26:53.614093Z","iopub.status.idle":"2023-11-28T13:27:14.426559Z","shell.execute_reply.started":"2023-11-28T13:26:53.614068Z","shell.execute_reply":"2023-11-28T13:27:14.425414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:\n    seed = 42\n    \n    train_csv_path = '/kaggle/input/UBC-OCEAN/train.csv'\n    train_thumbnails_path = '/kaggle/input/UBC-OCEAN/train_thumbnails'\n    train_images_path = '/kaggle/input/UBC-OCEAN/train_images'\n    \n    images_dir = '/kaggle/working/images'\n    \n    n_classes = 5\n    \n    batch_size = 8\n    learning_rate = 1e-2\n    epochs = 50\n    img_size = (300, 300)\n    min_lr = 1e-7\n    min_delta = 1e-4\n    \n    test_csv_path = '/kaggle/input/UBC-OCEAN/test.csv'\n    test_thumbnails_path = '/kaggle/input/UBC-OCEAN/test_thumbnails'\n    test_images_path = '/kaggle/input/UBC-OCEAN/test_images'\n    \nconfig = CFG()","metadata":{"execution":{"iopub.status.busy":"2023-11-28T13:27:14.428475Z","iopub.execute_input":"2023-11-28T13:27:14.429054Z","iopub.status.idle":"2023-11-28T13:27:14.435422Z","shell.execute_reply.started":"2023-11-28T13:27:14.429023Z","shell.execute_reply":"2023-11-28T13:27:14.434381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def set_seed(seed=42):\n    '''Sets the seed of the entire notebook so results are the same every time we run.\n    This is for REPRODUCIBILITY.'''\n    np.random.seed(seed)\n    # Set a fixed value for the hash seed\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    \nset_seed(config.seed)","metadata":{"execution":{"iopub.status.busy":"2023-11-28T13:27:14.436699Z","iopub.execute_input":"2023-11-28T13:27:14.437073Z","iopub.status.idle":"2023-11-28T13:27:14.471403Z","shell.execute_reply.started":"2023-11-28T13:27:14.437031Z","shell.execute_reply":"2023-11-28T13:27:14.470616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(config.train_csv_path)\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-28T13:27:14.474082Z","iopub.execute_input":"2023-11-28T13:27:14.47439Z","iopub.status.idle":"2023-11-28T13:27:14.518644Z","shell.execute_reply.started":"2023-11-28T13:27:14.474366Z","shell.execute_reply":"2023-11-28T13:27:14.517625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.describe()","metadata":{"execution":{"iopub.status.busy":"2023-11-28T13:27:14.519844Z","iopub.execute_input":"2023-11-28T13:27:14.520194Z","iopub.status.idle":"2023-11-28T13:27:14.546185Z","shell.execute_reply.started":"2023-11-28T13:27:14.520163Z","shell.execute_reply":"2023-11-28T13:27:14.545032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_path_thumbnails(image_id):\n    if os.path.exists(config.train_thumbnails_path + '/' + str(image_id) + '_thumbnail.png'):\n        return config.train_thumbnails_path + '/' + str(image_id) + '_thumbnail.png'\n    return config.train_images_path + '/' + str(image_id) + '.png'","metadata":{"execution":{"iopub.status.busy":"2023-11-28T13:27:14.547751Z","iopub.execute_input":"2023-11-28T13:27:14.548159Z","iopub.status.idle":"2023-11-28T13:27:14.554205Z","shell.execute_reply.started":"2023-11-28T13:27:14.54812Z","shell.execute_reply":"2023-11-28T13:27:14.553153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['f_path'] = train_df['image_id'].apply(get_path_thumbnails)","metadata":{"execution":{"iopub.status.busy":"2023-11-28T13:27:14.555581Z","iopub.execute_input":"2023-11-28T13:27:14.555904Z","iopub.status.idle":"2023-11-28T13:27:15.649493Z","shell.execute_reply.started":"2023-11-28T13:27:14.555876Z","shell.execute_reply":"2023-11-28T13:27:15.648611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plt.figure(figsize=(12,8))\n#sns.countplot(data=train_df, x='label')\n#plt.title('Distribution of classes');","metadata":{"execution":{"iopub.status.busy":"2023-11-28T13:27:15.650643Z","iopub.execute_input":"2023-11-28T13:27:15.650929Z","iopub.status.idle":"2023-11-28T13:27:15.654805Z","shell.execute_reply.started":"2023-11-28T13:27:15.650905Z","shell.execute_reply":"2023-11-28T13:27:15.653933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plt.figure(figsize=(12,8))\n#sns.violinplot(data=train_df, x='label', y='image_width')\n#plt.title('Distribution of image width by classes');","metadata":{"execution":{"iopub.status.busy":"2023-11-28T13:27:15.65589Z","iopub.execute_input":"2023-11-28T13:27:15.656142Z","iopub.status.idle":"2023-11-28T13:27:15.671494Z","shell.execute_reply.started":"2023-11-28T13:27:15.656121Z","shell.execute_reply":"2023-11-28T13:27:15.670633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plt.figure(figsize=(12,8))\n#sns.violinplot(data=train_df, x='label', y='image_height')\n#plt.title('Distribution of image height by classes');","metadata":{"execution":{"iopub.status.busy":"2023-11-28T13:27:15.676322Z","iopub.execute_input":"2023-11-28T13:27:15.67662Z","iopub.status.idle":"2023-11-28T13:27:15.681571Z","shell.execute_reply.started":"2023-11-28T13:27:15.676597Z","shell.execute_reply":"2023-11-28T13:27:15.680754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plt.figure(figsize=(10,8))\n#sns.countplot(data=train_df, x='is_tma')\n#plt.title('Count of tma vs non-tma images');","metadata":{"execution":{"iopub.status.busy":"2023-11-28T13:27:15.682455Z","iopub.execute_input":"2023-11-28T13:27:15.682679Z","iopub.status.idle":"2023-11-28T13:27:15.693066Z","shell.execute_reply.started":"2023-11-28T13:27:15.68266Z","shell.execute_reply":"2023-11-28T13:27:15.692411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#random_row = train_df.iloc[200]\n#image = cv2.imread(random_row['f_path'])\n#print(f'Original image size: {random_row.image_width} x {random_row.image_height}')\n#plt.imshow(image)\n#plt.title(random_row.label);","metadata":{"execution":{"iopub.status.busy":"2023-11-28T13:27:15.694103Z","iopub.execute_input":"2023-11-28T13:27:15.694391Z","iopub.status.idle":"2023-11-28T13:27:15.707087Z","shell.execute_reply.started":"2023-11-28T13:27:15.694367Z","shell.execute_reply":"2023-11-28T13:27:15.706205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#wide_images = train_df[train_df['image_width'] >= train_df['image_width'].mean()].sample(10)\n\n#plt.figure(figsize=(20, 8))\n#for i in range(10):\n#    wide_image = wide_images.iloc[i]\n#    ax = plt.subplot(2, 5, i+1)\n#    image = cv2.imread(wide_image['f_path'])\n#    ax.imshow(image)\n#    ax.set_title(wide_image.label);\n#plt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2023-11-28T13:27:15.708189Z","iopub.execute_input":"2023-11-28T13:27:15.708506Z","iopub.status.idle":"2023-11-28T13:27:15.719004Z","shell.execute_reply.started":"2023-11-28T13:27:15.708484Z","shell.execute_reply":"2023-11-28T13:27:15.718186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#tma_row = train_df[train_df.is_tma == True].iloc[0]\n#image = cv2.imread(tma_row['f_path'])\n#print(f'Original image size: {tma_row.image_width} x {tma_row.image_height}')\n#plt.imshow(image)\n#plt.title(tma_row.label);","metadata":{"execution":{"iopub.status.busy":"2023-11-28T13:27:15.720125Z","iopub.execute_input":"2023-11-28T13:27:15.720411Z","iopub.status.idle":"2023-11-28T13:27:15.728973Z","shell.execute_reply.started":"2023-11-28T13:27:15.720388Z","shell.execute_reply":"2023-11-28T13:27:15.728215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_cropped_images(file_path, image_id, th_area = 1000):\n    image = Image.open(file_path)\n    # Aspect ratio\n    as_ratio = image.size[0] / image.size[1]\n    \n    sxs, exs, sys, eys = [],[],[],[]\n    if as_ratio >= 1.5:\n        # Crop\n        mask = np.max( np.array(image) > 0, axis=-1 ).astype(np.uint8)\n        retval, labels = cv2.connectedComponents(mask)\n        if retval >= as_ratio:\n            x, y = np.meshgrid( np.arange(image.size[0]), np.arange(image.size[1]) )\n            for label in range(1, retval):\n                area = np.sum(labels == label)\n                if area < th_area:\n                    continue\n                xs, ys= x[ labels == label ], y[ labels == label ]\n                sx, ex = np.min(xs), np.max(xs)\n                cx = (sx + ex) // 2\n                crop_size = image.size[1]\n                sx = max(0, cx-crop_size//2)\n                ex = min(sx + crop_size - 1, image.size[0]-1)\n                sx = ex - crop_size + 1\n                sy, ey = 0, image.size[1]-1\n                sxs.append(sx)\n                exs.append(ex)\n                sys.append(sy)\n                eys.append(ey)\n        else:\n            crop_size = image.size[1]\n            for i in range(int(as_ratio)):\n                sxs.append( i * crop_size )\n                exs.append( (i+1) * crop_size - 1 )\n                sys.append( 0 )\n                eys.append( crop_size - 1 )\n    else:\n        # Not Crop (entire image)\n        sxs, exs, sys, eys = [0,],[image.size[0]-1],[0,],[image.size[1]-1]\n\n    df_crop = pd.DataFrame()\n    df_crop[\"image_id\"] = [image_id] * len(sxs)\n    df_crop[\"f_path\"] = [file_path] * len(sxs)\n    df_crop[\"sx\"] = sxs\n    df_crop[\"ex\"] = exs\n    df_crop[\"sy\"] = sys\n    df_crop[\"ey\"] = eys\n    return df_crop","metadata":{"execution":{"iopub.status.busy":"2023-11-28T13:27:15.730067Z","iopub.execute_input":"2023-11-28T13:27:15.730646Z","iopub.status.idle":"2023-11-28T13:27:15.746028Z","shell.execute_reply.started":"2023-11-28T13:27:15.730615Z","shell.execute_reply":"2023-11-28T13:27:15.745164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfs = [get_cropped_images(file_path, image_id) for (file_path, image_id) in zip(train_df[\"f_path\"], train_df[\"image_id\"])]\n\ndf_crop = pd.concat(dfs)","metadata":{"execution":{"iopub.status.busy":"2023-11-28T13:27:15.747087Z","iopub.execute_input":"2023-11-28T13:27:15.747383Z","iopub.status.idle":"2023-11-28T13:29:27.209899Z","shell.execute_reply.started":"2023-11-28T13:27:15.747359Z","shell.execute_reply":"2023-11-28T13:29:27.208988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#crop_sample = df_crop[df_crop['sx']>0].sample(10)\n\n#plt.figure(figsize=(20, 8))\n#for i in range(10):\n#    crop_image = crop_sample.iloc[i]\n#    ax = plt.subplot(2, 5, i+1)\n#    image = cv2.imread(crop_image['f_path'])\n#    resized_image = image[crop_image.sy : crop_image.ey, crop_image.sx : crop_image.ex, :]\n#    ax.set_title(crop_image.image_id)\n#    ax.imshow(resized_image);\n#plt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2023-11-28T13:29:27.210927Z","iopub.execute_input":"2023-11-28T13:29:27.211181Z","iopub.status.idle":"2023-11-28T13:29:27.215552Z","shell.execute_reply.started":"2023-11-28T13:29:27.211158Z","shell.execute_reply":"2023-11-28T13:29:27.214592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_crop = pd.merge(df_crop, train_df[['image_id', 'label', 'is_tma']], on='image_id', how='left')","metadata":{"execution":{"iopub.status.busy":"2023-11-28T13:29:27.21671Z","iopub.execute_input":"2023-11-28T13:29:27.217024Z","iopub.status.idle":"2023-11-28T13:29:27.269775Z","shell.execute_reply.started":"2023-11-28T13:29:27.216999Z","shell.execute_reply":"2023-11-28T13:29:27.268956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_crop.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-28T13:29:27.270845Z","iopub.execute_input":"2023-11-28T13:29:27.271135Z","iopub.status.idle":"2023-11-28T13:29:27.284058Z","shell.execute_reply.started":"2023-11-28T13:29:27.271109Z","shell.execute_reply":"2023-11-28T13:29:27.283121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocess_image(f_path, sx, ex, sy, ey):\n    image = Image.open(f_path)\n    image = image.crop((sx, sy, ex, ey))\n    return image","metadata":{"execution":{"iopub.status.busy":"2023-11-28T13:29:27.285288Z","iopub.execute_input":"2023-11-28T13:29:27.285644Z","iopub.status.idle":"2023-11-28T13:29:27.291647Z","shell.execute_reply.started":"2023-11-28T13:29:27.285613Z","shell.execute_reply":"2023-11-28T13:29:27.29077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def save_image(row, index):\n    image = preprocess_image(row.f_path, row.sx, row.ex, row.sy, row.ey)\n    if row.is_tma:\n        image.save(f'{config.images_dir}/{row.label}/{row.image_id}_{index}.png')\n    else: image.save(f'{config.images_dir}/{row.label}/{row.image_id}_{index}_thumbnails.png')","metadata":{"execution":{"iopub.status.busy":"2023-11-28T13:29:27.292749Z","iopub.execute_input":"2023-11-28T13:29:27.293071Z","iopub.status.idle":"2023-11-28T13:29:27.304628Z","shell.execute_reply.started":"2023-11-28T13:29:27.293034Z","shell.execute_reply":"2023-11-28T13:29:27.303856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def save_images(row):\n    for index, (_, row) in enumerate(df_crop[df_crop['image_id'] == row.image_id].iterrows()):\n        save_image(row, index)","metadata":{"execution":{"iopub.status.busy":"2023-11-28T13:29:27.305809Z","iopub.execute_input":"2023-11-28T13:29:27.306163Z","iopub.status.idle":"2023-11-28T13:29:27.316978Z","shell.execute_reply.started":"2023-11-28T13:29:27.306131Z","shell.execute_reply":"2023-11-28T13:29:27.316297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir images\n\nfor label in df_crop['label'].unique():\n    os.system(f'mkdir {config.images_dir}/{label}')\n    \ntrain_df.apply(save_images, axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-11-28T13:29:27.317939Z","iopub.execute_input":"2023-11-28T13:29:27.318303Z","iopub.status.idle":"2023-11-28T13:43:06.213843Z","shell.execute_reply.started":"2023-11-28T13:29:27.31827Z","shell.execute_reply":"2023-11-28T13:43:06.212784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = tf.keras.utils.image_dataset_from_directory(\n  config.images_dir,\n  validation_split=0.2,\n  subset=\"training\",\n  seed=config.seed,\n  image_size=config.img_size,\n  batch_size=config.batch_size)\n\ndf_val = tf.keras.utils.image_dataset_from_directory(\n  config.images_dir,\n  validation_split=0.2,\n  subset=\"validation\",\n  seed=config.seed,\n  image_size=config.img_size,\n  batch_size=config.batch_size)","metadata":{"execution":{"iopub.status.busy":"2023-11-28T13:43:06.216284Z","iopub.execute_input":"2023-11-28T13:43:06.217168Z","iopub.status.idle":"2023-11-28T13:43:12.131545Z","shell.execute_reply.started":"2023-11-28T13:43:06.217127Z","shell.execute_reply":"2023-11-28T13:43:12.130505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"AUTOTUNE = tf.data.AUTOTUNE\ndf_train = df_train.cache().shuffle(800).prefetch(buffer_size=AUTOTUNE)\ndf_val = df_val.cache().prefetch(buffer_size=AUTOTUNE)","metadata":{"execution":{"iopub.status.busy":"2023-11-28T13:43:12.132773Z","iopub.execute_input":"2023-11-28T13:43:12.133081Z","iopub.status.idle":"2023-11-28T13:43:12.150512Z","shell.execute_reply.started":"2023-11-28T13:43:12.133055Z","shell.execute_reply":"2023-11-28T13:43:12.149564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"resize_and_rescale = tf.keras.Sequential([\n  layers.Resizing(*config.img_size),\n  layers.Rescaling(1./255),\n])\n\ndf_train = df_train.map(lambda x, y: (resize_and_rescale(x), y),\n              num_parallel_calls=AUTOTUNE)\n\ndf_val = df_val.map(lambda x, y: (resize_and_rescale(x), y),\n              num_parallel_calls=AUTOTUNE)","metadata":{"execution":{"iopub.status.busy":"2023-11-28T13:43:12.151607Z","iopub.execute_input":"2023-11-28T13:43:12.151884Z","iopub.status.idle":"2023-11-28T13:43:12.26099Z","shell.execute_reply.started":"2023-11-28T13:43:12.151862Z","shell.execute_reply":"2023-11-28T13:43:12.260123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"effnet = hub.KerasLayer(\"https://www.kaggle.com/models/google/efficientnet-v2/frameworks/TensorFlow2/variations/imagenet21k-ft1k-b3-feature-vector/versions/1\")","metadata":{"execution":{"iopub.status.busy":"2023-11-28T13:43:12.262221Z","iopub.execute_input":"2023-11-28T13:43:12.262598Z","iopub.status.idle":"2023-11-28T13:43:21.549698Z","shell.execute_reply.started":"2023-11-28T13:43:12.262565Z","shell.execute_reply":"2023-11-28T13:43:21.54865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"effnet.trainable = False\n\nmodel = Sequential([\n    effnet,\n    tf.keras.layers.Flatten(),\n    tf.keras.layers.Dropout(0.3), # use dropout for regularizing our model\n    tf.keras.layers.Dense(1024, activation='relu'), # add a fully-connected top layer\n    tf.keras.layers.Dropout(0.2),\n    tf.keras.layers.Dense(config.n_classes, activation='softmax'), \n])\n\nmodel.compile(optimizer=tf.keras.optimizers.SGD(learning_rate=config.learning_rate),\n                loss=tf.keras.losses.SparseCategoricalCrossentropy(),\n                metrics=['accuracy'])\n\nreduce_lr = ReduceLROnPlateau(monitor='accuracy', factor=.4,\n                                     patience=5, verbose=0, mode='max',\n                                     min_delta=config.min_delta, cooldown=3,\n                                     min_lr=config.min_lr)\nmodel.build([None, 300, 300, 3])","metadata":{"execution":{"iopub.status.busy":"2023-11-28T13:43:21.554978Z","iopub.execute_input":"2023-11-28T13:43:21.555513Z","iopub.status.idle":"2023-11-28T13:43:22.747811Z","shell.execute_reply.started":"2023-11-28T13:43:21.555486Z","shell.execute_reply":"2023-11-28T13:43:22.746797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-11-28T13:43:22.748961Z","iopub.execute_input":"2023-11-28T13:43:22.749268Z","iopub.status.idle":"2023-11-28T13:43:22.794423Z","shell.execute_reply.started":"2023-11-28T13:43:22.74923Z","shell.execute_reply":"2023-11-28T13:43:22.793494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(\n  df_train,\n  validation_data=df_val,\n  epochs=config.epochs,\n  callbacks=[reduce_lr]\n)","metadata":{"execution":{"iopub.status.busy":"2023-11-28T13:43:22.79558Z","iopub.execute_input":"2023-11-28T13:43:22.79589Z","iopub.status.idle":"2023-11-28T13:47:47.139752Z","shell.execute_reply.started":"2023-11-28T13:43:22.795867Z","shell.execute_reply":"2023-11-28T13:47:47.138927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12,6))\n\nacc = history.history['accuracy']\nval_acc = history.history['val_accuracy']\n\nloss = history.history['loss']\nval_loss = history.history['val_loss']\n\nepochs_range = range(config.epochs)\n\nplt.subplot(1, 2, 1)\nplt.plot(epochs_range, acc, label='Training Accuracy')\nplt.plot(epochs_range, val_acc, label='Validation Accuracy')\nplt.legend(loc='lower right')\n\nplt.subplot(1, 2, 2)\nplt.plot(epochs_range, loss, label='Training Loss')\nplt.plot(epochs_range, val_loss, label='Validation Loss')\nplt.legend(loc='lower right')","metadata":{"execution":{"iopub.status.busy":"2023-11-28T13:47:47.141172Z","iopub.execute_input":"2023-11-28T13:47:47.141488Z","iopub.status.idle":"2023-11-28T13:47:48.193294Z","shell.execute_reply.started":"2023-11-28T13:47:47.141463Z","shell.execute_reply":"2023-11-28T13:47:48.192388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"effnet.trainable = True\n\n# https://stackoverflow.com/a/63635774\nclass SaveBestModel(tf.keras.callbacks.Callback):\n    def __init__(self, save_best_metric='val_loss', this_max=False):\n        self.save_best_metric = save_best_metric\n        self.max = this_max\n        if this_max:\n            self.best = float('-inf')\n        else:\n            self.best = float('inf')\n\n    def on_epoch_end(self, epoch, logs=None):\n        metric_value = logs[self.save_best_metric]\n        if self.max:\n            if metric_value > self.best:\n                self.best = metric_value\n                self.best_weights = self.model.get_weights()\n\n        else:\n            if metric_value < self.best:\n                self.best = metric_value\n                self.best_weights= self.model.get_weights()\n\nsave_best_model = SaveBestModel()","metadata":{"execution":{"iopub.status.busy":"2023-11-28T13:47:48.194582Z","iopub.execute_input":"2023-11-28T13:47:48.194873Z","iopub.status.idle":"2023-11-28T13:47:48.202592Z","shell.execute_reply.started":"2023-11-28T13:47:48.194848Z","shell.execute_reply":"2023-11-28T13:47:48.201728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"epochs = 100\n\nhistory = model.fit(\n  df_train,\n  validation_data=df_val,\n  epochs=epochs,\n  callbacks=[save_best_model, reduce_lr]\n)","metadata":{"execution":{"iopub.status.busy":"2023-11-28T13:47:48.203847Z","iopub.execute_input":"2023-11-28T13:47:48.204111Z","iopub.status.idle":"2023-11-28T13:54:53.209062Z","shell.execute_reply.started":"2023-11-28T13:47:48.204088Z","shell.execute_reply":"2023-11-28T13:54:53.208095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.set_weights(save_best_model.best_weights)","metadata":{"execution":{"iopub.status.busy":"2023-11-28T13:54:53.214748Z","iopub.execute_input":"2023-11-28T13:54:53.215633Z","iopub.status.idle":"2023-11-28T13:54:53.491659Z","shell.execute_reply.started":"2023-11-28T13:54:53.2156Z","shell.execute_reply":"2023-11-28T13:54:53.490845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"save_model(model, 'effnet_model.h5')","metadata":{"execution":{"iopub.status.busy":"2023-11-28T13:54:53.492775Z","iopub.execute_input":"2023-11-28T13:54:53.493075Z","iopub.status.idle":"2023-11-28T13:54:53.969068Z","shell.execute_reply.started":"2023-11-28T13:54:53.493049Z","shell.execute_reply":"2023-11-28T13:54:53.967901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# method to get appropriate image path\ndef get_img_path(image_id):\n    thumbnail_path = f\"{config.test_thumbnails_path}/{image_id}_thumbnail.png\"\n    img_path = f\"{config.test_images_path}/{image_id}.png\"\n    \n    if os.path.exists(thumbnail_path):\n        return thumbnail_path\n    else:\n        return img_path","metadata":{"execution":{"iopub.status.busy":"2023-11-28T13:54:53.971036Z","iopub.execute_input":"2023-11-28T13:54:53.971429Z","iopub.status.idle":"2023-11-28T13:54:53.976797Z","shell.execute_reply.started":"2023-11-28T13:54:53.971401Z","shell.execute_reply":"2023-11-28T13:54:53.975751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_names = ['CC', 'EC', 'HGSC', 'LGSC', 'MC']\n\n# load the test set\ndf_test = pd.read_csv(config.test_csv_path)\n\ndf_results = []\n\nfor image_id in df_test['image_id']:\n    img_path = get_img_path(image_id)\n    \n    # load the image and convert to array\n    img = tf.keras.utils.load_img(\n        img_path, target_size=config.img_size\n    )\n    img_array = tf.keras.utils.img_to_array(img)\n    img_array = tf.expand_dims(img_array, 0)\n\n    # use the model for prediction and display results\n    predictions = model.predict(img_array)\n    score = tf.nn.softmax(predictions[0])\n    \n    # add the results\n    df_results.append([image_id, class_names[np.argmax(score)]])\n\n# prepare results to be saved\ndf_results = pd.DataFrame(df_results)\ndf_results.columns=[\"image_id\", \"label\"]\n\n# save results\ndf_results.to_csv(\"submission.csv\", mode='w', index=False)\ndf_results","metadata":{"execution":{"iopub.status.busy":"2023-11-28T13:54:53.978713Z","iopub.execute_input":"2023-11-28T13:54:53.979029Z","iopub.status.idle":"2023-11-28T13:54:56.382672Z","shell.execute_reply.started":"2023-11-28T13:54:53.979004Z","shell.execute_reply":"2023-11-28T13:54:56.381771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}