{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"}],"dockerImageVersionId":30587,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras import layers, models\nimport keras_cv\nfrom keras_core import ops\nimport keras\n\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport pandas as pd\nimport pickle\n\nimport os","metadata":{"execution":{"iopub.status.busy":"2023-12-07T04:40:50.77506Z","iopub.execute_input":"2023-12-07T04:40:50.775456Z","iopub.status.idle":"2023-12-07T04:40:50.782151Z","shell.execute_reply.started":"2023-12-07T04:40:50.775421Z","shell.execute_reply":"2023-12-07T04:40:50.781086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!python --version","metadata":{"execution":{"iopub.status.busy":"2023-12-07T04:40:52.17993Z","iopub.execute_input":"2023-12-07T04:40:52.180371Z","iopub.status.idle":"2023-12-07T04:40:53.26242Z","shell.execute_reply.started":"2023-12-07T04:40:52.180338Z","shell.execute_reply":"2023-12-07T04:40:53.261371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Preprocessing","metadata":{}},{"cell_type":"code","source":"def read_image(path):\n    file = tf.io.read_file(path)\n    image = tf.io.decode_png(file, 3)\n    image = tf.image.resize(image, (256, 256))\n    image = tf.image.per_image_standardization(image)\n    return image","metadata":{"execution":{"iopub.status.busy":"2023-12-07T04:40:55.106892Z","iopub.execute_input":"2023-12-07T04:40:55.108093Z","iopub.status.idle":"2023-12-07T04:40:55.11477Z","shell.execute_reply.started":"2023-12-07T04:40:55.108041Z","shell.execute_reply":"2023-12-07T04:40:55.113737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"read_image('/kaggle/input/UBC-OCEAN/train_thumbnails/10077_thumbnail.png').shape\n#converts the input to (256, 256, 3) for the moddle","metadata":{"execution":{"iopub.status.busy":"2023-12-07T04:40:57.198494Z","iopub.execute_input":"2023-12-07T04:40:57.199099Z","iopub.status.idle":"2023-12-07T04:40:57.374117Z","shell.execute_reply.started":"2023-12-07T04:40:57.199056Z","shell.execute_reply":"2023-12-07T04:40:57.373077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/UBC-OCEAN/train.csv')\n\ndf = df[df[\"is_tma\"] == False]\n\n\n# Perform one-hot encoding of the 'label' \ndf_one_hot = pd.get_dummies(df[\"label\"], prefix=\"label\").astype(int)\n\n# Concatenate the original DataFrame with the one-hot encoded labels\ntrain_df = pd.concat([df[\"image_id\"], df_one_hot], axis=1)\n\n# Get the thumbnail image paths\ntrain_df[\"image_thumbnail_path\"] = train_df[\"image_id\"].apply(lambda x: f\"{'/kaggle/input/UBC-OCEAN/train_thumbnails'}/{x}_thumbnail.png\")\n\n\n# use these to setup the data pipeline\nimage_thumbnail_paths = train_df[\"image_thumbnail_path\"].values\nlabels = train_df[[col for col in train_df.columns if col.startswith(\"label_\")]].values","metadata":{"execution":{"iopub.status.busy":"2023-12-07T04:40:59.16419Z","iopub.execute_input":"2023-12-07T04:40:59.164573Z","iopub.status.idle":"2023-12-07T04:40:59.182949Z","shell.execute_reply.started":"2023-12-07T04:40:59.164544Z","shell.execute_reply":"2023-12-07T04:40:59.182145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#the dataset may be imbalanced\nlabel_counts = df['label'].value_counts()\n\n# Plot a histogram of the label distribution\nplt.figure(figsize=(10, 6))\nlabel_counts.plot(kind='bar')\nplt.title('Distribution of Labels')\nplt.xlabel('Labels')\nplt.ylabel('Frequency')\nplt.xticks(rotation=0)  # Rotates X-Axis Ticks by 45-degrees\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-07T03:34:51.821418Z","iopub.execute_input":"2023-12-07T03:34:51.821715Z","iopub.status.idle":"2023-12-07T03:34:52.068799Z","shell.execute_reply.started":"2023-12-07T03:34:51.82169Z","shell.execute_reply":"2023-12-07T03:34:52.067795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It is very imbalanced, so will assign class weights","metadata":{}},{"cell_type":"code","source":"from sklearn.utils.class_weight import compute_class_weight\n\n# Convert the one-hot encoded DataFrame to integer labels\ninteger_labels = np.argmax(df_one_hot.values, axis=1)\n\n# Calculate class weights\nclass_weights = compute_class_weight('balanced', classes=np.unique(integer_labels), y=integer_labels)\n\n# Create a class weights dictionary\nclass_weights_dict = {i: weight for i, weight in enumerate(class_weights)}\n","metadata":{"execution":{"iopub.status.busy":"2023-12-07T03:34:52.070126Z","iopub.execute_input":"2023-12-07T03:34:52.070424Z","iopub.status.idle":"2023-12-07T03:34:52.366047Z","shell.execute_reply.started":"2023-12-07T03:34:52.070398Z","shell.execute_reply":"2023-12-07T03:34:52.36481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#save the ids to reconvert\nlabel_names = [col for col in train_df.columns if col.startswith(\"label_\")]\nname_to_id = {key.replace(\"label_\", \"\"):value for value,key in enumerate(label_names)}\nid_to_name = {key:value for value, key in name_to_id.items()}\n\n# Save to dictionary to disk\nwith open(\"id_to_name.pkl\", \"wb\") as f:\n    pickle.dump(id_to_name, f)","metadata":{"execution":{"iopub.status.busy":"2023-12-07T03:34:52.367446Z","iopub.execute_input":"2023-12-07T03:34:52.367754Z","iopub.status.idle":"2023-12-07T03:34:52.374702Z","shell.execute_reply.started":"2023-12-07T03:34:52.367728Z","shell.execute_reply":"2023-12-07T03:34:52.373726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create the pipeline for the data\nx = (\n    tf.data.Dataset.from_tensor_slices(image_thumbnail_paths)\n    .map(read_image, num_parallel_calls=tf.data.AUTOTUNE)\n)\ny = tf.data.Dataset.from_tensor_slices(labels)\n\n# Zip the x and y together\nds = tf.data.Dataset.zip((x, y))\n\n# Create the training and validation splits\nval_ds = (\n    ds\n    .take(50)\n    .batch(7)\n    .prefetch(tf.data.AUTOTUNE)\n)\ntrain_ds = (\n    ds\n    .skip(50)\n    .shuffle(7 * 10)\n    .batch(7)\n    .prefetch(tf.data.AUTOTUNE)\n)","metadata":{"execution":{"iopub.status.busy":"2023-12-07T03:34:52.375919Z","iopub.execute_input":"2023-12-07T03:34:52.3762Z","iopub.status.idle":"2023-12-07T03:34:52.510592Z","shell.execute_reply.started":"2023-12-07T03:34:52.376172Z","shell.execute_reply":"2023-12-07T03:34:52.50963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Neural Network Building\n","metadata":{}},{"cell_type":"markdown","source":"### My model ","metadata":{}},{"cell_type":"code","source":"#(256, 256, 3) is shape of tensor\n\n# build, compile, fit model\n\nfrom tensorflow.keras import regularizers\n\n# Define the CNN model for multi-class classification\ndef create_model():\n    \n    data_augmentation = tf.keras.Sequential([\n        layers.RandomFlip(\"horizontal_and_vertical\"),\n        layers.RandomRotation(0.2),\n        layers.RandomZoom(0.1),\n        layers.RandomContrast(0.1)\n    ])\n    \n    model = models.Sequential([\n        data_augmentation,  # layer to augment images to not overfit\n        layers.Conv2D(32, (3, 3), activation='relu', input_shape=(256, 256, 3)),\n        layers.MaxPooling2D((2, 2)),\n        layers.Conv2D(64, (3, 3), activation='relu'),\n        layers.BatchNormalization(),\n        layers.MaxPooling2D((2, 2)),\n        layers.Conv2D(128, (3, 3), activation='relu'),\n        layers.BatchNormalization(),\n        layers.MaxPooling2D((2, 2)),\n        layers.Flatten(),\n        layers.Dense(128, activation='relu', kernel_regularizer=regularizers.l2(0.01)), #L2 regularization\n        layers.Dropout(0.5), #prevent overfitting\n        layers.Dense(5, activation='softmax')  \n    ])\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2023-12-07T03:34:52.514732Z","iopub.execute_input":"2023-12-07T03:34:52.515027Z","iopub.status.idle":"2023-12-07T03:34:52.52302Z","shell.execute_reply.started":"2023-12-07T03:34:52.515003Z","shell.execute_reply":"2023-12-07T03:34:52.522059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau, ModelCheckpoint\n\n# Define your callbacks\nearly_stopping = EarlyStopping(monitor='val_loss', patience=5, verbose=1, restore_best_weights=True)\nreduce_lr = ReduceLROnPlateau(monitor='val_loss', factor=0.2, patience=2, verbose=1, min_lr=1e-5)\nmodel_checkpoint = ModelCheckpoint('best_model.h5', monitor='val_loss', save_best_only=True, verbose=1)\n","metadata":{"execution":{"iopub.status.busy":"2023-12-07T03:34:52.524129Z","iopub.execute_input":"2023-12-07T03:34:52.524374Z","iopub.status.idle":"2023-12-07T03:34:52.532816Z","shell.execute_reply.started":"2023-12-07T03:34:52.524352Z","shell.execute_reply":"2023-12-07T03:34:52.532129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = create_model()\n\noptimizer = tf.keras.optimizers.Adam(learning_rate=0.001)\nmodel = create_model()\nmodel.compile(optimizer=optimizer,\n            loss='categorical_crossentropy',  \n           metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2023-12-07T03:41:58.725299Z","iopub.execute_input":"2023-12-07T03:41:58.725666Z","iopub.status.idle":"2023-12-07T03:41:58.787118Z","shell.execute_reply.started":"2023-12-07T03:41:58.725636Z","shell.execute_reply":"2023-12-07T03:41:58.786158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(\n    train_ds, \n    epochs=8, \n    validation_data=val_ds,\n    callbacks=[early_stopping, reduce_lr, model_checkpoint],\n    class_weight=class_weights_dict  # Add class weights here\n)","metadata":{"execution":{"iopub.status.busy":"2023-12-07T03:42:10.772318Z","iopub.execute_input":"2023-12-07T03:42:10.772847Z","iopub.status.idle":"2023-12-07T03:49:21.655989Z","shell.execute_reply.started":"2023-12-07T03:42:10.772773Z","shell.execute_reply":"2023-12-07T03:49:21.655105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"model.summary()","metadata":{}},{"cell_type":"code","source":"#visualize the training\ntraining_loss = history.history['loss']\nvalidation_loss = history.history['val_loss']\ntraining_accuracy = history.history['accuracy']\nvalidation_accuracy = history.history['val_accuracy']\n","metadata":{"execution":{"iopub.status.busy":"2023-12-07T03:51:09.635675Z","iopub.execute_input":"2023-12-07T03:51:09.636651Z","iopub.status.idle":"2023-12-07T03:51:09.642053Z","shell.execute_reply.started":"2023-12-07T03:51:09.636612Z","shell.execute_reply":"2023-12-07T03:51:09.641154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotting training and validation loss\nplt.figure(figsize=(12, 8))\nplt.plot(training_loss, label='Training Loss')\nplt.plot(validation_loss, label='Validation Loss')\nplt.title('Training and Validation Loss')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend()\nplt.show()\n\n# Plotting training and validation accuracy\nplt.figure(figsize=(12, 8))\nplt.plot(training_accuracy, label='Training Accuracy')\nplt.plot(validation_accuracy, label='Validation Accuracy')\nplt.title('Training and Validation Accuracy')\nplt.xlabel('Epochs')\nplt.ylabel('Accuracy')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-07T03:51:11.864017Z","iopub.execute_input":"2023-12-07T03:51:11.864387Z","iopub.status.idle":"2023-12-07T03:51:12.38381Z","shell.execute_reply.started":"2023-12-07T03:51:11.864357Z","shell.execute_reply":"2023-12-07T03:51:12.383007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## using preset backbone","metadata":{}},{"cell_type":"code","source":"# Load the image and text backbones with presets\nresnet_backbone = keras_cv.models.ResNetV2Backbone.from_preset(\n    \"resnet50_v2\",\n)\nresnet_backbone.trainable = False\n\nimage_inputs = resnet_backbone.input\nimage_embeddings = resnet_backbone(image_inputs)\nimage_embeddings = keras.layers.GlobalAveragePooling2D()(image_embeddings)\n\nx = layers.BatchNormalization(epsilon=1e-05, momentum=0.1)(image_embeddings)\nx = layers.Dense(units=1024, activation=\"relu\")(x)\nx = layers.Dropout(0.1)(x)\nx = layers.Dense(units=512, activation=\"relu\")(x)\nx = layers.Dropout(0.1)(x)\nx = layers.Dense(units=256, activation=\"relu\")(x)\noutputs = layers.Dense(units=5, activation=\"softmax\")(x)\n\n# Build the model with the Functional API (that's how they did it on their site, so don't want to mess it up by using models.Sequential instead)\nmodel = keras.Model(\n    inputs=image_inputs,\n    outputs=outputs,\n)\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-12-07T04:03:10.064482Z","iopub.execute_input":"2023-12-07T04:03:10.064867Z","iopub.status.idle":"2023-12-07T04:03:11.756728Z","shell.execute_reply.started":"2023-12-07T04:03:10.064816Z","shell.execute_reply":"2023-12-07T04:03:11.755915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"optimizer = tf.keras.optimizers.Adam(learning_rate=0.001)\nmodel.compile(optimizer=optimizer,\n            loss='categorical_crossentropy',  \n           metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2023-12-07T04:04:36.308615Z","iopub.execute_input":"2023-12-07T04:04:36.309313Z","iopub.status.idle":"2023-12-07T04:04:36.324739Z","shell.execute_reply.started":"2023-12-07T04:04:36.309277Z","shell.execute_reply":"2023-12-07T04:04:36.323875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(\n    train_ds, \n    epochs=8, \n    validation_data=val_ds,\n    callbacks=[early_stopping, reduce_lr, model_checkpoint],\n    class_weight=class_weights_dict  # Add class weights here\n)","metadata":{"execution":{"iopub.status.busy":"2023-12-07T04:05:18.668405Z","iopub.execute_input":"2023-12-07T04:05:18.669116Z","iopub.status.idle":"2023-12-07T04:10:17.977516Z","shell.execute_reply.started":"2023-12-07T04:05:18.669081Z","shell.execute_reply":"2023-12-07T04:10:17.976407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#visualize the training\ntraining_loss = history.history['loss']\nvalidation_loss = history.history['val_loss']\ntraining_accuracy = history.history['accuracy']\nvalidation_accuracy = history.history['val_accuracy']","metadata":{"execution":{"iopub.status.busy":"2023-12-07T04:10:35.37221Z","iopub.execute_input":"2023-12-07T04:10:35.373104Z","iopub.status.idle":"2023-12-07T04:10:35.377905Z","shell.execute_reply.started":"2023-12-07T04:10:35.373069Z","shell.execute_reply":"2023-12-07T04:10:35.376776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotting training and validation loss\nplt.figure(figsize=(12, 8))\nplt.plot(training_loss, label='Training Loss')\nplt.plot(validation_loss, label='Validation Loss')\nplt.title('Training and Validation Loss')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend()\nplt.show()\n\n# Plotting training and validation accuracy\nplt.figure(figsize=(12, 8))\nplt.plot(training_accuracy, label='Training Accuracy')\nplt.plot(validation_accuracy, label='Validation Accuracy')\nplt.title('Training and Validation Accuracy')\nplt.xlabel('Epochs')\nplt.ylabel('Accuracy')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-07T04:10:37.84634Z","iopub.execute_input":"2023-12-07T04:10:37.847176Z","iopub.status.idle":"2023-12-07T04:10:38.514752Z","shell.execute_reply.started":"2023-12-07T04:10:37.847144Z","shell.execute_reply":"2023-12-07T04:10:38.513881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Submission to Competition","metadata":{}},{"cell_type":"code","source":"from tensorflow.python.ops.numpy_ops import np_config\nnp_config.enable_numpy_behavior()","metadata":{"execution":{"iopub.status.busy":"2023-12-07T04:11:19.391189Z","iopub.execute_input":"2023-12-07T04:11:19.392196Z","iopub.status.idle":"2023-12-07T04:11:19.396582Z","shell.execute_reply.started":"2023-12-07T04:11:19.392159Z","shell.execute_reply":"2023-12-07T04:11:19.395652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read test.csv to get image IDs\ndf = pd.read_csv('/kaggle/input/UBC-OCEAN/test.csv')\ndf[\"image_path\"] = df[\"image_id\"].apply(lambda x: f\"/kaggle/input/UBC-OCEAN/test_thumbnails/{x}_thumbnail.png\")","metadata":{"execution":{"iopub.status.busy":"2023-12-07T04:11:25.289982Z","iopub.execute_input":"2023-12-07T04:11:25.290876Z","iopub.status.idle":"2023-12-07T04:11:25.302096Z","shell.execute_reply.started":"2023-12-07T04:11:25.290842Z","shell.execute_reply":"2023-12-07T04:11:25.301105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open(\"id_to_name.pkl\", \"rb\") as f:\n    id_to_name = pickle.load(f)","metadata":{"execution":{"iopub.status.busy":"2023-12-07T04:11:27.062407Z","iopub.execute_input":"2023-12-07T04:11:27.06326Z","iopub.status.idle":"2023-12-07T04:11:27.067857Z","shell.execute_reply.started":"2023-12-07T04:11:27.063224Z","shell.execute_reply":"2023-12-07T04:11:27.066947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predicted_labels = []\n\nfor index, row in df.iterrows():\n    # Get the image path\n    image_path = row[\"image_path\"]\n\n    # Get the image\n    image = read_image(image_path)[None, ...]\n\n    # Predict the label\n    logits = model.predict(image)\n    pred = ops.argmax(logits, axis=-1).tolist()[0]\n\n    # Map the pred to the name\n    label = id_to_name[pred]\n\n    predicted_labels.append(label)\n\n# Add the predicted labels to the csv\ndf[\"label\"] = predicted_labels","metadata":{"execution":{"iopub.status.busy":"2023-12-07T04:12:15.196296Z","iopub.execute_input":"2023-12-07T04:12:15.196666Z","iopub.status.idle":"2023-12-07T04:12:16.505989Z","shell.execute_reply.started":"2023-12-07T04:12:15.196637Z","shell.execute_reply":"2023-12-07T04:12:16.505194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a new dataframe with image IDs and predicted labels\nsubmission_df = pd.DataFrame({'image_id': df['image_id'], 'label': predicted_labels})\n\n# Save to CSV\nsubmission_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-12-07T04:12:41.683206Z","iopub.execute_input":"2023-12-07T04:12:41.683567Z","iopub.status.idle":"2023-12-07T04:12:41.691788Z","shell.execute_reply.started":"2023-12-07T04:12:41.683539Z","shell.execute_reply":"2023-12-07T04:12:41.690882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df","metadata":{"execution":{"iopub.status.busy":"2023-12-07T04:12:43.668204Z","iopub.execute_input":"2023-12-07T04:12:43.669096Z","iopub.status.idle":"2023-12-07T04:12:43.677234Z","shell.execute_reply.started":"2023-12-07T04:12:43.66906Z","shell.execute_reply":"2023-12-07T04:12:43.67624Z"},"trusted":true},"execution_count":null,"outputs":[]}]}