{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"}],"dockerImageVersionId":30558,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-01-28T19:20:07.724492Z","iopub.execute_input":"2024-01-28T19:20:07.72489Z","iopub.status.idle":"2024-01-28T19:20:07.743638Z","shell.execute_reply.started":"2024-01-28T19:20:07.72486Z","shell.execute_reply":"2024-01-28T19:20:07.742661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nos.environ[\"KERAS_BACKEND\"] = \"jax\" # or \"tensorflow\", \"torch\"\n\nimport cv2\nimport pickle\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Set the style for the plot\nsns.set(style=\"whitegrid\")\n\nimport tensorflow as tf\nimport keras_cv\nimport keras_core as keras\nfrom keras_core import ops","metadata":{"execution":{"iopub.status.busy":"2024-01-28T19:20:07.745567Z","iopub.execute_input":"2024-01-28T19:20:07.746451Z","iopub.status.idle":"2024-01-28T19:20:07.751769Z","shell.execute_reply.started":"2024-01-28T19:20:07.746418Z","shell.execute_reply":"2024-01-28T19:20:07.751071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image =plt.imread('/kaggle/input/UBC-OCEAN/train_thumbnails/10642_thumbnail.png')\nplt.imshow(image)\nplt.axis('off')","metadata":{"execution":{"iopub.status.busy":"2024-01-28T19:20:07.752835Z","iopub.execute_input":"2024-01-28T19:20:07.753064Z","iopub.status.idle":"2024-01-28T19:20:09.08869Z","shell.execute_reply.started":"2024-01-28T19:20:07.753043Z","shell.execute_reply":"2024-01-28T19:20:09.087231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\ndf = pd.read_csv('/kaggle/input/UBC-OCEAN/train.csv')\n\n# Create the thumbnail df where is_tma == False\ndf = df[df[\"is_tma\"] == False]\n\n# Get basic statistics about the dataset\nnum_rows = df.shape[0]\nnum_unique_images = df['image_id'].nunique()\nnum_unique_labels = df['label'].nunique()\nunique_labels = df['label'].unique()\n\nprint(f\"{num_rows=}\")\nprint(f\"{num_unique_images=}\")\nprint(f\"{num_unique_labels=}\")\nprint(f\"{unique_labels=}\")\n\n# Plot the distribution of the target classes\nplt.figure(figsize=(10, 6))\nsns.countplot(data=df, x='label', order=df['label'].value_counts().index)\nplt.title('Distribution of Target Classes')\nplt.xlabel('Label')\nplt.ylabel('Count')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-01-28T19:20:09.091554Z","iopub.execute_input":"2024-01-28T19:20:09.09193Z","iopub.status.idle":"2024-01-28T19:20:09.309324Z","shell.execute_reply.started":"2024-01-28T19:20:09.091857Z","shell.execute_reply":"2024-01-28T19:20:09.308331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2024-01-28T19:20:09.310365Z","iopub.execute_input":"2024-01-28T19:20:09.310642Z","iopub.status.idle":"2024-01-28T19:20:09.324636Z","shell.execute_reply.started":"2024-01-28T19:20:09.310619Z","shell.execute_reply":"2024-01-28T19:20:09.323458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_image(path):\n    file = tf.io.read_file(path)\n    image = tf.io.decode_png(file, 3)\n    image = tf.image.resize(image, (256, 256))\n    image = tf.image.per_image_standardization(image)\n    return image","metadata":{"execution":{"iopub.status.busy":"2024-01-28T19:20:09.325744Z","iopub.execute_input":"2024-01-28T19:20:09.326143Z","iopub.status.idle":"2024-01-28T19:20:09.332359Z","shell.execute_reply.started":"2024-01-28T19:20:09.326004Z","shell.execute_reply":"2024-01-28T19:20:09.331338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Perform one-hot encoding of the 'label' column and explicitly convert to integer type\ndf_one_hot = pd.get_dummies(df[\"label\"],prefix=\"label\").astype(int)\n\n# Concatenate the original DataFrame with the one-hot encoded labels\ntrain_df = pd.concat([df[\"image_id\"], df_one_hot], axis=1)   ","metadata":{"execution":{"iopub.status.busy":"2024-01-28T19:20:09.333696Z","iopub.execute_input":"2024-01-28T19:20:09.334041Z","iopub.status.idle":"2024-01-28T19:20:09.342152Z","shell.execute_reply.started":"2024-01-28T19:20:09.33401Z","shell.execute_reply":"2024-01-28T19:20:09.341248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2024-01-28T19:20:09.343144Z","iopub.execute_input":"2024-01-28T19:20:09.343423Z","iopub.status.idle":"2024-01-28T19:20:09.357631Z","shell.execute_reply.started":"2024-01-28T19:20:09.3434Z","shell.execute_reply":"2024-01-28T19:20:09.356728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get the thumbnail image paths\ntrain_df[\"image_thumbnail_path\"] = train_df[\"image_id\"].apply(lambda x: f\"{'/kaggle/input/UBC-OCEAN/train_thumbnails'}/{x}_thumbnail.png\")\nimage_thumbnail_paths = train_df[\"image_thumbnail_path\"].values\nlabels = train_df[[col for col in train_df.columns if col.startswith(\"label_\")]].values\n\nlabel_names = [col for col in train_df.columns if col.startswith(\"label_\")]\nname_to_id = {key.replace(\"label_\", \"\"):value for value,key in enumerate(label_names)}\nid_to_name = {key:value for value, key in name_to_id.items()}\n    \n    # Save to dictionary to disk\nwith open(\"id_to_name.pkl\", \"wb\") as f:\n    pickle.dump(id_to_name, f)","metadata":{"execution":{"iopub.status.busy":"2024-01-28T19:20:09.360606Z","iopub.execute_input":"2024-01-28T19:20:09.36087Z","iopub.status.idle":"2024-01-28T19:20:09.370482Z","shell.execute_reply.started":"2024-01-28T19:20:09.360848Z","shell.execute_reply":"2024-01-28T19:20:09.369276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_thumbnail_paths","metadata":{"execution":{"iopub.status.busy":"2024-01-28T19:20:09.371815Z","iopub.execute_input":"2024-01-28T19:20:09.372608Z","iopub.status.idle":"2024-01-28T19:20:09.382867Z","shell.execute_reply.started":"2024-01-28T19:20:09.372561Z","shell.execute_reply":"2024-01-28T19:20:09.381965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels\n","metadata":{"execution":{"iopub.status.busy":"2024-01-28T19:20:09.384077Z","iopub.execute_input":"2024-01-28T19:20:09.384372Z","iopub.status.idle":"2024-01-28T19:20:09.393308Z","shell.execute_reply.started":"2024-01-28T19:20:09.384348Z","shell.execute_reply":"2024-01-28T19:20:09.392327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_names","metadata":{"execution":{"iopub.status.busy":"2024-01-28T19:20:09.394725Z","iopub.execute_input":"2024-01-28T19:20:09.39505Z","iopub.status.idle":"2024-01-28T19:20:09.40192Z","shell.execute_reply.started":"2024-01-28T19:20:09.395025Z","shell.execute_reply":"2024-01-28T19:20:09.40119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = (\n    tf.data.Dataset.from_tensor_slices(image_thumbnail_paths)\n    .map(read_and_augment_image, num_parallel_calls=tf.data.AUTOTUNE)\n)\n\ny = tf.data.Dataset.from_tensor_slices(labels)\n\n# Zip the x and y together\nds = tf.data.Dataset.zip((x, y))\n\n# Create the training and validation splits\ntrain_size = 366\nval_size = 76\ntest_size = 71\n\n# Create training dataset\ntrain_ds = (\n    ds\n    .take(train_size)\n    .shuffle(8 * 10)\n    .batch(5)\n    .prefetch(tf.data.AUTOTUNE)\n)\n\n# Create validation dataset\nval_ds = (\n    ds\n    .skip(train_size)\n    .take(val_size)\n    .batch(5)\n    .prefetch(tf.data.AUTOTUNE)\n)\n\n# Create test dataset\ntest_ds = (\n    ds\n    .skip(train_size + val_size)\n    .take(test_size)\n    .batch(5)\n    .prefetch(tf.data.AUTOTUNE)\n)\n","metadata":{"execution":{"iopub.status.busy":"2024-01-28T19:20:09.403048Z","iopub.execute_input":"2024-01-28T19:20:09.403332Z","iopub.status.idle":"2024-01-28T19:20:09.480139Z","shell.execute_reply.started":"2024-01-28T19:20:09.403307Z","shell.execute_reply":"2024-01-28T19:20:09.479265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_image(path):\n    file = tf.io.read_file(path)\n    image = tf.io.decode_png(file, 3)\n    image = tf.image.resize(image, (256, 256))\n    \n    # Apply per_image_standardization for normalization\n    image = tf.image.per_image_standardization(image)\n    \n    return image\n# Add data augmentation to the image loading process\ndef read_and_augment_image(path):\n    file = tf.io.read_file(path)\n    image = tf.io.decode_png(file, 3)\n    image = tf.image.resize(image, (256, 256))\n    \n    # Data augmentation\n    image = tf.image.random_flip_left_right(image)\n    image = tf.image.random_flip_up_down(image)\n    image = tf.image.random_brightness(image, max_delta=0.2)\n    image = tf.image.random_contrast(image, lower=0.8, upper=1.2)\n    \n    # Apply per_image_standardization for normalization\n    image = tf.image.per_image_standardization(image)\n    \n    return image\n\n\n# ... (rest of your code)\n\n# Zip the x and y together\nds = tf.data.Dataset.zip((x, y))\n\n# Create the training and validation splits\ntrain_size = 366\nval_size = 76\ntest_size = 71\n\n# Create training dataset\ntrain_ds = (\n    ds\n    .take(train_size)\n    .shuffle(8 * 10)\n    .batch(5)\n    .prefetch(tf.data.AUTOTUNE)\n)\n\n# Create validation dataset\nval_ds = (\n    ds\n    .skip(train_size)\n    .take(val_size)\n    .batch(5)\n    .prefetch(tf.data.AUTOTUNE)\n)\n\n# Create test dataset\ntest_ds = (\n    ds\n    .skip(train_size + val_size)\n    .take(test_size)\n    .batch(5)\n    .prefetch(tf.data.AUTOTUNE)\n)\n","metadata":{"execution":{"iopub.status.busy":"2024-01-28T19:20:09.482003Z","iopub.execute_input":"2024-01-28T19:20:09.482499Z","iopub.status.idle":"2024-01-28T19:20:09.497351Z","shell.execute_reply.started":"2024-01-28T19:20:09.482463Z","shell.execute_reply":"2024-01-28T19:20:09.496254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras import layers, models, optimizers\n\n# Define the CNN model\nmodel = models.Sequential()\n\n# Convolutional layers\nmodel.add(layers.Conv2D(32, (3, 3), activation='relu', input_shape=(256, 256, 3)))\nmodel.add(layers.MaxPooling2D((2, 2)))\nmodel.add(layers.Conv2D(64, (3, 3), activation='relu'))\nmodel.add(layers.MaxPooling2D((2, 2)))\nmodel.add(layers.Conv2D(128, (3, 3), activation='relu'))\nmodel.add(layers.MaxPooling2D((2, 2)))\n# Add more convolutional layers\nmodel.add(layers.Conv2D(256, (3, 3), activation='relu'))\nmodel.add(layers.MaxPooling2D((2, 2)))\n# Add dropout layer\nmodel.add(layers.Dropout(0.5))\n\n# ... Continue with additional layers\n\n# Flatten layer\nmodel.add(layers.Flatten())\n\n# Dense layers\nmodel.add(layers.Dense(128, activation='relu'))\nmodel.add(layers.Dropout(0.5))  # Optional dropout layer for regularization\nmodel.add(layers.Dense(len(label_names), activation='softmax'))\n\n# Adjust the learning rate\ncustom_optimizer = optimizers.Adam(learning_rate=0.0001)  # Experiment with different values\nmodel.compile(optimizer=custom_optimizer, loss='categorical_crossentropy', metrics=['accuracy'])\n\n# Display the model summary\nmodel.summary()\n\n","metadata":{"execution":{"iopub.status.busy":"2024-01-28T19:20:09.49859Z","iopub.execute_input":"2024-01-28T19:20:09.498862Z","iopub.status.idle":"2024-01-28T19:20:09.649148Z","shell.execute_reply.started":"2024-01-28T19:20:09.498833Z","shell.execute_reply":"2024-01-28T19:20:09.648104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train the model\nhistory = model.fit(train_ds, epochs=10, validation_data=val_ds)\n","metadata":{"execution":{"iopub.status.busy":"2024-01-28T19:20:09.650646Z","iopub.execute_input":"2024-01-28T19:20:09.650995Z","iopub.status.idle":"2024-01-28T19:32:39.872123Z","shell.execute_reply.started":"2024-01-28T19:20:09.650964Z","shell.execute_reply":"2024-01-28T19:32:39.870737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming you have already trained your model and stored it in the 'model' variable\n\n# Evaluate the model on the test dataset\ntest_results = model.evaluate(test_ds)\n\n# Print the test loss and accuracy\ntest_loss, test_accuracy = test_results\nprint(f'Test Loss: {test_loss:.4f}')\nprint(f'Test Accuracy: {test_accuracy:.4f}')\n\n# Assuming you have an image you want to predict on\nnew_image_path = '/kaggle/input/UBC-OCEAN/train_thumbnails/10077_thumbnail.png'\n\n\n# Read and preprocess the new image\nnew_image = read_and_augment_image(new_image_path)\nnew_image = tf.expand_dims(new_image, axis=0)  # Add batch dimension\n\n# Make predictions\npredictions = model.predict(new_image)\n\n# Get the predicted class index\npredicted_class_index = np.argmax(predictions)\n\n# Map the predicted class index to the corresponding label\npredicted_label = label_names[predicted_class_index]\n\nprint(f'Predicted Label: {predicted_label}')\n","metadata":{"execution":{"iopub.status.busy":"2024-01-28T19:32:39.874051Z","iopub.execute_input":"2024-01-28T19:32:39.874402Z","iopub.status.idle":"2024-01-28T19:33:05.495135Z","shell.execute_reply.started":"2024-01-28T19:32:39.874372Z","shell.execute_reply":"2024-01-28T19:33:05.494258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.applications import VGG16\nfrom tensorflow.keras.layers import Dense, Flatten\nfrom tensorflow.keras.models import Model\n\n# Load the pre-trained VGG16 model without the top (fully connected) layers\nbase_model = VGG16(weights='imagenet', include_top=False, input_shape=(256, 256, 3))\n\n# Freeze the pre-trained layers\nfor layer in base_model.layers:\n    layer.trainable = False\n\n# Create a new model with additional layers on top of the pre-trained model\nmodel3 = Flatten()(base_model.output)\nmodel3 = Dense(4096, activation='relu')(model3)\nmodel3 = Dense(4096, activation='relu')(model3)\noutput = Dense(len(label_names), activation='softmax')(model3)\n\n# Final model\nfinal_model= Model(inputs=base_model.input, outputs=output)\n\n# Compile the model\nfinal_model.compile(optimizer='adam', loss='categorical_crossentropy', metrics=['accuracy'])\n\n# Print the model summary\nfinal_model.summary()\n","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:24:13.758381Z","iopub.execute_input":"2024-01-28T23:24:13.758826Z","iopub.status.idle":"2024-01-28T23:24:29.376049Z","shell.execute_reply.started":"2024-01-28T23:24:13.758792Z","shell.execute_reply":"2024-01-28T23:24:29.374297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history2 = final_model.fit(train_ds, validation_data=val_ds, epochs=10)\n","metadata":{"execution":{"iopub.status.busy":"2024-01-28T19:41:15.495894Z","iopub.execute_input":"2024-01-28T19:41:15.496253Z","iopub.status.idle":"2024-01-28T20:14:38.377246Z","shell.execute_reply.started":"2024-01-28T19:41:15.496226Z","shell.execute_reply":"2024-01-28T20:14:38.376551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming you have already trained your model and stored it in the 'model' variable\n\n# Evaluate the model on the test dataset\ntest_vgg_results = final_model.evaluate(test_ds)\n\n# Print the test loss and accuracy\ntest_vgg_loss, test_vgg_accuracy = test_vgg_results\nprint(f'Test Loss: {test_vgg_loss:.4f}')\nprint(f'Test Accuracy: {test_vgg_accuracy:.4f}')\n","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:24:01.774583Z","iopub.execute_input":"2024-01-28T23:24:01.77502Z","iopub.status.idle":"2024-01-28T23:24:02.110721Z","shell.execute_reply.started":"2024-01-28T23:24:01.774987Z","shell.execute_reply":"2024-01-28T23:24:02.108765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming you have an image you want to predict on\nnew_image_path = '/kaggle/input/UBC-OCEAN/train_thumbnails/10077_thumbnail.png'\n\n# Read and preprocess the new image\nnew_image = read_image(new_image_path)\nnew_image = tf.expand_dims(new_image, axis=0)  # Add batch dimension\n\n# Make predictions\npredictions_vgg = final_model.predict(new_image)\n\n# Get the predicted class index\npredicted_class_index2 = np.argmax(predictions_vgg)\n\n# Map the predicted class index to the corresponding label\npredicted_label2 = label_names[predicted_class_index2]\n\nprint(f'Predicted Label: {predicted_label2}')","metadata":{"execution":{"iopub.status.busy":"2024-01-28T20:15:43.608588Z","iopub.execute_input":"2024-01-28T20:15:43.608932Z","iopub.status.idle":"2024-01-28T20:15:44.232853Z","shell.execute_reply.started":"2024-01-28T20:15:43.608906Z","shell.execute_reply":"2024-01-28T20:15:44.232148Z"},"trusted":true},"execution_count":null,"outputs":[]}]}