{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-11-02T03:31:03.572468Z","iopub.execute_input":"2023-11-02T03:31:03.573783Z","iopub.status.idle":"2023-11-02T03:31:03.691209Z","shell.execute_reply.started":"2023-11-02T03:31:03.573743Z","shell.execute_reply":"2023-11-02T03:31:03.689815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**LOADING LIBRARIES**","metadata":{}},{"cell_type":"code","source":"import os\nos.environ[\"KERAS_BACKEND\"] = \"jax\" # or \"tensorflow\", \"torch\"\n\nimport cv2\nimport pickle\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Set the style for the plot\nsns.set(style=\"whitegrid\")\n\nimport tensorflow as tf\nimport keras_cv\nimport keras_core as keras\nfrom keras_core import ops","metadata":{"execution":{"iopub.status.busy":"2023-11-02T03:31:03.693966Z","iopub.execute_input":"2023-11-02T03:31:03.694788Z","iopub.status.idle":"2023-11-02T03:31:03.702424Z","shell.execute_reply.started":"2023-11-02T03:31:03.694741Z","shell.execute_reply":"2023-11-02T03:31:03.701027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**CONFIGURATION**","metadata":{}},{"cell_type":"code","source":"class Config:\n    is_submission = False\n    \n    # Reproducibility\n    SEED = 42\n    \n    # Training\n    train_csv_path = \"/kaggle/input/UBC-OCEAN/train.csv\"\n    train_thumbnail_paths = \"/kaggle/input/UBC-OCEAN/train_thumbnails\"\n    batch_size = 8\n    learning_rate = 1e-3\n    epochs = 2\n    \n    # Inference\n    test_csv_path = \"/kaggle/input/UBC-OCEAN/test.csv\"\n    test_thumbnail_paths = \"/kaggle/input/UBC-OCEAN/test_thumbnails\"\n\nconfig = Config()","metadata":{"execution":{"iopub.status.busy":"2023-11-02T03:31:03.70409Z","iopub.execute_input":"2023-11-02T03:31:03.704903Z","iopub.status.idle":"2023-11-02T03:31:03.716161Z","shell.execute_reply.started":"2023-11-02T03:31:03.70487Z","shell.execute_reply":"2023-11-02T03:31:03.715096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"keras.utils.set_random_seed(seed=config.SEED)","metadata":{"execution":{"iopub.status.busy":"2023-11-02T03:31:03.717688Z","iopub.execute_input":"2023-11-02T03:31:03.718555Z","iopub.status.idle":"2023-11-02T03:31:03.730861Z","shell.execute_reply.started":"2023-11-02T03:31:03.718513Z","shell.execute_reply":"2023-11-02T03:31:03.729733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**TRAINING OF DATASET**","metadata":{}},{"cell_type":"code","source":"if not config.is_submission:\n    df = pd.read_csv(config.train_csv_path)\n\n    # Create the thumbnail df where is_tma == False\n    df = df[df[\"is_tma\"] == False]\n    \n    # Get basic statistics about the dataset\n    num_rows = df.shape[0]\n    num_unique_images = df['image_id'].nunique()\n    num_unique_labels = df['label'].nunique()\n    unique_labels = df['label'].unique()\n\n    print(f\"{num_rows=}\")\n    print(f\"{num_unique_images=}\")\n    print(f\"{num_unique_labels=}\")\n    print(f\"{unique_labels=}\")\n    \n    # Plot the distribution of the target classes\n    plt.figure(figsize=(10, 6))\n    sns.countplot(data=df, x='label', order=df['label'].value_counts().index)\n    plt.title('Distribution of Target Classes')\n    plt.xlabel('Label')\n    plt.ylabel('Count')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-02T03:31:03.734188Z","iopub.execute_input":"2023-11-02T03:31:03.735238Z","iopub.status.idle":"2023-11-02T03:31:04.074281Z","shell.execute_reply.started":"2023-11-02T03:31:03.735199Z","shell.execute_reply":"2023-11-02T03:31:04.07311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**HOT ENCODING**","metadata":{}},{"cell_type":"code","source":"if not config.is_submission:\n    # Perform one-hot encoding of the 'label' column and explicitly convert to integer type\n    df_one_hot = pd.get_dummies(df[\"label\"], prefix=\"label\").astype(int)\n\n    # Concatenate the original DataFrame with the one-hot encoded labels\n    train_df = pd.concat([df[\"image_id\"], df_one_hot], axis=1)\n\n    # Get the thumbnail image paths\n    train_df[\"image_thumbnail_path\"] = train_df[\"image_id\"].apply(lambda x: f\"{config.train_thumbnail_paths}/{x}_thumbnail.png\")\n    \n    image_thumbnail_paths = train_df[\"image_thumbnail_path\"].values\n    labels = train_df[[col for col in train_df.columns if col.startswith(\"label_\")]].values\n\n    label_names = [col for col in train_df.columns if col.startswith(\"label_\")]\n    name_to_id = {key.replace(\"label_\", \"\"):value for value,key in enumerate(label_names)}\n    id_to_name = {key:value for value, key in name_to_id.items()}\n    \n    # Save to dictionary to disk\n    with open(\"id_to_name.pkl\", \"wb\") as f:\n        pickle.dump(id_to_name, f)","metadata":{"execution":{"iopub.status.busy":"2023-11-02T03:31:04.075749Z","iopub.execute_input":"2023-11-02T03:31:04.076169Z","iopub.status.idle":"2023-11-02T03:31:04.091526Z","shell.execute_reply.started":"2023-11-02T03:31:04.076122Z","shell.execute_reply":"2023-11-02T03:31:04.090122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not config.is_submission:\n    class_weights = np.sum(labels) - np.sum(labels, axis=0)\n    class_weights = class_weights / np.sum(class_weights) # Normalize the weights\n\n    class_weights = {idx:weight for idx, weight in enumerate(class_weights)}\n\n    for idx, weight in class_weights.items():\n        print(f\"{id_to_name[idx]}: {weight:0.2f}\")","metadata":{"execution":{"iopub.status.busy":"2023-11-02T03:31:04.093391Z","iopub.execute_input":"2023-11-02T03:31:04.093773Z","iopub.status.idle":"2023-11-02T03:31:04.101417Z","shell.execute_reply.started":"2023-11-02T03:31:04.093741Z","shell.execute_reply":"2023-11-02T03:31:04.10049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_image(path):\n    file = tf.io.read_file(path)\n    image = tf.io.decode_png(file, 3)\n    image = tf.image.resize(image, (256, 256))\n    image = tf.image.per_image_standardization(image)\n    return image","metadata":{"execution":{"iopub.status.busy":"2023-11-02T03:31:04.102693Z","iopub.execute_input":"2023-11-02T03:31:04.103237Z","iopub.status.idle":"2023-11-02T03:31:04.118199Z","shell.execute_reply.started":"2023-11-02T03:31:04.103208Z","shell.execute_reply":"2023-11-02T03:31:04.116895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not config.is_submission:\n    x = (\n        tf.data.Dataset.from_tensor_slices(image_thumbnail_paths)\n        .map(read_image, num_parallel_calls=tf.data.AUTOTUNE)\n    )\n    y = tf.data.Dataset.from_tensor_slices(labels)\n\n    # Zip the x and y together\n    ds = tf.data.Dataset.zip((x, y))\n    \n    # Create the training and validation splits\n    val_ds = (\n        ds\n        .take(50)\n        .batch(config.batch_size)\n        .prefetch(tf.data.AUTOTUNE)\n    )\n    train_ds = (\n        ds\n        .skip(50)\n        .shuffle(config.batch_size * 10)\n        .batch(config.batch_size)\n        .prefetch(tf.data.AUTOTUNE)\n    )","metadata":{"execution":{"iopub.status.busy":"2023-11-02T03:31:04.11981Z","iopub.execute_input":"2023-11-02T03:31:04.120483Z","iopub.status.idle":"2023-11-02T03:31:04.394094Z","shell.execute_reply.started":"2023-11-02T03:31:04.120441Z","shell.execute_reply":"2023-11-02T03:31:04.392867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not config.is_submission:\n    images, labels = train_ds.take(1).get_single_element()\n\n    keras_cv.visualization.plot_image_gallery(\n        images,\n        value_range=(0, 1),\n        rows=2,\n        cols=2,\n    )","metadata":{"execution":{"iopub.status.busy":"2023-11-02T03:31:04.395969Z","iopub.execute_input":"2023-11-02T03:31:04.396358Z","iopub.status.idle":"2023-11-02T03:31:17.336385Z","shell.execute_reply.started":"2023-11-02T03:31:04.396301Z","shell.execute_reply":"2023-11-02T03:31:17.335189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**BUILDING MODEL**","metadata":{}},{"cell_type":"code","source":"resnet_backbone = keras_cv.models.ResNetV2Backbone.from_preset(\n    \"resnet152_v2\",\n)\nresnet_backbone.trainable = False\n\nimage_inputs = resnet_backbone.input\nimage_embeddings = resnet_backbone(image_inputs)\nimage_embeddings = keras.layers.GlobalAveragePooling2D()(image_embeddings)\n\nx = keras.layers.BatchNormalization(epsilon=1e-05, momentum=0.1)(image_embeddings)\nx = keras.layers.Dense(units=1024, activation=\"relu\")(x)\nx = keras.layers.Dropout(0.1)(x)\nx = keras.layers.Dense(units=512, activation=\"relu\")(x)\nx = keras.layers.Dropout(0.1)(x)\nx = keras.layers.Dense(units=256, activation=\"relu\")(x)\noutputs = keras.layers.Dense(units=5, activation=\"softmax\")(x)\n\n# Build the model with the Functional API\nmodel = keras.Model(\n    inputs=image_inputs,\n    outputs=outputs,\n)\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-11-02T03:31:17.33816Z","iopub.execute_input":"2023-11-02T03:31:17.338876Z","iopub.status.idle":"2023-11-02T03:31:26.51388Z","shell.execute_reply.started":"2023-11-02T03:31:17.338836Z","shell.execute_reply":"2023-11-02T03:31:26.512311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not config.is_submission:\n    model.compile(\n        optimizer=keras.optimizers.Adam(learning_rate=config.learning_rate),\n        loss=keras.losses.CategoricalCrossentropy(),\n        metrics=[\"accuracy\"],\n    )\n\n    history = model.fit(\n        train_ds,\n        epochs=config.epochs,\n        validation_data=val_ds,\n        class_weight=class_weights,\n    )\n    \n    model.save_weights(\"ucb_ocean_checkpoint.weights.h5\")","metadata":{"execution":{"iopub.status.busy":"2023-11-02T03:31:26.515747Z","iopub.execute_input":"2023-11-02T03:31:26.516186Z","iopub.status.idle":"2023-11-02T03:52:25.624517Z","shell.execute_reply.started":"2023-11-02T03:31:26.516146Z","shell.execute_reply":"2023-11-02T03:52:25.623563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**INFERENCE**","metadata":{}},{"cell_type":"code","source":"if config.is_submission:\n    df = pd.read_csv(config.test_csv_path)\n    df[\"image_path\"] = df[\"image_id\"].apply(lambda x: f\"{config.test_thumbnail_paths}/{x}_thumbnail.png\")\n    \n    # Load the model weights\n    model.load_weights(\"/kaggle/input/kerascv-train-and-infer-on-thumbnails/ucb_ocean_checkpoint.weights.h5\")\n    \n    # Load the id to name dictionary\n    with open(\"/kaggle/input/kerascv-train-and-infer-on-thumbnails/id_to_name.pkl\", \"rb\") as f:\n        id_to_name = pickle.load(f)","metadata":{"execution":{"iopub.status.busy":"2023-11-02T03:52:25.626768Z","iopub.execute_input":"2023-11-02T03:52:25.627055Z","iopub.status.idle":"2023-11-02T03:52:25.632943Z","shell.execute_reply.started":"2023-11-02T03:52:25.62703Z","shell.execute_reply":"2023-11-02T03:52:25.631715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if config.is_submission:\n    predicted_labels = []\n\n    for index, row in df.iterrows():\n        # Get the image path\n        image_path = row[\"image_path\"]\n\n        # Get the image\n        image = read_image(image_path)[None, ...]\n\n        # Predict the label\n        logits = model.predict(image)\n        pred = ops.argmax(logits, axis=-1).tolist()[0]\n\n        # Map the pred to the name\n        label = id_to_name[pred]\n\n        predicted_labels.append(label)\n\n    # Add the predicted labels to the csv\n    df[\"label\"] = predicted_labels","metadata":{"execution":{"iopub.status.busy":"2023-11-02T03:52:25.636874Z","iopub.execute_input":"2023-11-02T03:52:25.637169Z","iopub.status.idle":"2023-11-02T03:52:25.647118Z","shell.execute_reply.started":"2023-11-02T03:52:25.637131Z","shell.execute_reply":"2023-11-02T03:52:25.646265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if config.is_submission:\n    # Create the submission\n    submission_csv = df[[\"image_id\", \"label\"]]\n    submission_csv.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-11-02T04:01:38.022499Z","iopub.execute_input":"2023-11-02T04:01:38.022892Z","iopub.status.idle":"2023-11-02T04:01:38.028587Z","shell.execute_reply.started":"2023-11-02T04:01:38.022863Z","shell.execute_reply":"2023-11-02T04:01:38.027581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}