{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":45867,"databundleVersionId":6924515},{"sourceType":"modelInstanceVersion","sourceId":6097,"databundleVersionId":7429358,"modelInstanceId":4634},{"sourceType":"kernelVersion","sourceId":155236951}],"dockerImageVersionId":30684,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-17T07:49:11.478775Z","iopub.execute_input":"2024-04-17T07:49:11.479187Z","iopub.status.idle":"2024-04-17T07:49:12.082994Z","shell.execute_reply.started":"2024-04-17T07:49:11.479156Z","shell.execute_reply":"2024-04-17T07:49:12.081867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install keras-core\n","metadata":{"execution":{"iopub.status.busy":"2024-04-17T07:49:12.085404Z","iopub.execute_input":"2024-04-17T07:49:12.085898Z","iopub.status.idle":"2024-04-17T07:49:30.360138Z","shell.execute_reply.started":"2024-04-17T07:49:12.085857Z","shell.execute_reply":"2024-04-17T07:49:30.358334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nos.environ[\"KERAS_BACKEND\"] = \"jax\" # or \"tensorflow\", \"torch\"\n\nimport cv2\nimport pickle\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Set the style for the plot\nsns.set(style=\"whitegrid\")\n\nimport tensorflow as tf\nimport keras_cv\nimport keras_core as keras\nfrom keras_core import ops","metadata":{"execution":{"iopub.status.busy":"2024-04-17T07:49:30.362234Z","iopub.execute_input":"2024-04-17T07:49:30.362642Z","iopub.status.idle":"2024-04-17T07:49:38.911427Z","shell.execute_reply.started":"2024-04-17T07:49:30.362607Z","shell.execute_reply":"2024-04-17T07:49:38.910274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Config:\n    is_submission = False\n    \n    # Reproducibility\n    SEED = 42\n    \n    # Training\n    train_csv_path = \"/kaggle/input/UBC-OCEAN/train.csv\"\n    train_thumbnail_paths = \"/kaggle/input/UBC-OCEAN/train_thumbnails\"\n    batch_size = 8\n    learning_rate = 1e-3\n    epochs = 2\n    \n    # Inference\n    test_csv_path = \"/kaggle/input/UBC-OCEAN/test.csv\"\n    test_thumbnail_paths = \"/kaggle/input/UBC-OCEAN/test_thumbnails\"\n\nconfig = Config()\n\n\n\n\n\n","metadata":{"execution":{"iopub.status.busy":"2024-04-17T07:49:38.917049Z","iopub.execute_input":"2024-04-17T07:49:38.917887Z","iopub.status.idle":"2024-04-17T07:49:38.929957Z","shell.execute_reply.started":"2024-04-17T07:49:38.917835Z","shell.execute_reply":"2024-04-17T07:49:38.927141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"keras.utils.set_random_seed(seed=config.SEED)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-17T07:49:38.931757Z","iopub.execute_input":"2024-04-17T07:49:38.932877Z","iopub.status.idle":"2024-04-17T07:49:39.012956Z","shell.execute_reply.started":"2024-04-17T07:49:38.932837Z","shell.execute_reply":"2024-04-17T07:49:39.011893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\nif not config.is_submission:\n    df = pd.read_csv(config.train_csv_path)\n\n    # Create the thumbnail df where is_tma == False\n    df = df[df[\"is_tma\"] == False]\n    \n    # Get basic statistics about the dataset\n    num_rows = df.shape[0]\n    num_unique_images = df['image_id'].nunique()\n    num_unique_labels = df['label'].nunique()\n    unique_labels = df['label'].unique()\n\n    print(f\"{num_rows=}\")\n    print(f\"{num_unique_images=}\")\n    print(f\"{num_unique_labels=}\")\n    print(f\"{unique_labels=}\")\n    \n    # Plot the distribution of the target classes\n    plt.figure(figsize=(10, 6))\n    sns.countplot(data=df, x='label', order=df['label'].value_counts().index)\n    plt.title('Distribution of Target Classes')\n    plt.xlabel('Label')\n    plt.ylabel('Count')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-17T07:49:39.014636Z","iopub.execute_input":"2024-04-17T07:49:39.015322Z","iopub.status.idle":"2024-04-17T07:49:39.445854Z","shell.execute_reply.started":"2024-04-17T07:49:39.015288Z","shell.execute_reply":"2024-04-17T07:49:39.444547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not config.is_submission:\n    # Perform one-hot encoding of the 'label' column and explicitly convert to integer type\n    df_one_hot = pd.get_dummies(df[\"label\"], prefix=\"label\").astype(int)\n\n    # Concatenate the original DataFrame with the one-hot encoded labels\n    train_df = pd.concat([df[\"image_id\"], df_one_hot], axis=1)\n\n    # Get the thumbnail image paths\n    train_df[\"image_thumbnail_path\"] = train_df[\"image_id\"].apply(lambda x: f\"{config.train_thumbnail_paths}/{x}_thumbnail.png\")\n    \n    image_thumbnail_paths = train_df[\"image_thumbnail_path\"].values\n    labels = train_df[[col for col in train_df.columns if col.startswith(\"label_\")]].values\n\n    label_names = [col for col in train_df.columns if col.startswith(\"label_\")]\n    name_to_id = {key.replace(\"label_\", \"\"):value for value,key in enumerate(label_names)}\n    id_to_name = {key:value for value, key in name_to_id.items()}\n    \n    # Save to dictionary to disk\n    with open(\"id_to_name.pkl\", \"wb\") as f:\n        pickle.dump(id_to_name, f)","metadata":{"execution":{"iopub.status.busy":"2024-04-17T07:49:39.447795Z","iopub.execute_input":"2024-04-17T07:49:39.449146Z","iopub.status.idle":"2024-04-17T07:49:39.465915Z","shell.execute_reply.started":"2024-04-17T07:49:39.449107Z","shell.execute_reply":"2024-04-17T07:49:39.464428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not config.is_submission:\n    class_weights = np.sum(labels) - np.sum(labels, axis=0)\n    class_weights = class_weights / np.sum(class_weights) # Normalize the weights\n\n    class_weights = {idx:weight for idx, weight in enumerate(class_weights)}\n\n    for idx, weight in class_weights.items():\n        print(f\"{id_to_name[idx]}: {weight:0.2f}\")","metadata":{"execution":{"iopub.status.busy":"2024-04-17T07:49:39.467403Z","iopub.execute_input":"2024-04-17T07:49:39.467866Z","iopub.status.idle":"2024-04-17T07:49:39.483263Z","shell.execute_reply.started":"2024-04-17T07:49:39.467834Z","shell.execute_reply":"2024-04-17T07:49:39.482229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_image(path):\n    file = tf.io.read_file(path)\n    image = tf.io.decode_png(file, 3)\n    image = tf.image.resize(image, (256, 256))\n    image = tf.image.per_image_standardization(image)\n    return image","metadata":{"execution":{"iopub.status.busy":"2024-04-17T07:49:39.48486Z","iopub.execute_input":"2024-04-17T07:49:39.48524Z","iopub.status.idle":"2024-04-17T07:49:39.495682Z","shell.execute_reply.started":"2024-04-17T07:49:39.485211Z","shell.execute_reply":"2024-04-17T07:49:39.494357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not config.is_submission:\n    x = (\n        tf.data.Dataset.from_tensor_slices(image_thumbnail_paths)\n        .map(read_image, num_parallel_calls=tf.data.AUTOTUNE)\n    )\n    y = tf.data.Dataset.from_tensor_slices(labels)\n\n    # Zip the x and y together\n    ds = tf.data.Dataset.zip((x, y))\n    \n    # Create the training and validation splits\n    val_ds = (\n        ds\n        .take(50)\n        .batch(config.batch_size)\n        .prefetch(tf.data.AUTOTUNE)\n    )\n    train_ds = (\n        ds\n        .skip(50)\n        .shuffle(config.batch_size * 10)\n        .batch(config.batch_size)\n        .prefetch(tf.data.AUTOTUNE)\n    )","metadata":{"execution":{"iopub.status.busy":"2024-04-17T07:49:39.500212Z","iopub.execute_input":"2024-04-17T07:49:39.5009Z","iopub.status.idle":"2024-04-17T07:49:39.696842Z","shell.execute_reply.started":"2024-04-17T07:49:39.500861Z","shell.execute_reply":"2024-04-17T07:49:39.695145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not config.is_submission:\n    images, labels = train_ds.take(1).get_single_element()\n\n    keras_cv.visualization.plot_image_gallery(\n        images,\n        value_range=(0, 1),\n        rows=2,\n        cols=2,\n    )","metadata":{"execution":{"iopub.status.busy":"2024-04-17T07:49:39.699012Z","iopub.execute_input":"2024-04-17T07:49:39.69952Z","iopub.status.idle":"2024-04-17T07:49:52.999915Z","shell.execute_reply.started":"2024-04-17T07:49:39.699477Z","shell.execute_reply":"2024-04-17T07:49:52.998672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install keras_cv","metadata":{"execution":{"iopub.status.busy":"2024-04-17T07:49:53.001819Z","iopub.execute_input":"2024-04-17T07:49:53.002264Z","iopub.status.idle":"2024-04-17T07:50:08.112981Z","shell.execute_reply.started":"2024-04-17T07:49:53.002223Z","shell.execute_reply":"2024-04-17T07:50:08.111246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow.keras as keras\n\n# Load the ResNet152V2 model with ImageNet weights\nresnet_backbone = keras.applications.ResNet152V2(include_top=False, weights='imagenet')\nresnet_backbone.trainable = False\n\nimage_inputs = resnet_backbone.input\nimage_embeddings = resnet_backbone(image_inputs)\nimage_embeddings = keras.layers.GlobalAveragePooling2D()(image_embeddings)\n\nx = keras.layers.BatchNormalization(epsilon=1e-05, momentum=0.1)(image_embeddings)\nx = keras.layers.Dense(units=1024, activation=\"relu\")(x)\nx = keras.layers.Dropout(0.1)(x)\nx = keras.layers.Dense(units=512, activation=\"relu\")(x)\nx = keras.layers.Dropout(0.1)(x)\nx = keras.layers.Dense(units=256, activation=\"relu\")(x)\noutputs = keras.layers.Dense(units=5, activation=\"softmax\")(x)\n\n# Build the model with the Functional API\nmodel = keras.Model(\n    inputs=image_inputs,\n    outputs=outputs,\n)\n\nmodel.summary()\n\n\n","metadata":{"execution":{"iopub.status.busy":"2024-04-17T07:50:08.115639Z","iopub.execute_input":"2024-04-17T07:50:08.116131Z","iopub.status.idle":"2024-04-17T07:50:15.078005Z","shell.execute_reply.started":"2024-04-17T07:50:08.116078Z","shell.execute_reply":"2024-04-17T07:50:15.076936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not config.is_submission:\n    model.compile(\n        optimizer=keras.optimizers.Adam(learning_rate=config.learning_rate),\n        loss=keras.losses.CategoricalCrossentropy(),\n        metrics=[\"accuracy\"],\n    )\n\n    history = model.fit(\n        train_ds,\n        epochs=config.epochs,\n        validation_data=val_ds,\n        class_weight=class_weights,\n    )\n    \n    model.save_weights(\"ucb_ocean_checkpoint.weights.h5\")","metadata":{"execution":{"iopub.status.busy":"2024-04-17T07:50:15.079154Z","iopub.execute_input":"2024-04-17T07:50:15.079536Z","iopub.status.idle":"2024-04-17T07:57:24.116926Z","shell.execute_reply.started":"2024-04-17T07:50:15.079505Z","shell.execute_reply":"2024-04-17T07:57:24.11598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if config.is_submission:\n    df = pd.read_csv(config.test_csv_path)\n    df[\"image_path\"] = df[\"image_id\"].apply(lambda x: f\"{config.test_thumbnail_paths}/{x}_thumbnail.png\")\n    \n    # Load the model weights\n    model.load_weights(\"/kaggle/input/kerascv-train-and-infer-on-thumbnails/ucb_ocean_checkpoint.weights.h5\")\n    \n    # Load the id to name dictionary\n    with open(\"/kaggle/input/kerascv-train-and-infer-on-thumbnails/id_to_name.pkl\", \"rb\") as f:\n        id_to_name = pickle.load(f)","metadata":{"execution":{"iopub.status.busy":"2024-04-17T07:57:24.119097Z","iopub.execute_input":"2024-04-17T07:57:24.120196Z","iopub.status.idle":"2024-04-17T07:57:24.1278Z","shell.execute_reply.started":"2024-04-17T07:57:24.120152Z","shell.execute_reply":"2024-04-17T07:57:24.126389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if config.is_submission:\n    predicted_labels = []\n\n    for index, row in df.iterrows():\n        # Get the image path\n        image_path = row[\"image_path\"]\n\n        # Get the image\n        image = read_image(image_path)[None, ...]\n\n        # Predict the label\n        logits = model.predict(image)\n        pred = ops.argmax(logits, axis=-1).tolist()[0]\n\n        # Map the pred to the name\n        label = id_to_name[pred]\n\n        predicted_labels.append(label)\n\n    # Add the predicted labels to the csv\n    df[\"label\"] = predicted_labels","metadata":{"execution":{"iopub.status.busy":"2024-04-17T07:57:24.129393Z","iopub.execute_input":"2024-04-17T07:57:24.129875Z","iopub.status.idle":"2024-04-17T07:57:24.144645Z","shell.execute_reply.started":"2024-04-17T07:57:24.12981Z","shell.execute_reply":"2024-04-17T07:57:24.143061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if config.is_submission:\n    # Create the submission\n    submission_df = df[[\"image_id\", \"label\"]]\n    submission_df.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2024-04-17T07:58:35.810078Z","iopub.execute_input":"2024-04-17T07:58:35.810823Z","iopub.status.idle":"2024-04-17T07:58:35.816798Z","shell.execute_reply.started":"2024-04-17T07:58:35.810788Z","shell.execute_reply":"2024-04-17T07:58:35.815488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}