{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"},{"sourceId":5830,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":4603},{"sourceId":5861,"sourceType":"modelInstanceVersion","modelInstanceId":4634}],"dockerImageVersionId":30626,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport cv2\nimport os\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport tensorflow as tf\nimport random\nimport math\nimport PIL\nimport PIL.Image as Image\nimport pickle\nimport keras_cv\nimport keras_core as keras\nfrom keras_core import ops\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-01-04T06:49:38.606858Z","iopub.execute_input":"2024-01-04T06:49:38.607313Z","iopub.status.idle":"2024-01-04T06:50:02.14203Z","shell.execute_reply.started":"2024-01-04T06:49:38.607278Z","shell.execute_reply":"2024-01-04T06:50:02.140922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.random.seed(42)\nrandom.seed(42)\nSEED = 42\nkeras.utils.set_random_seed(seed=42)","metadata":{"execution":{"iopub.status.busy":"2024-01-04T06:50:35.775419Z","iopub.execute_input":"2024-01-04T06:50:35.776546Z","iopub.status.idle":"2024-01-04T06:50:35.78574Z","shell.execute_reply.started":"2024-01-04T06:50:35.776492Z","shell.execute_reply":"2024-01-04T06:50:35.783148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROOT_DIR = '/kaggle/input/UBC-OCEAN'\nTRAIN_DIR='/kaggle/input/UBC-OCEAN/train_thumbnails'\nALT_TRAIN_DIR='/kaggle/input/UBC-OCEAN/train_images'\nTEST_DIR = '/kaggle/input/UBC-OCEAN/test_thumbnails'\nALT_TEST_DIR = '/kaggle/input/UBC-OCEAN/test_images'","metadata":{"execution":{"iopub.status.busy":"2024-01-04T06:50:36.382571Z","iopub.execute_input":"2024-01-04T06:50:36.382985Z","iopub.status.idle":"2024-01-04T06:50:36.389232Z","shell.execute_reply.started":"2024-01-04T06:50:36.382953Z","shell.execute_reply":"2024-01-04T06:50:36.388079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_test_file_path(image_id):\n    if os.path.exists(f\"{TEST_DIR}/{image_id}_thumbnail.png\"):\n        return f\"{TEST_DIR}/{image_id}_thumbnail.png\"\n    else:\n        return f\"{ALT_TEST_DIR}/{image_id}.png\"","metadata":{"execution":{"iopub.status.busy":"2024-01-04T06:50:52.827172Z","iopub.execute_input":"2024-01-04T06:50:52.827685Z","iopub.status.idle":"2024-01-04T06:50:52.835866Z","shell.execute_reply.started":"2024-01-04T06:50:52.827646Z","shell.execute_reply":"2024-01-04T06:50:52.83344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_train_file_path(image_id):\n    if os.path.exists(f\"{TRAIN_DIR}/{image_id}_thumbnail.png\"):\n        return f\"{TRAIN_DIR}/{image_id}_thumbnail.png\"\n    else:\n        return f\"{ALT_TRAIN_DIR}/{image_id}.png\"","metadata":{"execution":{"iopub.status.busy":"2024-01-04T06:50:53.568825Z","iopub.execute_input":"2024-01-04T06:50:53.570131Z","iopub.status.idle":"2024-01-04T06:50:53.578312Z","shell.execute_reply.started":"2024-01-04T06:50:53.570079Z","shell.execute_reply":"2024-01-04T06:50:53.576505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df=pd.read_csv(f'{ROOT_DIR}/train.csv')\ntrain_df['file_path'] = train_df['image_id'].apply(get_train_file_path)\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2024-01-04T06:50:54.803607Z","iopub.execute_input":"2024-01-04T06:50:54.804046Z","iopub.status.idle":"2024-01-04T06:50:55.225714Z","shell.execute_reply.started":"2024-01-04T06:50:54.804016Z","shell.execute_reply":"2024-01-04T06:50:55.224609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df=pd.read_csv(f'{ROOT_DIR}/test.csv')\ntest_df['file_path'] = test_df['image_id'].apply(get_test_file_path)\ntest_df","metadata":{"execution":{"iopub.status.busy":"2024-01-04T06:50:56.166992Z","iopub.execute_input":"2024-01-04T06:50:56.167458Z","iopub.status.idle":"2024-01-04T06:50:56.1899Z","shell.execute_reply.started":"2024-01-04T06:50:56.167407Z","shell.execute_reply":"2024-01-04T06:50:56.188407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_cropped_images(file_path, image_id, th_area = 1000):\n    image = Image.open(file_path)\n    # Aspect ratio\n    as_ratio = image.size[0] / image.size[1]\n    \n    sxs, exs, sys, eys = [],[],[],[]\n    if as_ratio >= 1.5:\n        # Crop\n        mask = np.max( np.array(image) > 0, axis=-1 ).astype(np.uint8)\n        retval, labels = cv2.connectedComponents(mask)\n        if retval >= as_ratio:\n            x, y = np.meshgrid( np.arange(image.size[0]), np.arange(image.size[1]) )\n            for label in range(1, retval):\n                area = np.sum(labels == label)\n                if area < th_area:\n                    continue\n                xs, ys= x[ labels == label ], y[ labels == label ]\n                sx, ex = np.min(xs), np.max(xs)\n                cx = (sx + ex) // 2\n                crop_size = image.size[1]\n                sx = max(0, cx-crop_size//2)\n                ex = min(sx + crop_size - 1, image.size[0]-1)\n                sx = ex - crop_size + 1\n                sy, ey = 0, image.size[1]-1\n                sxs.append(sx)\n                exs.append(ex)\n                sys.append(sy)\n                eys.append(ey)\n        else:\n            crop_size = image.size[1]\n            for i in range(int(as_ratio)):\n                sxs.append( i * crop_size )\n                exs.append( (i+1) * crop_size - 1 )\n                sys.append( 0 )\n                eys.append( crop_size - 1 )\n    else:\n        # Not Crop (entire image)\n        sxs, exs, sys, eys = [0,],[image.size[0]-1],[0,],[image.size[1]-1]\n\n    df_crop = pd.DataFrame()\n    df_crop[\"image_id\"] = [image_id] * len(sxs)\n    df_crop[\"file_path\"] = [file_path] * len(sxs)\n    df_crop[\"sx\"] = sxs\n    df_crop[\"ex\"] = exs\n    df_crop[\"sy\"] = sys\n    df_crop[\"ey\"] = eys\n    return df_crop","metadata":{"execution":{"iopub.status.busy":"2024-01-04T06:50:56.919181Z","iopub.execute_input":"2024-01-04T06:50:56.919772Z","iopub.status.idle":"2024-01-04T06:50:56.941741Z","shell.execute_reply.started":"2024-01-04T06:50:56.919726Z","shell.execute_reply":"2024-01-04T06:50:56.940209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dfs = []\nfor (file_path, image_id) in zip(train_df[\"file_path\"], train_df[\"image_id\"]):\n    train_dfs.append( get_cropped_images(file_path, image_id) )\n\ntraindf_crop = pd.concat(train_dfs)\n","metadata":{"execution":{"iopub.status.busy":"2024-01-04T06:50:57.749873Z","iopub.execute_input":"2024-01-04T06:50:57.750293Z","iopub.status.idle":"2024-01-04T06:53:28.256557Z","shell.execute_reply.started":"2024-01-04T06:50:57.75026Z","shell.execute_reply":"2024-01-04T06:53:28.255122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"traindf_crop = traindf_crop.drop_duplicates(subset=[\"image_id\", \"sx\", \"ex\", \"sy\", \"ey\"]).reset_index(drop=True)\n# traindf_crop","metadata":{"execution":{"iopub.status.busy":"2024-01-04T06:53:28.258985Z","iopub.execute_input":"2024-01-04T06:53:28.25949Z","iopub.status.idle":"2024-01-04T06:53:28.276665Z","shell.execute_reply.started":"2024-01-04T06:53:28.259444Z","shell.execute_reply":"2024-01-04T06:53:28.27531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dfs = []\nfor (file_path, image_id) in zip(test_df[\"file_path\"], test_df[\"image_id\"]):\n    test_dfs.append( get_cropped_images(file_path, image_id))\n\ntestdf_crop = pd.concat(test_dfs)\n","metadata":{"execution":{"iopub.status.busy":"2024-01-04T06:53:28.278381Z","iopub.execute_input":"2024-01-04T06:53:28.2792Z","iopub.status.idle":"2024-01-04T06:53:28.883664Z","shell.execute_reply.started":"2024-01-04T06:53:28.279155Z","shell.execute_reply":"2024-01-04T06:53:28.882651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testdf_crop = testdf_crop.drop_duplicates(subset=[\"image_id\", \"sx\", \"ex\", \"sy\", \"ey\"]).reset_index(drop=True)\ntestdf_crop","metadata":{"execution":{"iopub.status.busy":"2024-01-04T06:53:28.886656Z","iopub.execute_input":"2024-01-04T06:53:28.887171Z","iopub.status.idle":"2024-01-04T06:53:28.905112Z","shell.execute_reply.started":"2024-01-04T06:53:28.887126Z","shell.execute_reply":"2024-01-04T06:53:28.903772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_image(filepath,sx,ex,sy,ey):\n\n    file = tf.io.read_file(filepath)\n    image = tf.io.decode_png(file, 3)\n    image = image[ sy:ey, sx:ex, : ]\n    image = tf.image.resize(image, (256, 256))\n    image = tf.image.per_image_standardization(image)\n    return image","metadata":{"execution":{"iopub.status.busy":"2024-01-04T06:53:28.906633Z","iopub.execute_input":"2024-01-04T06:53:28.907008Z","iopub.status.idle":"2024-01-04T06:53:28.914172Z","shell.execute_reply.started":"2024-01-04T06:53:28.906976Z","shell.execute_reply":"2024-01-04T06:53:28.912832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n    # Perform one-hot encoding of the 'label' column and explicitly convert to integer type\ndf_one_hot = pd.get_dummies(train_df[\"label\"], prefix=\"label\").astype(int)\n\n    # Concatenate the original DataFrame with the one-hot encoded labels\ntrain_df = pd.concat([train_df, df_one_hot], axis=1)\n\n    # Get the thumbnail image paths\n    \nlabels = df_one_hot.values\n\nlabel_names = [col for col in train_df.columns if col.startswith(\"label_\")]\nname_to_id = {key.replace(\"label_\", \"\"):value for value,key in enumerate(label_names)}\nid_to_name = {key:value for value, key in name_to_id.items()}\n    \n    # Save to dictionary to disk\nwith open(\"id_to_name.pkl\", \"wb\") as f:\n        pickle.dump(id_to_name, f)","metadata":{"execution":{"iopub.status.busy":"2024-01-04T06:53:28.915634Z","iopub.execute_input":"2024-01-04T06:53:28.915959Z","iopub.status.idle":"2024-01-04T06:53:28.932136Z","shell.execute_reply.started":"2024-01-04T06:53:28.915931Z","shell.execute_reply":"2024-01-04T06:53:28.930897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file_path=traindf_crop['file_path']\nsx,sy,ex,ey=traindf_crop['sx'],traindf_crop['sy'],traindf_crop['ex'],traindf_crop['ey']\n","metadata":{"execution":{"iopub.status.busy":"2024-01-04T06:53:49.384904Z","iopub.execute_input":"2024-01-04T06:53:49.385495Z","iopub.status.idle":"2024-01-04T06:53:49.391952Z","shell.execute_reply.started":"2024-01-04T06:53:49.385446Z","shell.execute_reply":"2024-01-04T06:53:49.391031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x=tf.data.Dataset.from_tensor_slices(list(map(read_image,file_path,sx,ex,sy,ey)))","metadata":{"execution":{"iopub.status.busy":"2024-01-04T06:53:50.984838Z","iopub.execute_input":"2024-01-04T06:53:50.985264Z","iopub.status.idle":"2024-01-04T06:55:54.426516Z","shell.execute_reply.started":"2024-01-04T06:53:50.985232Z","shell.execute_reply":"2024-01-04T06:55:54.424557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" y = tf.data.Dataset.from_tensor_slices(labels)\n    ","metadata":{"execution":{"iopub.status.busy":"2024-01-04T06:55:54.429135Z","iopub.execute_input":"2024-01-04T06:55:54.429568Z","iopub.status.idle":"2024-01-04T06:55:54.439178Z","shell.execute_reply.started":"2024-01-04T06:55:54.429535Z","shell.execute_reply":"2024-01-04T06:55:54.437942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# y","metadata":{"execution":{"iopub.status.busy":"2024-01-04T06:56:06.696143Z","iopub.execute_input":"2024-01-04T06:56:06.696618Z","iopub.status.idle":"2024-01-04T06:56:06.703193Z","shell.execute_reply.started":"2024-01-04T06:56:06.696581Z","shell.execute_reply":"2024-01-04T06:56:06.701715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ds = tf.data.Dataset.zip((x, y))\n    \n    # Create the training and validation splits\nval_ds = (\n        ds\n        .take(50)\n        .batch(8)\n        .prefetch(tf.data.AUTOTUNE)\n    )\ntrain_ds = (\n        ds\n        .skip(50)\n        .shuffle(8 * 10)\n        .batch(8)\n        .prefetch(tf.data.AUTOTUNE)\n    )","metadata":{"execution":{"iopub.status.busy":"2024-01-04T06:56:07.019568Z","iopub.execute_input":"2024-01-04T06:56:07.020846Z","iopub.status.idle":"2024-01-04T06:56:07.073211Z","shell.execute_reply.started":"2024-01-04T06:56:07.020801Z","shell.execute_reply":"2024-01-04T06:56:07.071855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trans_arc = keras_cv.models.EfficientNetV2Backbone.from_preset(\"efficientnetv2_b0\",\n                         input_shape=(256, 256,3))\nfor l in trans_arc.layers:\n    l.trainable = False\ninputs = trans_arc.input\nx = trans_arc.output\nx=tf.keras.layers.Flatten()(x)\n\nx = tf.keras.layers.Dense(256, activation='relu')(x)\nx = tf.keras.layers.BatchNormalization()(x)\nx = tf.keras.layers.Dropout(0.3)(x)\n\nx = tf.keras.layers.Dense(128, activation='relu')(x)\nx = tf.keras.layers.BatchNormalization()(x)\n\nx = tf.keras.layers.Dense(5, activation='relu')(x)\noutputs=tf.keras.layers.Dense(5, activation='softmax')(x)\n\nmodel = tf.keras.Model(inputs=inputs, outputs=outputs)","metadata":{"execution":{"iopub.status.busy":"2024-01-04T06:56:09.437329Z","iopub.execute_input":"2024-01-04T06:56:09.438598Z","iopub.status.idle":"2024-01-04T06:56:14.080594Z","shell.execute_reply.started":"2024-01-04T06:56:09.438543Z","shell.execute_reply":"2024-01-04T06:56:14.079365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model.summary()","metadata":{"execution":{"iopub.status.busy":"2024-01-04T06:56:17.963766Z","iopub.execute_input":"2024-01-04T06:56:17.964185Z","iopub.status.idle":"2024-01-04T06:56:17.969165Z","shell.execute_reply.started":"2024-01-04T06:56:17.964153Z","shell.execute_reply":"2024-01-04T06:56:17.967877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(\n        optimizer=tf.keras.optimizers.Adam(learning_rate=0.001),\n        loss=keras.losses.CategoricalCrossentropy(),\n        metrics=[\"accuracy\"],\n    )\n","metadata":{"execution":{"iopub.status.busy":"2024-01-04T06:56:34.643256Z","iopub.execute_input":"2024-01-04T06:56:34.643703Z","iopub.status.idle":"2024-01-04T06:56:34.676413Z","shell.execute_reply.started":"2024-01-04T06:56:34.643667Z","shell.execute_reply":"2024-01-04T06:56:34.675237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(\n        train_ds,\n        epochs=100,\n        validation_data=val_ds,\n    )\n    \n    ","metadata":{"execution":{"iopub.status.busy":"2024-01-04T06:56:40.019166Z","iopub.execute_input":"2024-01-04T06:56:40.019665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save_weights(\"ucb_ocean_checkpoint.weights.h5\")","metadata":{"execution":{"iopub.status.busy":"2024-01-03T18:06:58.443688Z","iopub.execute_input":"2024-01-03T18:06:58.44412Z","iopub.status.idle":"2024-01-03T18:06:59.281806Z","shell.execute_reply.started":"2024-01-03T18:06:58.444082Z","shell.execute_reply":"2024-01-03T18:06:59.28069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file_pathtest=testdf_crop['file_path']\nsx_,sy_,ex_,ey_=testdf_crop['sx'],testdf_crop['sy'],testdf_crop['ex'],testdf_crop['ey']","metadata":{"execution":{"iopub.status.busy":"2024-01-03T18:08:22.31291Z","iopub.execute_input":"2024-01-03T18:08:22.313327Z","iopub.status.idle":"2024-01-03T18:08:22.319091Z","shell.execute_reply.started":"2024-01-03T18:08:22.313295Z","shell.execute_reply":"2024-01-03T18:08:22.318005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data=tf.data.Dataset.from_tensor_slices(list(map(read_image,file_pathtest,sx_,ex_,sy_,ey_)))","metadata":{"execution":{"iopub.status.busy":"2024-01-03T18:08:23.237087Z","iopub.execute_input":"2024-01-03T18:08:23.237522Z","iopub.status.idle":"2024-01-03T18:08:23.371843Z","shell.execute_reply.started":"2024-01-03T18:08:23.237476Z","shell.execute_reply":"2024-01-03T18:08:23.370942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model.load_weights(\"/kaggle/input/ubc-ovatian/ucb_ocean_checkpoint.weights.h5\")","metadata":{"execution":{"iopub.status.busy":"2024-01-04T06:48:29.788319Z","iopub.execute_input":"2024-01-04T06:48:29.789906Z","iopub.status.idle":"2024-01-04T06:48:30.235314Z","shell.execute_reply.started":"2024-01-04T06:48:29.789853Z","shell.execute_reply":"2024-01-04T06:48:30.233469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# with open(\"/kaggle/working/id_to_name.pkl\", \"rb\") as f:\n#     id_to_name = pickle.load(f)","metadata":{"execution":{"iopub.status.busy":"2024-01-03T18:08:25.856116Z","iopub.execute_input":"2024-01-03T18:08:25.856541Z","iopub.status.idle":"2024-01-03T18:08:25.861684Z","shell.execute_reply.started":"2024-01-03T18:08:25.856509Z","shell.execute_reply":"2024-01-03T18:08:25.86086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" predicted_labels = []\n\nfor index, row in testdf_crop.iterrows():\n        # Get the image path\n    image_path = row[\"file_path\"]\n\n        # Get the image\n    image = read_image(image_path,row['sx'],row['ex'],row['sy'],row['ey'])[None, ...]\n\n        # Predict the label\n    logits = model.predict(image)\n    pred = np.argmax(logits)\n        # Map the pred to the name\n    label = id_to_name[pred]\n\n    predicted_labels.append(label)\n\n    # Add the predicted labels to the csv\n\ntestdf_crop[\"label\"] = predicted_labels","metadata":{"execution":{"iopub.status.busy":"2024-01-03T18:08:27.036703Z","iopub.execute_input":"2024-01-03T18:08:27.037106Z","iopub.status.idle":"2024-01-03T18:08:29.456728Z","shell.execute_reply.started":"2024-01-03T18:08:27.037075Z","shell.execute_reply":"2024-01-03T18:08:29.455773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df = testdf_crop[[\"image_id\", \"label\"]]\nsubmission_df.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2024-01-03T18:08:35.626653Z","iopub.execute_input":"2024-01-03T18:08:35.627019Z","iopub.status.idle":"2024-01-03T18:08:35.6346Z","shell.execute_reply.started":"2024-01-03T18:08:35.62699Z","shell.execute_reply":"2024-01-03T18:08:35.633329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df","metadata":{"execution":{"iopub.status.busy":"2024-01-03T18:08:36.547571Z","iopub.execute_input":"2024-01-03T18:08:36.548322Z","iopub.status.idle":"2024-01-03T18:08:36.558049Z","shell.execute_reply.started":"2024-01-03T18:08:36.548287Z","shell.execute_reply":"2024-01-03T18:08:36.557054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}