{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"},{"sourceId":992,"sourceType":"modelInstanceVersion","modelInstanceId":846}],"dockerImageVersionId":30559,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import cv2\nimport pickle\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom tensorflow.keras.applications.resnet50 import ResNet50\nfrom tensorflow.keras.models import Sequential, Model\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout, GlobalAveragePooling2D\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import EarlyStopping\nimport tensorflow_hub as hub\nimport tensorflow as tf\nimport keras_cv\nimport keras_core as keras\nfrom keras_core import ops\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder","metadata":{"execution":{"iopub.status.busy":"2023-11-29T17:32:22.836287Z","iopub.execute_input":"2023-11-29T17:32:22.836989Z","iopub.status.idle":"2023-11-29T17:32:22.845195Z","shell.execute_reply.started":"2023-11-29T17:32:22.836955Z","shell.execute_reply":"2023-11-29T17:32:22.843148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Config:\n    # Training\n    train_csv_path = \"/kaggle/input/UBC-OCEAN/train.csv\"\n    train_thumbnail_paths = \"/kaggle/input/UBC-OCEAN/train_thumbnails\"\n    batch_size = 8\n    learning_rate = 1e-3\n    epochs = 20\n    \n    # Inference\n    test_csv_path = \"/kaggle/input/UBC-OCEAN/test.csv\"\n    test_thumbnail_paths = \"/kaggle/input/UBC-OCEAN/test_thumbnails\"\n\nconfig = Config()","metadata":{"execution":{"iopub.status.busy":"2023-11-29T17:32:22.84738Z","iopub.execute_input":"2023-11-29T17:32:22.847928Z","iopub.status.idle":"2023-11-29T17:32:22.873251Z","shell.execute_reply.started":"2023-11-29T17:32:22.847891Z","shell.execute_reply":"2023-11-29T17:32:22.872132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"To help with reproducibility we set the seed of the Pseudo Random Number Generator.","metadata":{}},{"cell_type":"code","source":"# Load metadata from CSV files\ntrain_metadata = pd.read_csv('/kaggle/input/UBC-OCEAN/train.csv')\ntest_metadata = pd.read_csv('/kaggle/input/UBC-OCEAN/test.csv')\n\n# Filter rows where \"is_tma\" is False\ntrain_metadata = train_metadata[train_metadata['is_tma'] == False]\n\n# Split the training data into training and validation sets\ntrain_data, val_data = train_test_split(train_metadata, test_size=0.2, random_state=42)\n\n\n# Define paths to image files\ntrain_path_prefix = '/kaggle/input/UBC-OCEAN/train_thumbnails/'\ntrain_image_paths = [train_path_prefix + f\"{img_id}_thumbnail.png\" for img_id in train_data['image_id']]\nval_image_paths = [train_path_prefix + f\"{img_id}_thumbnail.png\" for img_id in val_data['image_id']]","metadata":{"execution":{"iopub.status.busy":"2023-11-29T17:32:22.875655Z","iopub.execute_input":"2023-11-29T17:32:22.876127Z","iopub.status.idle":"2023-11-29T17:32:22.895789Z","shell.execute_reply.started":"2023-11-29T17:32:22.876085Z","shell.execute_reply":"2023-11-29T17:32:22.894589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_image(path):\n    try:\n        file = tf.io.read_file(path)\n        image = tf.io.decode_png(file, 3)\n        image = tf.image.resize(image, (224, 224))\n        image = tf.image.per_image_standardization(image)\n    except Exception as e:\n        print(f\"Error processing image {image_path}: {e}\")\n        return None\n    return image","metadata":{"execution":{"iopub.status.busy":"2023-11-29T17:32:22.897037Z","iopub.execute_input":"2023-11-29T17:32:22.897358Z","iopub.status.idle":"2023-11-29T17:32:22.903941Z","shell.execute_reply.started":"2023-11-29T17:32:22.897332Z","shell.execute_reply":"2023-11-29T17:32:22.902792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load image paths and labels as NumPy arrays\ntrain_image_paths_np = np.array(train_image_paths)\nval_image_paths_np = np.array(val_image_paths)\n# Encode labels using one_hot\ntrain_labels = pd.get_dummies(train_data['label']).astype(int).values\nval_labels = pd.get_dummies(val_data['label']).astype(int).values\n\n# Read and preprocess images\ntrain_images = np.array([read_image(path) for path in train_image_paths_np])\nval_images = np.array([read_image(path) for path in val_image_paths_np])","metadata":{"execution":{"iopub.status.busy":"2023-11-29T17:32:22.906201Z","iopub.execute_input":"2023-11-29T17:32:22.906543Z","iopub.status.idle":"2023-11-29T17:34:32.633367Z","shell.execute_reply.started":"2023-11-29T17:32:22.906515Z","shell.execute_reply":"2023-11-29T17:34:32.63242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_one_hot = pd.get_dummies(train_metadata[\"label\"], prefix=\"label\").astype(int)\n\n# Concatenate the original DataFrame with the one-hot encoded labels\ntrain_df = pd.concat([train_metadata[\"image_id\"], df_one_hot], axis=1)\n\n# Get the thumbnail image paths\ntrain_df[\"image_thumbnail_path\"] = train_df[\"image_id\"].apply(lambda x: f\"{config.train_thumbnail_paths}/{x}_thumbnail.png\")\n\nimage_thumbnail_paths = train_df[\"image_thumbnail_path\"].values\nlabels = train_df[[col for col in train_df.columns if col.startswith(\"label_\")]].values\n\nlabel_names = [col for col in train_df.columns if col.startswith(\"label_\")]\nname_to_id = {key.replace(\"label_\", \"\"):value for value,key in enumerate(label_names)}\nid_to_name = {key:value for value, key in name_to_id.items()}\n\nprint(id_to_name)","metadata":{"execution":{"iopub.status.busy":"2023-11-29T17:34:32.634627Z","iopub.execute_input":"2023-11-29T17:34:32.634938Z","iopub.status.idle":"2023-11-29T17:34:32.651685Z","shell.execute_reply.started":"2023-11-29T17:34:32.634912Z","shell.execute_reply":"2023-11-29T17:34:32.650563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define a function to download the ResNet50 model from TensorFlow Hub\ndef download_resnet_model():\n    resnet_url = \"https://www.kaggle.com/models/tensorflow/resnet-50/frameworks/TensorFlow2/variations/classification/versions/1\"  # ResNet50\n    resnet_model = hub.KerasLayer(resnet_url, input_shape=(224, 224, 3))\n    resnet_model.trainable = True\n    return resnet_model","metadata":{"execution":{"iopub.status.busy":"2023-11-29T17:34:32.653587Z","iopub.execute_input":"2023-11-29T17:34:32.653969Z","iopub.status.idle":"2023-11-29T17:34:32.668907Z","shell.execute_reply.started":"2023-11-29T17:34:32.653936Z","shell.execute_reply":"2023-11-29T17:34:32.667908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the ResNet50 model\nresnet_model = download_resnet_model()\n\n# Create a new model with ResNet50 as the base\nmodel = Sequential([\n    resnet_model,\n    Flatten(),\n    Dense(896, activation='relu'),\n    Dropout(0.5),\n    Dense(448, activation='relu'),\n    Dropout(0.5),\n    Dense(224, activation='relu'),\n    Dense(5, activation='softmax')  # Adjust to the number of classes\n])\nmodel.build([None, 224, 224, 3])  # Batch input shape.\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-11-29T17:34:32.671253Z","iopub.execute_input":"2023-11-29T17:34:32.671577Z","iopub.status.idle":"2023-11-29T17:34:44.003358Z","shell.execute_reply.started":"2023-11-29T17:34:32.671551Z","shell.execute_reply":"2023-11-29T17:34:44.002545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 定義學習率\nlearning_rate = config.learning_rate\n\n# 使用Adam優化器並設定學習率\noptimizer = Adam(learning_rate=learning_rate)\n\nmodel.compile(optimizer='adam', loss='categorical_crossentropy', metrics=['accuracy'])\n\n# Create early stopping callback\nearly_stopping = EarlyStopping(monitor='val_loss', patience=5)\n\n\n# Train the model\nhistory = model.fit(\n    train_images, train_labels,\n    epochs=config.epochs,\n    validation_data=(val_images, val_labels),\n    batch_size=config.batch_size,\n    callbacks=[early_stopping]\n)\n\nmodel.save_weights(\"ucb_ocean_checkpoint.weights.h5\")","metadata":{"execution":{"iopub.status.busy":"2023-11-29T17:34:44.00466Z","iopub.execute_input":"2023-11-29T17:34:44.004991Z","iopub.status.idle":"2023-11-29T17:36:46.273842Z","shell.execute_reply.started":"2023-11-29T17:34:44.004964Z","shell.execute_reply":"2023-11-29T17:36:46.27258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(config.test_csv_path)\ndf[\"image_path\"] = df[\"image_id\"].apply(lambda x: f\"{config.test_thumbnail_paths}/{x}_thumbnail.png\")","metadata":{"execution":{"iopub.status.busy":"2023-11-29T17:36:46.275749Z","iopub.execute_input":"2023-11-29T17:36:46.276164Z","iopub.status.idle":"2023-11-29T17:36:46.285382Z","shell.execute_reply.started":"2023-11-29T17:36:46.276126Z","shell.execute_reply":"2023-11-29T17:36:46.284294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predicted_labels = []\n\nfor index, row in df.iterrows():\n    # Get the image path\n    image_path = row[\"image_path\"]\n\n    # Get the image\n    image = read_image(image_path)[None, ...]\n    if image is None:\n        predicted_labels.append('HGSC')\n        continue\n    print(image.shape)\n\n    # Predict the label\n    predictions = model.predict(image)\n    # Convert predicted probabilities to class labels\n    predicted_label_list = [np.argmax(predictions) for prediction in predictions]\n   \n    predicted_subtypes = id_to_name[predicted_label_list[0]]\n\n    # Print predicted subtypes\n    print(predicted_subtypes)\n\n    predicted_labels.append(predicted_subtypes)\n\n# Add the predicted labels to the csv\ndf[\"label\"] = predicted_labels","metadata":{"execution":{"iopub.status.busy":"2023-11-29T17:36:46.286579Z","iopub.execute_input":"2023-11-29T17:36:46.286996Z","iopub.status.idle":"2023-11-29T17:36:47.875541Z","shell.execute_reply.started":"2023-11-29T17:36:46.28695Z","shell.execute_reply":"2023-11-29T17:36:47.87428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create the submission\nsubmission_df = df[[\"image_id\", \"label\"]]\nsubmission_df.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-11-29T17:36:47.876819Z","iopub.execute_input":"2023-11-29T17:36:47.877128Z","iopub.status.idle":"2023-11-29T17:36:47.885646Z","shell.execute_reply.started":"2023-11-29T17:36:47.877102Z","shell.execute_reply":"2023-11-29T17:36:47.884624Z"},"trusted":true},"execution_count":null,"outputs":[]}]}