{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"},{"sourceId":153776482,"sourceType":"kernelVersion"}],"dockerImageVersionId":30587,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\n\nimport pandas as pd\nimport numpy as np\nfrom tqdm.auto import tqdm\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport pickle\n\n\nfrom skimage import io\nfrom skimage.color import rgb2gray\nfrom skimage.transform import rescale, resize, downscale_local_mean\n\n\nfrom sklearn.preprocessing import LabelEncoder, OneHotEncoder\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score\n\nimport matplotlib.pyplot as plt\n\n\nfrom tensorflow.keras.models import Sequential, load_model\nfrom tensorflow.keras.layers import Dense, Conv2D, Flatten, Dropout\nfrom keras.utils import to_categorical, set_random_seed\n\nimport tensorflow as tf\nimport keras_cv\nimport keras_core as keras\nfrom keras_core import ops","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-12-11T11:46:36.966324Z","iopub.execute_input":"2023-12-11T11:46:36.967313Z","iopub.status.idle":"2023-12-11T11:46:36.975637Z","shell.execute_reply.started":"2023-12-11T11:46:36.967269Z","shell.execute_reply":"2023-12-11T11:46:36.974557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Reproducibility","metadata":{}},{"cell_type":"markdown","source":"# configuration","metadata":{}},{"cell_type":"code","source":"def set_reproducibility(seed=42):\n    np.random.seed(seed)\n    set_random_seed(seed)","metadata":{"execution":{"iopub.status.busy":"2023-12-11T11:46:36.977327Z","iopub.execute_input":"2023-12-11T11:46:36.977646Z","iopub.status.idle":"2023-12-11T11:46:36.991784Z","shell.execute_reply.started":"2023-12-11T11:46:36.97762Z","shell.execute_reply":"2023-12-11T11:46:36.990857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Config:\n    is_submission = False\n    \n    # Reproducibility\n    SEED = 42\n    \n    # Training\n    train_csv_path =        \"/kaggle/input/UBC-OCEAN/train.csv\"\n    train_thumbnail_paths = \"/kaggle/input/UBC-OCEAN/train_thumbnails\"\n    train_dir =             \"/kaggle/input/UBC-OCEAN/train_images\"\n    batch_size = 8\n    epochs = 2\n    \n    # Test\n    test_csv_path =        \"/kaggle/input/UBC-OCEAN/test.csv\"\n    test_thumbnail_paths = \"/kaggle/input/UBC-OCEAN/test_thumbnails\"\n    test_dir =             \"/kaggle/input/UBC-OCEAN/test_images\"\n    \n    # Experiment\n    experiment_name = \"experiment_1\"\n    exp_id = \"id1\"\n    activation_function = keras.activations.softmax\n    loss_func = keras.losses.categorical_crossentropy\n    momentum = 0.9\n    lr = 0.0001\n    \n\n\n\n","metadata":{"execution":{"iopub.status.busy":"2023-12-11T11:46:36.992737Z","iopub.execute_input":"2023-12-11T11:46:36.993084Z","iopub.status.idle":"2023-12-11T11:46:37.002981Z","shell.execute_reply.started":"2023-12-11T11:46:36.993052Z","shell.execute_reply":"2023-12-11T11:46:37.002003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"config = Config()\nset_reproducibility(config.SEED)","metadata":{"execution":{"iopub.status.busy":"2023-12-11T11:46:37.005302Z","iopub.execute_input":"2023-12-11T11:46:37.005659Z","iopub.status.idle":"2023-12-11T11:46:37.018312Z","shell.execute_reply.started":"2023-12-11T11:46:37.005627Z","shell.execute_reply":"2023-12-11T11:46:37.017627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# loading train data","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv(config.train_csv_path)\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-11T11:46:37.02028Z","iopub.execute_input":"2023-12-11T11:46:37.02072Z","iopub.status.idle":"2023-12-11T11:46:37.038407Z","shell.execute_reply.started":"2023-12-11T11:46:37.020689Z","shell.execute_reply":"2023-12-11T11:46:37.037662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# dataset pre processing\n","metadata":{}},{"cell_type":"code","source":"if not config.is_submission:\n    df = pd.read_csv(config.train_csv_path)\n\n    # Create the thumbnail df where is_tma == False\n    df = df[df[\"is_tma\"] == False]\n    \n    # Get basic statistics about the dataset\n    num_rows = df.shape[0]\n    num_unique_images = df['image_id'].nunique()\n    num_unique_labels = df['label'].nunique()\n    unique_labels = df['label'].unique()\n\n    print(f\"{num_rows=}\")\n    print(f\"{num_unique_images=}\")\n    print(f\"{num_unique_labels=}\")\n    print(f\"{unique_labels=}\")\n    \n    # Plot the distribution of the target classes\n    plt.figure(figsize=(10, 6))\n    sns.countplot(data=df, x='label', order=df['label'].value_counts().index)\n    plt.title('Distribution of Target Classes')\n    plt.xlabel('Label')\n    plt.ylabel('Count')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-11T11:46:37.06427Z","iopub.execute_input":"2023-12-11T11:46:37.064575Z","iopub.status.idle":"2023-12-11T11:46:37.338511Z","shell.execute_reply.started":"2023-12-11T11:46:37.064543Z","shell.execute_reply":"2023-12-11T11:46:37.337623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# encoder (for training only)","metadata":{}},{"cell_type":"markdown","source":" ### Perform one-hot encoding of the 'label' column and explicitly convert to integer type","metadata":{}},{"cell_type":"code","source":"df_one_hot = pd.get_dummies(df[\"label\"], prefix=\"label\").astype(int)\ndf_one_hot.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-11T11:46:37.340301Z","iopub.execute_input":"2023-12-11T11:46:37.340729Z","iopub.status.idle":"2023-12-11T11:46:37.355149Z","shell.execute_reply.started":"2023-12-11T11:46:37.34069Z","shell.execute_reply":"2023-12-11T11:46:37.354177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Concatenate the original DataFrame with the one-hot encoded labels\n","metadata":{}},{"cell_type":"code","source":"train_df = pd.concat([df[\"image_id\"], df_one_hot], axis=1)\ntrain_df.head()    ","metadata":{"execution":{"iopub.status.busy":"2023-12-11T11:46:37.356585Z","iopub.execute_input":"2023-12-11T11:46:37.356868Z","iopub.status.idle":"2023-12-11T11:46:37.371754Z","shell.execute_reply.started":"2023-12-11T11:46:37.356836Z","shell.execute_reply":"2023-12-11T11:46:37.370539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Get the thumbnail image paths","metadata":{}},{"cell_type":"code","source":"train_df[\"image_thumbnail_path\"] = train_df[\"image_id\"].apply(lambda x: f\"{config.train_thumbnail_paths}/{x}_thumbnail.png\")\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-11T11:46:37.372992Z","iopub.execute_input":"2023-12-11T11:46:37.373517Z","iopub.status.idle":"2023-12-11T11:46:37.393758Z","shell.execute_reply.started":"2023-12-11T11:46:37.373456Z","shell.execute_reply":"2023-12-11T11:46:37.392625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### make a list with all png thumbnail path","metadata":{}},{"cell_type":"code","source":"image_thumbnail_paths = train_df[\"image_thumbnail_path\"].values\nprint(image_thumbnail_paths[:5])","metadata":{"execution":{"iopub.status.busy":"2023-12-11T11:46:37.395543Z","iopub.execute_input":"2023-12-11T11:46:37.395866Z","iopub.status.idle":"2023-12-11T11:46:37.40308Z","shell.execute_reply.started":"2023-12-11T11:46:37.39584Z","shell.execute_reply":"2023-12-11T11:46:37.401984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = train_df[[col for col in train_df.columns if col.startswith(\"label_\")]].values\nprint(labels[:5])","metadata":{"execution":{"iopub.status.busy":"2023-12-11T11:46:37.40436Z","iopub.execute_input":"2023-12-11T11:46:37.404736Z","iopub.status.idle":"2023-12-11T11:46:37.415702Z","shell.execute_reply.started":"2023-12-11T11:46:37.404708Z","shell.execute_reply":"2023-12-11T11:46:37.414446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_names = [col for col in train_df.columns if col.startswith(\"label_\")]\nprint(label_names)","metadata":{"execution":{"iopub.status.busy":"2023-12-11T11:46:37.416845Z","iopub.execute_input":"2023-12-11T11:46:37.417193Z","iopub.status.idle":"2023-12-11T11:46:37.430591Z","shell.execute_reply.started":"2023-12-11T11:46:37.417168Z","shell.execute_reply":"2023-12-11T11:46:37.429709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"name_to_id = {key.replace(\"label_\", \"\"):value for value,key in enumerate(label_names)}\nprint(name_to_id)","metadata":{"execution":{"iopub.status.busy":"2023-12-11T11:46:37.43431Z","iopub.execute_input":"2023-12-11T11:46:37.434611Z","iopub.status.idle":"2023-12-11T11:46:37.441881Z","shell.execute_reply.started":"2023-12-11T11:46:37.434587Z","shell.execute_reply":"2023-12-11T11:46:37.440969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id_to_name = {key:value for value, key in name_to_id.items()}\nprint(id_to_name)","metadata":{"execution":{"iopub.status.busy":"2023-12-11T11:46:37.443162Z","iopub.execute_input":"2023-12-11T11:46:37.444103Z","iopub.status.idle":"2023-12-11T11:46:37.455525Z","shell.execute_reply.started":"2023-12-11T11:46:37.444076Z","shell.execute_reply":"2023-12-11T11:46:37.454533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"     # Save to dictionary to disk\nwith open(\"id_to_name.pkl\", \"wb\") as f:\n    pickle.dump(id_to_name, f)","metadata":{"execution":{"iopub.status.busy":"2023-12-11T11:46:37.458991Z","iopub.execute_input":"2023-12-11T11:46:37.459792Z","iopub.status.idle":"2023-12-11T11:46:37.464543Z","shell.execute_reply.started":"2023-12-11T11:46:37.459756Z","shell.execute_reply":"2023-12-11T11:46:37.463541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Création des poids d'apprentissage","metadata":{}},{"cell_type":"code","source":"if not config.is_submission:\n    class_weights = np.sum(labels) - np.sum(labels, axis=0)\n    class_weights = class_weights / np.sum(class_weights) # Normalize the weights\n\n    class_weights = {idx:weight for idx, weight in enumerate(class_weights)}\n\n    for idx, weight in class_weights.items():\n        print(f\"{id_to_name[idx]}: {weight:0.2f}\")","metadata":{"execution":{"iopub.status.busy":"2023-12-11T11:46:37.465816Z","iopub.execute_input":"2023-12-11T11:46:37.466122Z","iopub.status.idle":"2023-12-11T11:46:37.476637Z","shell.execute_reply.started":"2023-12-11T11:46:37.466086Z","shell.execute_reply":"2023-12-11T11:46:37.475472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# dataset pipeline","metadata":{}},{"cell_type":"code","source":"def read_image(path):\n    file = tf.io.read_file(path)\n    image = tf.io.decode_png(file, 3)\n    image = tf.image.resize(image, (256, 256))\n    image = tf.image.per_image_standardization(image)\n    return image","metadata":{"execution":{"iopub.status.busy":"2023-12-11T11:46:37.477994Z","iopub.execute_input":"2023-12-11T11:46:37.478306Z","iopub.status.idle":"2023-12-11T11:46:37.49054Z","shell.execute_reply.started":"2023-12-11T11:46:37.478281Z","shell.execute_reply":"2023-12-11T11:46:37.489606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not config.is_submission:\n    x = (\n        tf.data.Dataset.from_tensor_slices(image_thumbnail_paths)\n        .map(read_image, num_parallel_calls=tf.data.AUTOTUNE)\n    )\n    y = tf.data.Dataset.from_tensor_slices(labels)\n\n    # Zip the x and y together\n    ds = tf.data.Dataset.zip((x, y))\n    \n    # Create the training and validation splits\n    val_ds = (\n        ds\n        .take(50)\n        .batch(config.batch_size)\n        .prefetch(tf.data.AUTOTUNE)\n    )\n    train_ds = (\n        ds\n        .skip(50)\n        .shuffle(config.batch_size * 10)\n        .batch(config.batch_size)\n        .prefetch(tf.data.AUTOTUNE)\n    )","metadata":{"execution":{"iopub.status.busy":"2023-12-11T11:46:37.491689Z","iopub.execute_input":"2023-12-11T11:46:37.491954Z","iopub.status.idle":"2023-12-11T11:46:37.551938Z","shell.execute_reply.started":"2023-12-11T11:46:37.491932Z","shell.execute_reply":"2023-12-11T11:46:37.550938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_ds.enumerate)","metadata":{"execution":{"iopub.status.busy":"2023-12-11T11:46:37.553061Z","iopub.execute_input":"2023-12-11T11:46:37.553337Z","iopub.status.idle":"2023-12-11T11:46:37.558462Z","shell.execute_reply.started":"2023-12-11T11:46:37.553313Z","shell.execute_reply":"2023-12-11T11:46:37.557471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# build the model","metadata":{}},{"cell_type":"code","source":"lr = 0.0001\n\nmodel = keras.models.Sequential([\n    keras.layers.Conv2D(32, (3, 3), activation=keras.activations.relu, input_shape=(256, 256, 3), padding='same'),\n    keras.layers.Conv2D(32, (3, 3), activation=keras.activations.relu, input_shape=(32, 32, 3), padding='same'),\n    keras.layers.MaxPooling2D((2, 2)),\n\n\n    keras.layers.Conv2D(64, (3, 3), activation=keras.activations.relu, input_shape=(32, 32, 3), padding='same'),\n    keras.layers.Conv2D(64, (3, 3), activation=keras.activations.relu, input_shape=(32, 32, 3), padding='same'),\n    keras.layers.MaxPooling2D((2, 2)),\n\n    keras.layers.Conv2D(128, (3, 3), activation=keras.activations.relu, input_shape=(32, 32, 3), padding='same'),\n    keras.layers.Conv2D(128, (3, 3), activation=keras.activations.relu, input_shape=(32, 32, 3), padding='same'),\n    keras.layers.MaxPooling2D((2, 2)),\n\n    keras.layers.Flatten(),\n    keras.layers.Dense(5, activation=keras.activations.softmax),\n])\nmodel.compile(\n    loss=config.loss_func,\n    optimizer=keras.optimizers.Adam(learning_rate=lr),\n    metrics=['accuracy']\n)\n","metadata":{"execution":{"iopub.status.busy":"2023-12-11T11:53:45.644535Z","iopub.execute_input":"2023-12-11T11:53:45.644898Z","iopub.status.idle":"2023-12-11T11:53:45.700636Z","shell.execute_reply.started":"2023-12-11T11:53:45.644872Z","shell.execute_reply":"2023-12-11T11:53:45.699622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-12-11T11:53:49.164256Z","iopub.execute_input":"2023-12-11T11:53:49.164758Z","iopub.status.idle":"2023-12-11T11:53:49.195322Z","shell.execute_reply.started":"2023-12-11T11:53:49.164712Z","shell.execute_reply":"2023-12-11T11:53:49.194466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# training\n","metadata":{}},{"cell_type":"code","source":"val_ds\n","metadata":{"execution":{"iopub.status.busy":"2023-12-11T11:46:37.684804Z","iopub.execute_input":"2023-12-11T11:46:37.685061Z","iopub.status.idle":"2023-12-11T11:46:37.690399Z","shell.execute_reply.started":"2023-12-11T11:46:37.685038Z","shell.execute_reply":"2023-12-11T11:46:37.689574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Inspecter train_ds\nfor images, labels in train_ds.take(1):  # Prend le premier batch pour l'inspection\n    print(\"Taille du batch dans train_ds:\")\n    print(\"  - Images:\", images.shape)  # Taille des images dans le batch\n    print(\"  - Labels:\", labels.shape)  # Taille des labels dans le batch\n\n# Inspecter val_ds\nfor images, labels in val_ds.take(1):  # Prend le premier batch pour l'inspection\n    print(\"Taille du batch dans val_ds:\")\n    print(\"  - Images:\", images.shape)  # Taille des images dans le batch\n    print(\"  - Labels:\", labels.shape)  # Taille des labels dans le batch","metadata":{"execution":{"iopub.status.busy":"2023-12-11T11:48:28.020318Z","iopub.execute_input":"2023-12-11T11:48:28.021121Z","iopub.status.idle":"2023-12-11T11:48:35.290401Z","shell.execute_reply.started":"2023-12-11T11:48:28.021085Z","shell.execute_reply":"2023-12-11T11:48:35.289383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#import datetime\n\n#if not config.is_submission:\n    #date_str = datetime.datetime.now().strftime(\"%Y%m%d-%H%M%S\")\n    #log_dir = f\"./logdir/{date_str}{config.experiment_name}_{str(config.exp_id)}\"\nmodel.fit(\n    train_ds,\n    epochs=2,\n    validation_data=val_ds,\n    class_weight= class_weights,\n )\n\nmodel.save_weights(\"ucb_ocean_checkpoint.weights.h5\")","metadata":{"execution":{"iopub.status.busy":"2023-12-11T11:54:55.496795Z","iopub.execute_input":"2023-12-11T11:54:55.497152Z","iopub.status.idle":"2023-12-11T11:54:55.660778Z","shell.execute_reply.started":"2023-12-11T11:54:55.497124Z","shell.execute_reply":"2023-12-11T11:54:55.659519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#","metadata":{"execution":{"iopub.status.busy":"2023-12-11T11:46:37.877139Z","iopub.status.idle":"2023-12-11T11:46:37.877482Z","shell.execute_reply.started":"2023-12-11T11:46:37.877311Z","shell.execute_reply":"2023-12-11T11:46:37.877326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# testing","metadata":{}},{"cell_type":"code","source":"test = pd.read_csv(\"/kaggle/input/UBC-OCEAN/test.csv\")\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-11T11:46:37.878947Z","iopub.status.idle":"2023-12-11T11:46:37.879293Z","shell.execute_reply.started":"2023-12-11T11:46:37.879134Z","shell.execute_reply":"2023-12-11T11:46:37.87915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# saving for submition","metadata":{}},{"cell_type":"code","source":"if config.is_submission:\n    sample_submission = pd.read_csv(\"/kaggle/input/UBC-OCEAN/sample_submission.csv\")\n    sample_submission['label'] = preds\n    sample_submission","metadata":{"execution":{"iopub.status.busy":"2023-12-11T11:46:37.880453Z","iopub.status.idle":"2023-12-11T11:46:37.880962Z","shell.execute_reply.started":"2023-12-11T11:46:37.880719Z","shell.execute_reply":"2023-12-11T11:46:37.880741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if config.is_submission:\n    sample_submission.to_csv('submission.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2023-12-11T11:46:37.882752Z","iopub.status.idle":"2023-12-11T11:46:37.883189Z","shell.execute_reply.started":"2023-12-11T11:46:37.882962Z","shell.execute_reply":"2023-12-11T11:46:37.882983Z"},"trusted":true},"execution_count":null,"outputs":[]}]}