{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"},{"sourceId":7530723,"sourceType":"datasetVersion","datasetId":4386165},{"sourceId":7552440,"sourceType":"datasetVersion","datasetId":4398777}],"dockerImageVersionId":30648,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport seaborn as sns\nimport pickle\nimport json\n\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\n\nfrom keras.utils import set_random_seed\n\nimport tensorflow as tf\nimport tensorflow.keras as keras\nimport datetime\n\nimport itertools\nfrom tensorboard.plugins.hparams import api as hp\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n#    for filename in filenames:\n#        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\ndef set_reproducibility(seed=42):\n    np.random.seed(seed)\n    set_random_seed(seed)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-02-04T13:59:32.212317Z","iopub.execute_input":"2024-02-04T13:59:32.212615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n#PATH\n#work    \nds_src = \"/kaggle/input/UBC-OCEAN/\"\ndst_path = \"/kaggle/working/\"\n\n#home\n#ds_src =\n#dst_path =\n\n#kaggle\n#ds_src = \"/kaggle/input/UBC-OCEAN/\"\n#dst_path = \"\"\n\n    \n# Training\ntrain_csv_path =        f\"{ds_src}train.csv\"\ntrain_thumbnail_paths = f\"{ds_src}train_thumbnails\"\ntrain_dir =             f\"{ds_src}train_images\"\n\n    \n# Test\ntest_csv_path =        f\"{ds_src}test.csv\"\ntest_thumbnail_paths = f\"{ds_src}test_thumbnails\"\ntest_dir =             f\"{ds_src}test_images\"\n    \nactivation_function = keras.activations.softmax\nloss_func = keras.losses.categorical_crossentropy\n\n\n# Experiment\nexperiment_name = \"mlp_4\"\ngrid_search_folder = \"input/\"\n\nwith open(f'/kaggle/input/mlp-version-3/mlp_4.json', 'r') as f:\n    global_grid = json.load(f)\n\nparams_grid = global_grid['params_grid']\nimage_size = global_grid['image_size']\n# dictionnary \nid_to_name_dst = f\"{dst_path}id_to_name.pkl\"\n# model\nmodel_name = \"thumbnail-weighted_linear\"\nmodel_path = f\"{dst_path}{model_name}.weights.h5\"\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"set_reproducibility(global_grid['seed'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(tf.config.list_physical_devices('GPU'))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(train_csv_path)\ntrain.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(train_csv_path)\n\n# Create the thumbnail df where is_tma == False\ndf = df[df[\"is_tma\"] == False]\n\n# Get basic statistics about the dataset\nnum_rows = df.shape[0]\nnum_unique_images = df['image_id'].nunique()\nnum_unique_labels = df['label'].nunique()\nunique_labels = df['label'].unique()\n\nprint(f\"{num_rows=}\")\nprint(f\"{num_unique_images=}\")\nprint(f\"{num_unique_labels=}\")\nprint(f\"{unique_labels=}\")\n\n# Plot the distribution of the target classes\nplt.figure(figsize=(10, 6))\nsns.countplot(data=df, x='label', order=df['label'].value_counts().index)\nplt.title('Distribution of Target Classes')\nplt.xlabel('Label')\nplt.ylabel('Count')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_one_hot = pd.get_dummies(df[\"label\"], prefix=\"label\").astype(int)\ndf_one_hot.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.concat([df[\"image_id\"], df_one_hot], axis=1)\ntrain_df.head()    ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[\"image_thumbnail_path\"] = train_df[\"image_id\"].apply(lambda x: f\"{train_thumbnail_paths}/{x}_thumbnail.png\")\ntrain_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_thumbnail_paths = train_df[\"image_thumbnail_path\"].values\nprint(image_thumbnail_paths[:5])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = train_df[[col for col in train_df.columns if col.startswith(\"label_\")]].values\nprint(labels[:5])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_names = [col for col in train_df.columns if col.startswith(\"label_\")]\nprint(label_names)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"name_to_id = {key.replace(\"label_\", \"\"):value for value,key in enumerate(label_names)}\nprint(name_to_id)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id_to_name = {key:value for value, key in name_to_id.items()}\nprint(id_to_name)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"     # Save to dictionary to disk\nwith open(id_to_name_dst, \"wb\") as f:\n    pickle.dump(id_to_name, f)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_weights = np.sum(labels) - np.sum(labels, axis=0)\nclass_weights = class_weights / np.sum(class_weights) # Normalize the weights\n\nclass_weights = {idx:weight for idx, weight in enumerate(class_weights)}\n\nfor idx, weight in class_weights.items():\n    print(f\"{id_to_name[idx]}: {weight:0.2f}\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_image(path):\n    file = tf.io.read_file(path)\n    image = tf.io.decode_png(file, 3)\n    image = tf.image.resize(image, (image_size, image_size))\n    image = tf.image.per_image_standardization(image)\n    return image","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = (\n    tf.data.Dataset.from_tensor_slices(image_thumbnail_paths)\n    .map(read_image, num_parallel_calls=tf.data.AUTOTUNE)\n)\ny = tf.data.Dataset.from_tensor_slices(labels)\n\n# Zip the x and y together\nds = tf.data.Dataset.zip((x, y))\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_dataset(batch_size, ds):\n    val_ds = (\n        ds\n        .take(50)\n        .batch(batch_size)\n        .prefetch(tf.data.AUTOTUNE)\n    )\n    train_ds = (\n        ds\n        .skip(50)\n        .shuffle(batch_size * 10)\n        .batch(batch_size)\n        .prefetch(tf.data.AUTOTUNE)\n    )\n    return train_ds, val_ds","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_optimizer(param_optimizer):\n    optimizer_name = param_optimizer['optimizer']\n    if optimizer_name == \"adam\":\n        return keras.optimizers.Adam(learning_rate=param_optimizer['learning_rate'])\n    if optimizer_name == \"rmsprops\":\n        return keras.optimizers.RMSprop(learning_rate=param_optimizer['learning_rate'])\n    if optimizer_name == \"sgd\":\n        return keras.optimizers.SGD(learning_rate=param_optimizer['learning_rate'])\n    if optimizer_name == \"sgd_momentum\":\n        return keras.optimizers.SGD(learning_rate=param_optimizer['learning_rate'], momentum=param_optimizer['momentum'])\n    print(\"optimizer not found\")\n    print(param_optimizer)\n    return None\n    \n        ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_layer(layer, input):\n    if layer[0] == \"Conv2D\":\n        return keras.layers.Conv2D(layer[1],(3, 3), padding=layer[2], activation=layer[3])(input)\n    if layer[0] == \"MaxPooling2D\":\n        return keras.layers.MaxPooling2D((layer[1],layer[1]))(input)\n    if layer[0] == \"Dense\":\n        return keras.layers.Dense(layer[1], activation=layer[2])(input)\n    if layer[0] == \"Dropout\":\n        return keras.layers.Dropout(layer[1])(input)\n    if layer[0] == \"BatchNormalization\":\n        return keras.layers.BatchNormalization()(input)\n    if layer[0] == \"RandomFlip\":\n        return keras.layers.RandomFlip(layer[1])(input)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_model(params, model_grid):\n    input_tensor = keras.layers.Input(shape=(image_size, image_size, 3))\n    \n    x = input_tensor\n    for layer in model_grid:\n        x = get_layer(layer, x)\n    \n    \n    flattened_input_tensor = keras.layers.Flatten()(x)\n    output_tensor = keras.layers.Dense(5, activation=keras.activations.softmax)(flattened_input_tensor)\n    \n    model = keras.models.Model(inputs=[input_tensor], outputs=[output_tensor])\n    \n    model.compile(\n        loss=loss_func,\n        optimizer= get_optimizer(params),\n        metrics=['accuracy'],\n        )\n    return model","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"early_stopping = keras.callbacks.EarlyStopping(\n    patience=5,\n    min_delta=0.001,\n    restore_best_weights=True,\n    monitor='loss'\n)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class EarlyStoppingLogging(tf.keras.callbacks.Callback):\n    def __init__(self, early_stopping_callback, log_dir):\n        super().__init__()\n        self.early_stopping = early_stopping_callback\n        self.stopped_epoch = 0\n        self.writer = tf.summary.create_file_writer(log_dir)\n\n    def on_epoch_end(self, epoch, logs=None):\n        if self.early_stopping.stopped_epoch > 0:\n            self.stopped_epoch = self.early_stopping.stopped_epoch\n            with self.writer.as_default():\n                tf.summary.scalar('early_stopping_epoch', self.stopped_epoch, step=epoch)\n                self.writer.flush()\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_model(params_dict, log_dir, ds, model_grid):\n    model = create_model(params_dict, model_grid)\n    train_ds, val_ds = build_dataset(params_dict['batch_size'], ds)\n\n    tensorboard_callback = keras.callbacks.TensorBoard(log_dir=log_dir, histogram_freq=1)\n    early_stopping_logging_callback = EarlyStoppingLogging(early_stopping, log_dir)\n\n    model.fit(\n        train_ds,\n        epochs=params_dict['epochs'],\n        validation_data=val_ds,\n        callbacks=[\n            early_stopping,\n            tensorboard_callback,\n            hp.KerasCallback(log_dir, params_dict),\n            early_stopping_logging_callback,\n\n        ],\n        class_weight= class_weights,\n        verbose=1\n    )\n    _, val_acc = model.evaluate(val_ds)\n    return val_acc\n    ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def run_grid_search(param_grid, ds, model_config):\n    total_iterations = 1\n    for values in param_grid.values():\n        total_iterations *= len(values)\n        \n    for params in tqdm(itertools.product(*param_grid.values()), total=total_iterations):\n        params_dict = dict(zip(params_grid.keys(), params))\n        print(params_dict)\n\n        log_dir = (f\"log/fit/{experiment_name}/\" \n                   + \"_\".join(f\"{v}\" for k, v in params_dict.items()) \n                   + \"_\" \n                   + datetime.datetime.now().strftime(\"%Y%m%d-%H%M%S\")\n                   )\n        \n        \n        print(log_dir)\n        \n        #build dataset\n        with tf.summary.create_file_writer(log_dir).as_default():\n            params_dict_with_name = params_dict\n            params_dict_with_name['run'] = experiment_name\n            hp.hparams(params_dict_with_name)  # record the values used in this trial\n            accuracy = train_model(params_dict_with_name, log_dir, ds, model_config)\n            tf.summary.scalar(\"accuracy\", accuracy, step=1)\n        ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(params_grid)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"run_grid_search(params_grid, ds, global_grid['model'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!zip -r pmc_1.zip /kaggle/working/log/fit","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import FileLink\nFileLink(r'pmc_1.zip')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}