{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Published on October 19, 2023. By Marília Prata, mpwolke ","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-10-20T00:33:36.000271Z","iopub.execute_input":"2023-10-20T00:33:36.003495Z","iopub.status.idle":"2023-10-20T00:33:36.507358Z","shell.execute_reply.started":"2023-10-20T00:33:36.00342Z","shell.execute_reply":"2023-10-20T00:33:36.505215Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"![](https://pbs.twimg.com/media/F0w3RwLaYAInwqD.jpg:large)X.com","metadata":{}},{"cell_type":"markdown","source":"#I tried to adapt the code made by Keras team on UBC-OCEAN. And got stucked with the tif.\n\nThat's the Original Notebook made by \n\nAritra Roy Gosthipaty, Martin Görner, Gusthema and Phil Culliton\n\nhttps://www.kaggle.com/code/aritrag/kerascv-train-and-infer-on-thumbnails","metadata":{}},{"cell_type":"code","source":"import os\nos.environ[\"KERAS_BACKEND\"] = \"jax\" # or \"tensorflow\", \"torch\"\n\nimport cv2\nimport pickle\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Set the style for the plot\nsns.set(style=\"whitegrid\")\n\nimport tensorflow as tf\nimport keras_cv\nimport keras_core as keras\nfrom keras_core import ops","metadata":{"execution":{"iopub.status.busy":"2023-10-20T00:25:40.803043Z","iopub.execute_input":"2023-10-20T00:25:40.803476Z","iopub.status.idle":"2023-10-20T00:25:58.53474Z","shell.execute_reply.started":"2023-10-20T00:25:40.803442Z","shell.execute_reply":"2023-10-20T00:25:58.533535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from PIL import Image\n\nImage.MAX_IMAGE_PIXELS = None","metadata":{"execution":{"iopub.status.busy":"2023-10-20T00:34:55.70683Z","iopub.execute_input":"2023-10-20T00:34:55.707243Z","iopub.status.idle":"2023-10-20T00:34:55.712886Z","shell.execute_reply.started":"2023-10-20T00:34:55.707217Z","shell.execute_reply":"2023-10-20T00:34:55.711699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Config:\n    is_submission = False\n    \n    # Reproducibility\n    SEED = 42\n    \n    # Training\n    train_csv_path = \"/kaggle/input/mayo-clinic-strip-ai/train.csv\"\n    train_thumbnail_paths = \"/kaggle/input/mayo-clinic-strip-ai/train\"\n    batch_size = 8\n    learning_rate = 1e-3\n    epochs = 2\n    \n    # Inference\n    test_csv_path = \"/kaggle/input/mayo-clinic-strip-ai/test.csv\"\n    test_thumbnail_paths = \"/kaggle/input/mayo-clinic-strip-ai/test\"\n\nconfig = Config()","metadata":{"execution":{"iopub.status.busy":"2023-10-20T00:25:58.536262Z","iopub.execute_input":"2023-10-20T00:25:58.536867Z","iopub.status.idle":"2023-10-20T00:25:58.542818Z","shell.execute_reply.started":"2023-10-20T00:25:58.536835Z","shell.execute_reply":"2023-10-20T00:25:58.541695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"keras.utils.set_random_seed(seed=config.SEED)","metadata":{"execution":{"iopub.status.busy":"2023-10-20T00:26:02.045166Z","iopub.execute_input":"2023-10-20T00:26:02.045505Z","iopub.status.idle":"2023-10-20T00:26:02.052935Z","shell.execute_reply.started":"2023-10-20T00:26:02.045476Z","shell.execute_reply":"2023-10-20T00:26:02.051304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not config.is_submission:\n    df = pd.read_csv(config.train_csv_path)\n\n    # Create the thumbnail df where is_tma == False\n    #df = df[df[\"is_tma\"] == False]  #We don't have any bool here\n    \n    # Get basic statistics about the dataset\n    num_rows = df.shape[0]\n    num_unique_images = df['image_id'].nunique()\n    num_unique_labels = df['label'].nunique()\n    unique_labels = df['label'].unique()\n\n    print(f\"{num_rows=}\")\n    print(f\"{num_unique_images=}\")\n    print(f\"{num_unique_labels=}\")\n    print(f\"{unique_labels=}\")\n    \n    # Plot the distribution of the target classes\n    plt.figure(figsize=(10, 6))\n    sns.countplot(data=df, x='label', order=df['label'].value_counts().index)\n    plt.title('Distribution of Target Classes')\n    plt.xlabel('Label')\n    plt.ylabel('Count')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-20T00:26:08.232573Z","iopub.execute_input":"2023-10-20T00:26:08.232941Z","iopub.status.idle":"2023-10-20T00:26:08.567804Z","shell.execute_reply.started":"2023-10-20T00:26:08.232915Z","shell.execute_reply":"2023-10-20T00:26:08.566731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#I didn't changed the thumbnails from UBC_OCEAN competition to avoid risking ruin it soon.","metadata":{}},{"cell_type":"code","source":"if not config.is_submission:\n    # Perform one-hot encoding of the 'label' column and explicitly convert to integer type\n    df_one_hot = pd.get_dummies(df[\"label\"], prefix=\"label\").astype(int)\n\n    # Concatenate the original DataFrame with the one-hot encoded labels\n    train_df = pd.concat([df[\"image_id\"], df_one_hot], axis=1)\n\n    # Get the thumbnail image paths\n    train_df[\"image_thumbnail_path\"] = train_df[\"image_id\"].apply(lambda x: f\"{config.train_thumbnail_paths}/{x}.tif\")#Original is {x}_thumbnail.png\"\n    \n    image_thumbnail_paths = train_df[\"image_thumbnail_path\"].values\n    labels = train_df[[col for col in train_df.columns if col.startswith(\"label_\")]].values\n\n    label_names = [col for col in train_df.columns if col.startswith(\"label_\")]\n    name_to_id = {key.replace(\"label_\", \"\"):value for value,key in enumerate(label_names)}\n    id_to_name = {key:value for value, key in name_to_id.items()}\n    \n    # Save to dictionary to disk\n    with open(\"id_to_name.pkl\", \"wb\") as f:\n        pickle.dump(id_to_name, f)","metadata":{"execution":{"iopub.status.busy":"2023-10-20T00:26:21.432795Z","iopub.execute_input":"2023-10-20T00:26:21.433309Z","iopub.status.idle":"2023-10-20T00:26:21.457523Z","shell.execute_reply.started":"2023-10-20T00:26:21.433266Z","shell.execute_reply":"2023-10-20T00:26:21.456317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not config.is_submission:\n    class_weights = np.sum(labels) - np.sum(labels, axis=0)\n    class_weights = class_weights / np.sum(class_weights) # Normalize the weights\n\n    class_weights = {idx:weight for idx, weight in enumerate(class_weights)}\n\n    for idx, weight in class_weights.items():\n        print(f\"{id_to_name[idx]}: {weight:0.2f}\")","metadata":{"execution":{"iopub.status.busy":"2023-10-20T00:26:26.648686Z","iopub.execute_input":"2023-10-20T00:26:26.649079Z","iopub.status.idle":"2023-10-20T00:26:26.656905Z","shell.execute_reply.started":"2023-10-20T00:26:26.649049Z","shell.execute_reply":"2023-10-20T00:26:26.655524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#To decode tif files we need Tensorflow-io\n\nOriginal code on UBC-OCEAN competition has png files.","metadata":{}},{"cell_type":"code","source":"!pip install tensorflow-io","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-10-20T00:26:35.469455Z","iopub.execute_input":"2023-10-20T00:26:35.469834Z","iopub.status.idle":"2023-10-20T00:26:47.557346Z","shell.execute_reply.started":"2023-10-20T00:26:35.469805Z","shell.execute_reply":"2023-10-20T00:26:47.556262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#https://stackoverflow.com/questions/64006585/attributeerror-module-tensorflow-io-has-no-attribute-experimental\n\nimport tensorflow_io as tfio\ntfio.__version__\n# 0.15.0","metadata":{"execution":{"iopub.status.busy":"2023-10-20T00:26:51.557315Z","iopub.execute_input":"2023-10-20T00:26:51.557731Z","iopub.status.idle":"2023-10-20T00:26:51.603715Z","shell.execute_reply.started":"2023-10-20T00:26:51.557698Z","shell.execute_reply":"2023-10-20T00:26:51.602737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#https://stackoverflow.com/questions/64006585/attributeerror-module-tensorflow-io-has-no-attribute-experimental\n#https://stackoverflow.com/questions/41985509/no-tensorflow-decoder-for-tiff-images\n\nimport tensorflow as tf\n#import tensorflow.io as tfio #It's an underscore not a dot (It'stensorflow_io)\n\n\ndef read_image(path):\n    file = tf.io.read_file(path)\n    image = tfio.experimental.image.decode_tiff(file, 3)#Original was tf.io.decode_png(file, 3)\n    image = tf.image.resize(image, (256, 256))\n    image = tf.image.per_image_standardization(image)\n    return image","metadata":{"execution":{"iopub.status.busy":"2023-10-20T00:26:58.428465Z","iopub.execute_input":"2023-10-20T00:26:58.429279Z","iopub.status.idle":"2023-10-20T00:26:58.437019Z","shell.execute_reply.started":"2023-10-20T00:26:58.429231Z","shell.execute_reply":"2023-10-20T00:26:58.435527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not config.is_submission:\n    x = (\n        tf.data.Dataset.from_tensor_slices(image_thumbnail_paths)\n        .map(read_image, num_parallel_calls=tf.data.AUTOTUNE)\n    )\n    y = tf.data.Dataset.from_tensor_slices(labels)\n\n    # Zip the x and y together\n    ds = tf.data.Dataset.zip((x, y))\n    \n    # Create the training and validation splits\n    val_ds = (\n        ds\n        .take(50)\n        .batch(config.batch_size)\n        .prefetch(tf.data.AUTOTUNE)\n    )\n    train_ds = (\n        ds\n        .skip(50)\n        .shuffle(config.batch_size * 10)\n        .batch(config.batch_size)\n        .prefetch(tf.data.AUTOTUNE)\n    )","metadata":{"execution":{"iopub.status.busy":"2023-10-20T00:27:04.882054Z","iopub.execute_input":"2023-10-20T00:27:04.882473Z","iopub.status.idle":"2023-10-20T00:27:05.536887Z","shell.execute_reply.started":"2023-10-20T00:27:04.882443Z","shell.execute_reply":"2023-10-20T00:27:05.534724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Below, InvalidArgumentError:\n\n{{function_node __wrapped__DatasetToSingleElement_output_types_2_device_/job:localhost/replica:0/task:0/device:CPU:0}} unable to set TIFF directory to 3\n\t [[{{node IO>DecodeTiff}}]] [Op:DatasetToSingleElement]","metadata":{}},{"cell_type":"code","source":"if not config.is_submission:\n    images, labels = train_ds.take(1).get_single_element()\n\n    keras_cv.visualization.plot_image_gallery(\n        images,\n        value_range=(0, 1),\n        rows=2,\n        cols=2,\n    )","metadata":{"execution":{"iopub.status.busy":"2023-10-20T00:27:53.230163Z","iopub.execute_input":"2023-10-20T00:27:53.230697Z","iopub.status.idle":"2023-10-20T00:28:13.069609Z","shell.execute_reply.started":"2023-10-20T00:27:53.230655Z","shell.execute_reply":"2023-10-20T00:28:13.067932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#ResNet Model","metadata":{}},{"cell_type":"code","source":"# Load the image and text backbones with presets\nresnet_backbone = keras_cv.models.ResNetV2Backbone.from_preset(\n    \"resnet152_v2\",\n)\nresnet_backbone.trainable = False\n\nimage_inputs = resnet_backbone.input\nimage_embeddings = resnet_backbone(image_inputs)\nimage_embeddings = keras.layers.GlobalAveragePooling2D()(image_embeddings)\n\nx = keras.layers.BatchNormalization(epsilon=1e-05, momentum=0.1)(image_embeddings)\nx = keras.layers.Dense(units=1024, activation=\"relu\")(x)\nx = keras.layers.Dropout(0.1)(x)\nx = keras.layers.Dense(units=512, activation=\"relu\")(x)\nx = keras.layers.Dropout(0.1)(x)\nx = keras.layers.Dense(units=256, activation=\"relu\")(x)\noutputs = keras.layers.Dense(units=5, activation=\"softmax\")(x)\n\n# Build the model with the Functional API\nmodel = keras.Model(\n    inputs=image_inputs,\n    outputs=outputs,\n)\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-10-20T00:28:22.511778Z","iopub.execute_input":"2023-10-20T00:28:22.512174Z","iopub.status.idle":"2023-10-20T00:28:31.598168Z","shell.execute_reply.started":"2023-10-20T00:28:22.512117Z","shell.execute_reply":"2023-10-20T00:28:31.597456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Epochs. Turn on GPU!\n\nI didn't cause I got hat TIFF error.","metadata":{}},{"cell_type":"code","source":"if not config.is_submission:\n    model.compile(\n        optimizer=keras.optimizers.Adam(learning_rate=config.learning_rate),\n        loss=keras.losses.CategoricalCrossentropy(),\n        metrics=[\"accuracy\"],\n    )\n\n    history = model.fit(\n        train_ds,\n        epochs=config.epochs,\n        validation_data=val_ds,\n        class_weight=class_weights,\n    )\n    \n    model.save_weights(\"ucb_ocean_checkpoint.weights.h5\")","metadata":{"execution":{"iopub.status.busy":"2023-10-20T00:28:56.253262Z","iopub.execute_input":"2023-10-20T00:28:56.253632Z","iopub.status.idle":"2023-10-20T00:29:03.013805Z","shell.execute_reply.started":"2023-10-20T00:28:56.253606Z","shell.execute_reply":"2023-10-20T00:29:03.011087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Inference","metadata":{}},{"cell_type":"code","source":"if config.is_submission:\n    df = pd.read_csv(config.test_csv_path)\n    df[\"image_path\"] = df[\"image_id\"].apply(lambda x: f\"{config.test_thumbnail_paths}/{x}_thumbnail.png\")\n    \n    # Load the model weights\n    model.load_weights(\"/kaggle/input/kerascv-train-and-infer-on-thumbnails/ucb_ocean_checkpoint.weights.h5\")\n    \n    # Load the id to name dictionary\n    with open(\"/kaggle/input/kerascv-train-and-infer-on-thumbnails/id_to_name.pkl\", \"rb\") as f:\n        id_to_name = pickle.load(f)","metadata":{"execution":{"iopub.status.busy":"2023-10-20T00:29:29.799676Z","iopub.execute_input":"2023-10-20T00:29:29.800107Z","iopub.status.idle":"2023-10-20T00:29:29.808107Z","shell.execute_reply.started":"2023-10-20T00:29:29.800076Z","shell.execute_reply":"2023-10-20T00:29:29.806577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if config.is_submission:\n    predicted_labels = []\n\n    for index, row in df.iterrows():\n        # Get the image path\n        image_path = row[\"image_path\"]\n\n        # Get the image\n        image = read_image(image_path)[None, ...]\n\n        # Predict the label\n        logits = model.predict(image)\n        pred = ops.argmax(logits, axis=-1).tolist()[0]\n\n        # Map the pred to the name\n        label = id_to_name[pred]\n\n        predicted_labels.append(label)\n\n    # Add the predicted labels to the csv\n    df[\"label\"] = predicted_labels","metadata":{"execution":{"iopub.status.busy":"2023-10-20T00:30:01.510049Z","iopub.execute_input":"2023-10-20T00:30:01.510565Z","iopub.status.idle":"2023-10-20T00:30:01.517686Z","shell.execute_reply.started":"2023-10-20T00:30:01.510439Z","shell.execute_reply":"2023-10-20T00:30:01.516296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##Jirca's Borovec\n\ndisplay(df.head())\ndf[[\"image_id\", \"label\"]].to_csv(\"submission.csv\", index=False)\n\n! head submission.csv","metadata":{"execution":{"iopub.status.busy":"2023-10-20T00:30:26.129114Z","iopub.execute_input":"2023-10-20T00:30:26.12953Z","iopub.status.idle":"2023-10-20T00:30:27.344691Z","shell.execute_reply.started":"2023-10-20T00:30:26.129501Z","shell.execute_reply":"2023-10-20T00:30:27.343186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Oh Boy! Oh Dear! Oh Crap! Another code that I ruined! \n\nI couldn't find any answer on my searchs. Though I'll keep trying. ","metadata":{}},{"cell_type":"markdown","source":"![](https://miro.medium.com/v2/resize:fit:1204/1*ic1zTLBPAWFhtefyP6IGhw.jpeg)Pallawi- Medium","metadata":{}},{"cell_type":"markdown","source":"#Acknowledgements:\n\nAritra Roy Gosthipaty, Martin Görner, Gusthema and Phil Culliton\n\nhttps://www.kaggle.com/code/aritrag/kerascv-train-and-infer-on-thumbnails","metadata":{}}]}