{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing , CSV file I/O (e.g. pd.read_csv)\nimport os\nimport cv2\nimport tifffile\nimport gc\nfrom time import sleep\nfrom tqdm import tqdm\n\nfrom keras import layers\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.models import Sequential\nfrom keras.layers.normalization.batch_normalization import BatchNormalization\nfrom keras.layers.convolutional import Conv2D\nfrom keras.layers.convolutional import MaxPooling2D\nfrom keras.layers.convolutional import SeparableConv2D\nfrom keras.layers.core import Activation\nfrom keras.layers.core import Flatten\nfrom keras.layers.core import Dropout\nfrom keras.layers.core import Dense\nfrom keras.layers import UpSampling2D\nfrom keras import backend as k\nimport keras\nimport tensorflow as tf\n\nfrom tensorflow.keras.optimizers import SGD\nfrom keras.callbacks import EarlyStopping\nfrom keras.callbacks import ModelCheckpoint\n\nfrom sklearn.metrics import classification_report\nimport matplotlib.pyplot as plt\n\nfrom collections import defaultdict\nfrom openslide import OpenSlide","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Initialization of required variables here.**","metadata":{}},{"cell_type":"code","source":"# initialize variables here\nis_resize_image = False\nsave_image_size = (256, 256)\ntarget_image_size = (256, 256)  # 1024x1024 512x512 will exceed the memory limit allocated by Kaggle\n\nresized_image_path = {'train': \"./train/\", 'test':\"./test/\"}\n\ninput_images_path = \"../input/mayo-clinic-strip-preprocess-image-256x256-128/train/filtered/\"\ntest_images_path = \"../input/mayo-clinic-strip-preprocess-image-256x256-128/test/filtered/\"\n\nmodel_dir = \"./model/\"\nfull_model_path = os.path.join(model_dir, \"efficientnetb0_strip_ai.hdf5\")\nweight_path = os.path.join(model_dir, \"efficientnetb0_strip_ai.hdf5\")\n\nif is_resize_image:\n    input_images_path = \"./train/\"\n    test_images_path = \"./test/\"\n    \nbatch_size = 32\ndata_pipeline = \"tensorflow_dataset\"  # image_data_generator","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **UNet**","metadata":{}},{"cell_type":"code","source":"def DoubleSeparableConv2D(unet_layers, filters):\n    unet_layers = Activation(\"relu\")(unet_layers)\n    unet_layers = SeparableConv2D(filters, (3, 3), padding=\"same\")(unet_layers)\n    unet_layers = BatchNormalization()(unet_layers)\n\n    unet_layers = Activation(\"relu\")(unet_layers)\n    unet_layers = SeparableConv2D(filters, (3, 3), padding=\"same\")(unet_layers)\n    unet_layers = BatchNormalization()(unet_layers)\n    \n    return unet_layers\n\ndef DoubleConv2DTranspose(unet_layers, filters):\n    unet_layers = Activation(\"relu\")(unet_layers)\n    unet_layers = SeparableConv2D(filters, (3, 3), padding=\"same\")(unet_layers)\n    unet_layers = BatchNormalization()(unet_layers)\n\n    unet_layers = Activation(\"relu\")(unet_layers)\n    unet_layers = SeparableConv2D(filters, (3, 3), padding=\"same\")(unet_layers)\n    unet_layers = BatchNormalization()(unet_layers)\n    \n    return unet_layers","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class UNet:\n    @staticmethod\n    def build(width, height, depth, classes):\n        # initialize the input dimension\n        input_shape = (height, width, depth)\n        chan_dim = -1\n        # if we are using \"channel first\", update the input shape\n        if k.image_data_format() == \"channel_first\":\n            input_shape = (depth, height, width)\n            chan_dim = 1\n        \n        inputs = layers.Input(shape=input_shape)\n        \n        unet_layers = Conv2D(32, (3, 3), strides=(2, 2), padding=\"same\")(inputs)\n        unet_layers = layers.BatchNormalization()(unet_layers)\n        unet_layers = layers.Activation(\"relu\")(unet_layers)\n        \n        previous_block = unet_layers  # Set aside residual\n        \n        # downsampling\n        for filters in [64, 128, 256]:\n            \n            unet_layers = DoubleSeparableConv2D(unet_layers, filters)\n            unet_layers = MaxPooling2D(pool_size=(3, 3), strides=(2, 2), padding=\"same\")(unet_layers)\n\n            # Project residual\n            residual = Conv2D(filters, (1, 1), strides=(2, 2), padding=\"same\")(previous_block)\n            unet_layers = layers.add([unet_layers, residual])  # Add back residual\n            previous_block = unet_layers  # Set aside next residual\n            \n        for filters in [256, 128, 64, 32]:\n            \n            unet_layers = DoubleConv2DTranspose(unet_layers, filters)\n            unet_layers = UpSampling2D((2, 2))(unet_layers)\n\n            # Project residual\n            residual = UpSampling2D((2, 2))(previous_block)\n            residual = Conv2D(filters, (1, 1), padding=\"same\")(residual)\n            unet_layers = layers.add([unet_layers, residual])  # Add back residual\n            previous_block = unet_layers  # Set aside next residual\n\n        outputs = Conv2D(3, (3, 3), activation=\"softmax\", padding=\"same\")(unet_layers)\n        \n        model = keras.Model(inputs, outputs, name=\"U-Net\")\n        \n        return model","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Function**","metadata":{}},{"cell_type":"code","source":"def labels_to_dataset(labels, label_mode, num_classes):\n    \"\"\"Create a tf.data.Dataset from the list/tuple of labels.\n    Args:\n    labels: list/tuple of labels to be converted into a tf.data.Dataset.\n    label_mode: String describing the encoding of `labels`. Options are:\n    - 'binary' indicates that the labels (there can be only 2) are encoded as\n      `float32` scalars with values 0 or 1 (e.g. for `binary_crossentropy`).\n    - 'categorical' means that the labels are mapped into a categorical vector.\n      (e.g. for `categorical_crossentropy` loss).\n    num_classes: number of classes of labels.\n    Returns:\n    A `Dataset` instance.\n    \"\"\"\n    label_ds = tf.data.Dataset.from_tensor_slices(labels)\n    print(label_ds)\n    if label_mode == 'binary':\n        label_ds = label_ds.map(lambda x: tf.expand_dims(tf.cast(x, 'float32'), axis=-1), num_parallel_calls=tf.data.AUTOTUNE)\n    elif label_mode == 'categorical':\n        label_ds = label_ds.map(lambda x: tf.one_hot(x, num_classes), num_parallel_calls=tf.data.AUTOTUNE)\n    return label_ds\n\ndef parse_image_with_label(filename, label):\n    image = tf.io.read_file(filename)\n    image = tf.io.decode_jpeg(image, channels=3)\n    image = tf.image.convert_image_dtype(image, tf.float32)\n    image.set_shape((target_image_size[0], target_image_size[1], 3))\n    return image, label\n\ndef parse_image(filename):\n    image = tf.io.read_file(filename)\n    image = tf.io.decode_jpeg(image, channels=3)\n    image = tf.image.convert_image_dtype(image, tf.float32)\n    image.set_shape((target_image_size[0], target_image_size[1], 3))\n    return image","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Load the training datasets**","metadata":{}},{"cell_type":"code","source":"# load the images and labels\ntrain_data = pd.read_csv(\"../input/mayo-clinic-strip-ai/train.csv\")\ntest_data = pd.read_csv(\"../input/mayo-clinic-strip-ai/test.csv\")\n\nsliced_train_data = pd.read_csv(\"../input/mayo-clinic-strip-preprocess-image-256x256-256/sliced_train.csv\")\nsliced_test_data = pd.read_csv(\"../input/mayo-clinic-strip-preprocess-image-256x256-256/sliced_test.csv\")\n\nclass_labels = [str(x) for x in np.unique(train_data['label'])]\n\nsliced_train_size = len(sliced_train_data)\nsliced_test_size = len(sliced_test_data)\n\nprint(train_data.head())\nprint(class_labels) ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def append_ext(fn):\n    return fn+\".jpg\"\n\n\nif data_pipeline == \"image_data_generator\":\n    train_data.insert(1, \"image_id_path\", train_data[\"image_id\"].apply(append_ext))\n    test_data.insert(1, \"image_id_path\", test_data[\"image_id\"].apply(append_ext))\n\n    sliced_train_data.insert(2, \"image_slice_id_path\", sliced_train_data[\"image_slice_id\"].apply(append_ext))\n    sliced_test_data.insert(2, \"image_slice_id_path\", sliced_test_data[\"image_slice_id\"].apply(append_ext))\n\nif data_pipeline == \"tensorflow_dataset\":\n    train_data.insert(1, \"image_id_path\", train_data[\"image_id\"].apply(append_ext))\n    test_data.insert(1, \"image_id_path\", test_data[\"image_id\"].apply(append_ext))\n    \n    sliced_train_data.insert(2, \"image_slice_id_path\", sliced_train_data[\"image_slice_id\"].apply(append_ext))\n    sliced_test_data.insert(2, \"image_slice_id_path\", sliced_test_data[\"image_slice_id\"].apply(append_ext))\n    \n    #sliced_train_data['image_slice_id_fullpath'] = sliced_train_data['image_slice_id_path'].map(lambda x: os.path.join(input_images_path, x))\n    #sliced_test_data['image_slice_id_fullpath'] = sliced_test_data['image_slice_id_path'].map(lambda x: os.path.join(input_images_path, x))\n\n\nprint((train_data.head()))\nprint((test_data.head()))\nprint((sliced_train_data.head()))\nprint((sliced_test_data.head()))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **TensorFlow Dataset**","metadata":{}},{"cell_type":"code","source":"if data_pipeline == \"tensorflow_dataset\":\n    normalization_layer = tf.keras.layers.Rescaling(1./255)\n    test_tfdataset = tf.keras.utils.image_dataset_from_directory(directory=test_images_path,\n                                                                 labels=None, label_mode=None, seed=42,\n                                                                 image_size=target_image_size, batch_size=batch_size,\n                                                                 shuffle=False)\n    file_paths = test_tfdataset.file_paths\n    # get the filenames\n    filenames = list([os.path.split(i)[1] for i in file_paths])\n\n\n  \n    test_tfdataset = test_tfdataset.map(lambda x: (normalization_layer(x)))\n    \n    training_tfdataset = tf.keras.utils.image_dataset_from_directory(directory=input_images_path,\n                                                                     validation_split=0.2, subset=\"training\", seed=42,\n                                                                     image_size=target_image_size, batch_size=batch_size,\n                                                                     shuffle=True, label_mode=\"categorical\")\n    class_names = training_tfdataset.class_names # returns list()\n    labels = dict()\n    for i in range(len(class_names)):\n        labels[i] = class_names[i]\n        \n    training_tfdataset = training_tfdataset.map(lambda x, y: (normalization_layer(x), y))\n    \n    validation_tfdataset = tf.keras.utils.image_dataset_from_directory(directory=input_images_path,\n                                                                       validation_split=0.2, subset=\"validation\", seed=42,\n                                                                       image_size=target_image_size, batch_size=batch_size, \n                                                                       shuffle=True, label_mode=\"categorical\")\n    \n    validation_tfdataset = validation_tfdataset.map(lambda x, y: (normalization_layer(x), y))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if False:\n    epochs = 1\n    image_batch = next(iter(test_tfdataset))\n    first_image = image_batch[0]\n    # Notice the pixel values are now in `[0,1]`.\n    print(np.min(first_image), np.max(first_image))\n\n    def show_sliced_image(image):\n        plt.figure()\n        plt.imshow(image)\n        plt.axis('off')\n        \n    show_sliced_image(first_image)\n    \n    #class_names = training_tfdataset.class_names\n    print(class_names)\n    print(labels)\n    print(len(file_paths))\n    print(file_paths[0:3])\n    filenames = list([os.path.split(i)[1] for i in file_paths])\n    print(filenames[0:3])\n    print(len(filenames))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Image Data Generator**","metadata":{}},{"cell_type":"code","source":"if data_pipeline == \"image_data_generator\":\n    image_data_gen=ImageDataGenerator(rescale=1.0/255.0,validation_split=0.25)\n    test_data_gen=ImageDataGenerator(rescale=1.0/255.0)\n\n    training_generator=image_data_gen.flow_from_dataframe(\n        dataframe=sliced_train_data, directory=input_images_path, x_col=\"image_slice_id_path\", y_col=\"label\",\n        subset=\"training\", batch_size=batch_size, seed=42, shuffle=True, class_mode=\"categorical\",\n        target_size=target_image_size)\n    validation_generator=image_data_gen.flow_from_dataframe(\n        dataframe=sliced_train_data, directory=input_images_path, x_col=\"image_slice_id_path\", y_col=\"label\",\n        subset=\"validation\", batch_size=batch_size, seed=42, shuffle=True, class_mode=\"categorical\",\n        target_size=target_image_size)\n\n    test_generator=test_data_gen.flow_from_dataframe(\n        dataframe=sliced_test_data, directory=test_images_path, x_col=\"image_slice_id_path\", y_col=None,\n        batch_size=batch_size, seed=42, shuffle=False, class_mode=None,target_size=target_image_size)\n    \n    filenames = test_generator.filenames\n    labels = (training_generator.class_indices) # returns dict()\n    print(labels)\n    labels = dict((v, k) for k, v in labels.items())\n    print(labels)\n    print(training_generator)\n    print(validation_generator)\n    print(test_generator)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Build the network**","metadata":{}},{"cell_type":"code","source":"# train the model using SGD\ntry:\n    os.mkdir(model_dir)\nexcept:\n    pass\nlearning_decay = 0.1 / 40 \nepochs = 100\npatience = max(10, np.ceil(epochs * 0.20).astype(int))\n\n# build the model\nsgd = SGD(learning_rate=0.01, decay=learning_decay, momentum=0.9, nesterov=True)\n#model = UNet.build(width=target_image_size[0], height=target_image_size[1], depth=3, \n#                         classes=len(class_labels))\n\nfrom keras.applications.xception import Xception\n\nfrom keras.applications.resnet import ResNet50\n\nfrom keras.applications.efficientnet import EfficientNetB0\n\nmodel = EfficientNetB0(input_shape=(256, 256, 3), classes=2, weights=None)\n\nmodel.compile(loss=\"categorical_crossentropy\", optimizer=sgd, metrics=[\"accuracy\"])\n\ncheckpoint = ModelCheckpoint(filepath=full_model_path, save_weights_only=False,\n                             monitor=\"val_loss\", mode=\"min\", save_best_only=True, verbose=1)\n\nearly_stopping = EarlyStopping(monitor = 'val_loss', patience = patience )\n\ncallbacks = [checkpoint, early_stopping]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Train the network using TensorFlow Dataset**","metadata":{}},{"cell_type":"code","source":"if data_pipeline == \"tensorflow_dataset\":\n   # step_per_epoch_training = train_size // batch_size\n   # validation_steps = val_size // batch_size\n   # test_steps = sliced_test_size // batch_size\n\n    # train the network\n    H = model.fit(training_tfdataset, validation_data=validation_tfdataset,\n                  callbacks=callbacks,\n                  epochs=epochs, verbose=1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Train the network using image generator**","metadata":{}},{"cell_type":"code","source":"if data_pipeline == \"image_data_generator\":\n    \n    step_per_epoch_training = training_generator.n // training_generator.batch_size\n    validation_steps = validation_generator.n // validation_generator.batch_size\n    test_steps = test_generator.n // test_generator.batch_size\n\n    # train the network\n    H = model.fit(training_generator, validation_data=validation_generator,\n                  steps_per_epoch=step_per_epoch_training, callbacks=callbacks,\n                  validation_steps=validation_steps, \n                  epochs=epochs, verbose=1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(H.history.keys())\nepochs_ran = len(H.history['loss'])\n# plot the training loss and accuracy\nplt.style.use(\"ggplot\")\nplt.figure()\nplt.plot(np.arange(0, epochs_ran), H.history[\"loss\"], label=\"train_loss\")\nplt.plot(np.arange(0, epochs_ran), H.history[\"accuracy\"], label=\"accuracy\")\nplt.title(\"Training Loss and Accuracy\")\nplt.xlabel(\"Epoch #\")\nplt.ylabel(\"Loss/Accuracy\")\nplt.legend()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(H.history.keys())\nepochs_ran = len(H.history['loss'])\n# plot the training loss and accuracy\nplt.style.use(\"ggplot\")\nplt.figure()\n#plt.plot(np.arange(0, epochs_ran), H.history[\"loss\"], label=\"train_loss\")\nplt.plot(np.arange(0, epochs_ran), H.history[\"val_loss\"], label=\"val_loss\")\n#plt.plot(np.arange(0, epochs_ran), H.history[\"accuracy\"], label=\"accuracy\")\nplt.plot(np.arange(0, epochs_ran), H.history[\"val_accuracy\"], label=\"val_accuracy\")\nplt.title(\"Training Loss and Accuracy\")\nplt.xlabel(\"Epoch #\")\nplt.ylabel(\"Loss/Accuracy\")\nplt.legend()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(H.history.keys())\nepochs_ran = len(H.history['loss'])\n# plot the training loss and accuracy\nplt.style.use(\"ggplot\")\nplt.figure()\nplt.plot(np.arange(0, epochs_ran), H.history[\"loss\"], label=\"train_loss\")\nplt.plot(np.arange(0, epochs_ran), H.history[\"val_loss\"], label=\"val_loss\")\nplt.plot(np.arange(0, epochs_ran), H.history[\"accuracy\"], label=\"accuracy\")\nplt.plot(np.arange(0, epochs_ran), H.history[\"val_accuracy\"], label=\"val_accuracy\")\nplt.title(\"Training Loss and Accuracy\")\nplt.xlabel(\"Epoch #\")\nplt.ylabel(\"Loss/Accuracy\")\nplt.legend()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Evaluate the model**","metadata":{}},{"cell_type":"code","source":"if data_pipeline == \"image_data_generator\":\n    predictions = model.predict(test_generator, batch_size=batch_size)\n    ce, laa = predictions[:, 0], predictions[:, 1]\n    print(predictions)   \n    \nif data_pipeline == \"tensorflow_dataset\":\n    predictions = model.predict(test_tfdataset, batch_size=batch_size)\n    ce, laa = predictions[:, 0], predictions[:, 1]\n    print(predictions)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Evaluate the model**","metadata":{}},{"cell_type":"code","source":"# Save to csv\nprint(\"[INFO] predictions size: {}\".format(len(predictions)))\nprint(\"[INFO] predictions shape: {}\".format(predictions.shape))\n\npredicted_class_indices = np.argmax(predictions, axis=1)\nprint(\"[INFO] predicted_class_indices shape: {}\".format(predicted_class_indices.shape))\nprint(predicted_class_indices)\npredictions_label = [labels[k] for k in predicted_class_indices]\n\n# Match the filenames with the inference result\nprint(\"[INFO] filenames size: {}\".format(len(filenames)))\nresults = pd.DataFrame({\"image_slice_id_path\":filenames, \"label\":predictions_label, 'ce': ce, 'laa': laa})\nprint(\"[INFO] store the result to the dataframe\")\nprint(results.head(14))\n\nprint(\"[INFO] merge the result with the sliced images dataset\")\nprediction_sliced_test_data = sliced_test_data.merge(results, on='image_slice_id_path')\nprint(prediction_sliced_test_data[[\"image_slice_id\", \"patient_id\", \"label\", \"ce\", \"laa\"]].head(14))\n\n## use mean for the meantime to compute the final prediction of patient's sample lab result\nprint(\"[INFO] compute the mean of images belong to the same patient_id\")\nprediction_result = prediction_sliced_test_data[['patient_id','ce', 'laa']].groupby(['patient_id']).mean()\nprint(prediction_result)\n\nprint(\"[INFO] read sample submission data\")\nsubmission = pd.read_csv(\"../input/mayo-clinic-strip-ai/sample_submission.csv\")\nprint(submission)\n\nsubmission.patient_id = prediction_result.index.to_list()\nsubmission.CE = prediction_result.ce.to_list()\nsubmission.LAA = prediction_result.laa.to_list()\n\nprint(\"[INFO] save submission data\")\nprint(submission)\nsubmission.to_csv(\"sample_submission.csv\", index = False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}