{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Import necessary packages**","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# Input data files are available in the read-only \"../input/\" directory\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output \n# when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the \n# current session\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\n\nimport cv2\nimport tifffile\nimport gc\nfrom time import sleep\nfrom tqdm import tqdm\n\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.models import Sequential\nfrom keras.layers.normalization.batch_normalization import BatchNormalization\nfrom keras.layers.convolutional import Conv2D\nfrom keras.layers.convolutional import MaxPooling2D\nfrom keras.layers.core import Activation\nfrom keras.layers.core import Flatten\nfrom keras.layers.core import Dropout\nfrom keras.layers.core import Dense\nfrom keras import backend as k\n\nfrom tensorflow.keras.optimizers import SGD\nfrom keras.callbacks import EarlyStopping\nfrom keras.callbacks import ModelCheckpoint\n\nfrom sklearn.metrics import classification_report\nimport matplotlib.pyplot as plt\n\nfrom collections import defaultdict\nfrom openslide import OpenSlide","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Initialization of required variables here.**","metadata":{}},{"cell_type":"code","source":"# initialize variables here\nis_resize_image = False\nsave_image_size = (1024, 1024)\ntarget_image_size = (256, 256)  # 1024x1024 512x512 will exceed the memory limit allocated by Kaggle\n\nresized_image_path = {'train': \"./train/\", 'test':\"./test/\"}\n\ninput_images_path = \"../input/mayo-clinic-strip-ai-preprocess-image/train/\"\ntest_images_path = \"../input/mayo-clinic-strip-ai-preprocess-image/test/\"\n\nmodel_dir = \"./model/\"\nfull_model_path = os.path.join(model_dir, \"minivggnet_strip_ai.hdf5\")\nweight_path = os.path.join(model_dir, \"weight_strip_ai.hdf5\")\n\nif is_resize_image:\n    input_images_path = \"./train/\"\n    test_images_path = \"./test/\"\n    \nbatch_size = 32","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Log Loss Function**","metadata":{}},{"cell_type":"code","source":"class WeightedMultiClassLogarithmicLoss():\n    def __init__(self, epsilon=10e-15, weights=np.ones(2, dtype='float64'), class_proportions=np.ones(2, dtype='float64')):\n        self.epsilon = epsilon\n        self.weights = weights\n        self.class_proportions = class_proportions\n        #super().__init__()\n    \n    def call(self, y_true, y_pred):\n        print(self.weights)\n        print(self.class_proportions)\n        print(self.epsilon)\n        # probability 𝑝 is replaced with epsilon to avoid extremes of the log function\n        y_pred_clipped = k.clip(y_pred, self.epsilon, 1-self.epsilon)\n        # ground-truth labels are weighted\n        y_true_weighted = (y_true * self.weights)/self.class_proportions\n        # compute the log loss\n        loss = -1*( k.sum(y_true_weighted * k.log(y_pred_clipped)) / k.sum(self.weights) )\n        return loss","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **MiniVGGNet**","metadata":{}},{"cell_type":"code","source":"# From Deep Learning For Computer Vision Start Bundle (p. 229-233), \n# by Rosebrock, Adrian, 2017, PyImageSearch \n# (https://www.pyimagesearch.com/deep-learning-computer-vision-python-book). \n# Copyright 2017 by Adrian Rosebrock, PyImageSearch.com. \nclass MiniVGGNet:\n    @staticmethod\n    def build(width, height, depth, classes):\n        # initialize the model\n        model = Sequential()\n        input_shape = (height, width, depth)\n        chan_dim = -1\n        # if we are using \"channel first\", update the input shape\n        if k.image_data_format() == \"channel_first\":\n            input_shape = (depth, height, width)\n            chan_dim = 1\n\n        # first set of CONV => RELU => CONV => RELU => POOL layers\n        model.add(Conv2D(32, (3, 3), padding=\"same\", input_shape=input_shape))\n        model.add(Activation(\"relu\"))\n        model.add(BatchNormalization(axis=chan_dim))\n        model.add(Conv2D(32, (3, 3), padding=\"same\"))\n        model.add(Activation(\"relu\"))\n        model.add(BatchNormalization(axis=chan_dim))\n        model.add(MaxPooling2D(pool_size=(2, 2)))\n        model.add(Dropout(0.25))\n\n        # second set of CONV => RELU => CONV => RELU => POOL layers\n        model.add(Conv2D(64, (3, 3), padding=\"same\"))\n        model.add(Activation(\"relu\"))\n        model.add(BatchNormalization(axis=chan_dim))\n        model.add(Conv2D(64, (3, 3), padding=\"same\"))\n        model.add(Activation(\"relu\"))\n        model.add(BatchNormalization(axis=chan_dim))\n        model.add(MaxPooling2D(pool_size=(2, 2)))\n        model.add(Dropout(0.25))\n\n        # first and only set of FC => RELU layers\n        model.add(Flatten())\n        model.add(Dense(512))\n        model.add(Activation(\"relu\"))\n        model.add(BatchNormalization())\n\n        # softmax classifier\n        model.add(Dense(classes))\n        model.add(Activation(\"softmax\"))\n\n        return model\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Load the training datasets**","metadata":{}},{"cell_type":"code","source":"# load the images and labels\ntrain_data = pd.read_csv(\"../input/mayo-clinic-strip-ai/train.csv\")\ntest_data = pd.read_csv(\"../input/mayo-clinic-strip-ai/test.csv\")\n\nsliced_train_data = pd.read_csv(\"../input/mayo-clinic-strip-ai-preprocess-image/sliced_train.csv\")\nsliced_test_data = pd.read_csv(\"../input/mayo-clinic-strip-ai-preprocess-image/sliced_test.csv\")\n\nclass_labels = [str(x) for x in np.unique(train_data['label'])]\n\nprint(train_data.head())\nprint(class_labels)  ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **For debugging**","metadata":{}},{"cell_type":"code","source":"# test only the batch size and recompute the new batch size\nif False:\n    train_data = train_data.head(batch_size)\n    batch_size = batch_size // 8\n    input_images_path = \"./train/\"\n    test_images_path = \"./test/\"\n    is_resize_image = False\nprint(train_data.head(4))\n\nif False:\n    \n    from sklearn.utils import shuffle\n    from sklearn.utils import resample\n    print(sliced_train_data)\n    \n    #sliced_train_data = shuffle(sliced_train_data)\n    #print(sliced_train_data)\n    #sliced_train_data.reset_index(inplace=True, drop=True)\n    \n    sliced_train_data = resample(sliced_train_data, n_samples=30, replace=False, stratify=sliced_train_data, random_state=0 )\n    sliced_train_data.reset_index(inplace=True, drop=True)\n    print(sliced_train_data)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Solution to overconsumption of memory**","metadata":{}},{"cell_type":"code","source":"if False:\n    \n    from sklearn.utils import shuffle\n    from sklearn.utils import resample\n    print(sliced_train_data)\n    \n    sliced_train_data = resample(sliced_train_data, n_samples=4000, replace=False, stratify=sliced_train_data, random_state=0 )\n    sliced_train_data.reset_index(inplace=True, drop=True)\n    print(sliced_train_data)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Resize Function**","metadata":{}},{"cell_type":"code","source":"def resize_images_v1(data, size=(128,128), input_path=None, output_path=None):\n    # copy tif image to jpg in [512,512]\n    if input_path is None:\n        input_path = \"../input/mayo-clinic-strip-ai/test/\"\n    if output_path is None:\n        output_path = \"./train/\"\n    try:\n        os.mkdir(output_path)\n    except:\n        pass\n\n    with tqdm(total=len(data['image_id'])) as pbar:\n        for x in data['image_id']:       \n            img_id = x\n            img = cv2.resize(tifffile.imread(\n                os.path.join(input_path, \"{}.tif\".format(img_id))), size)\n            cv2.imwrite(os.path.join(output_path, \"{}.jpg\".format(img_id)), img)\n            del img\n            gc.collect()\n\n            pbar.write('processed: %s' %x)\n            pbar.update(1)\n            sleep(1)\n    return None","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if is_resize_image:\n    resize_images_v1(train_data, size=save_image_size, \n                  input_path=\"../input/mayo-clinic-strip-ai/train/\",\n                  output_path=resized_image_path['train'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if is_resize_image:\n    resize_images_v1(test_data, size=save_image_size, \n                  input_path=\"../input/mayo-clinic-strip-ai/test/\",\n                  output_path=resized_image_path['test'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def append_ext(fn):\n    return fn+\".jpg\"\n\ntrain_data.insert(1, \"image_id_path\", train_data[\"image_id\"].apply(append_ext))\ntest_data.insert(1, \"image_id_path\", test_data[\"image_id\"].apply(append_ext))\n\nsliced_train_data.insert(2, \"image_slice_id_path\", sliced_train_data[\"image_slice_id\"].apply(append_ext))\nsliced_test_data.insert(2, \"image_slice_id_path\", sliced_test_data[\"image_slice_id\"].apply(append_ext))\n\nprint((train_data.head()))\nprint((test_data.head()))\nprint((sliced_train_data.head()))\nprint((sliced_test_data.head()))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Data Generator**","metadata":{}},{"cell_type":"code","source":"image_data_gen=ImageDataGenerator(rescale=1.0/255.0,validation_split=0.25)\ntest_data_gen=ImageDataGenerator(rescale=1.0/255.0)\n\ntraining_generator=image_data_gen.flow_from_dataframe(\n    dataframe=sliced_train_data, directory=input_images_path, x_col=\"image_slice_id_path\", y_col=\"label\",\n    subset=\"training\", batch_size=batch_size, seed=42, shuffle=True, class_mode=\"categorical\",\n    target_size=target_image_size)\nvalidation_generator=image_data_gen.flow_from_dataframe(\n    dataframe=sliced_train_data, directory=input_images_path, x_col=\"image_slice_id_path\", y_col=\"label\",\n    subset=\"validation\", batch_size=batch_size, seed=42, shuffle=True, class_mode=\"categorical\",\n    target_size=target_image_size)\n\ntest_generator=test_data_gen.flow_from_dataframe(\n    dataframe=sliced_test_data, directory=test_images_path, x_col=\"image_slice_id_path\", y_col=None,\n    batch_size=batch_size, seed=42, shuffle=False, class_mode=None,target_size=target_image_size)\n\nprint(training_generator)\nprint(validation_generator)\nprint(test_generator)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Train the network**","metadata":{}},{"cell_type":"code","source":"# train the model using SGD\ntry:\n    os.mkdir(model_dir)\nexcept:\n    pass\nlearning_decay = 0.1 / 40 \nepochs = 100\n\n# build the model\nsgd = SGD(learning_rate=0.01, decay=learning_decay, momentum=0.9, nesterov=True)\nmodel = MiniVGGNet.build(width=target_image_size[0], height=target_image_size[1], depth=3, \n                         classes=len(class_labels))\n\n# define the loss\nweights = np.ones(2, dtype='float64')\nweights[0] = 0.5\nweights[1] = 0.5\nclass_counts = train_data.groupby('label')['image_id'].count().values \nclass_proportions = class_counts/np.max(class_counts)\nk.set_floatx('float64')\nlog_loss = WeightedMultiClassLogarithmicLoss(epsilon=10e-15, weights=weights, class_proportions=class_proportions)\n\n# compile the model\n#model.compile(loss=log_loss, optimizer=sgd, metrics=[\"accuracy\"])\nmodel.compile(loss=\"categorical_crossentropy\", optimizer=sgd, metrics=[\"accuracy\"])\n\nstep_per_epoch_training = training_generator.n // training_generator.batch_size\nvalidation_steps = validation_generator.n // validation_generator.batch_size\ntest_steps = test_generator.n // test_generator.batch_size\n\ncheckpoint = ModelCheckpoint(filepath=full_model_path, save_weights_only=False,\n                             monitor=\"val_loss\", mode=\"min\", save_best_only=True, verbose=1)\n\nearly_stopping = EarlyStopping(monitor = 'val_loss', patience = np.ceil(epochs * 0.20).astype(int) )\ncallbacks = [checkpoint, early_stopping]\n\n# train the network\nH = model.fit(training_generator, validation_data=validation_generator,\n              steps_per_epoch=step_per_epoch_training, callbacks=callbacks,\n              validation_steps=validation_steps, \n              epochs=epochs, verbose=1)\n\n# The model weights (that are considered the best) are re-loaded into the model.\n#model.load_weights(weight_path)\n# save the network\n#model.save(full_model_path)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(H.history.keys())\nepochs_ran = len(H.history['loss'])\n# plot the training loss and accuracy\nplt.style.use(\"ggplot\")\nplt.figure()\nplt.plot(np.arange(0, epochs_ran), H.history[\"loss\"], label=\"train_loss\")\nplt.plot(np.arange(0, epochs_ran), H.history[\"accuracy\"], label=\"accuracy\")\nplt.title(\"Training Loss and Accuracy\")\nplt.xlabel(\"Epoch #\")\nplt.ylabel(\"Loss/Accuracy\")\nplt.legend()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(H.history.keys())\nepochs_ran = len(H.history['loss'])\n# plot the training loss and accuracy\nplt.style.use(\"ggplot\")\nplt.figure()\n#plt.plot(np.arange(0, epochs_ran), H.history[\"loss\"], label=\"train_loss\")\nplt.plot(np.arange(0, epochs_ran), H.history[\"val_loss\"], label=\"val_loss\")\n#plt.plot(np.arange(0, epochs_ran), H.history[\"accuracy\"], label=\"accuracy\")\nplt.plot(np.arange(0, epochs_ran), H.history[\"val_accuracy\"], label=\"val_accuracy\")\nplt.title(\"Training Loss and Accuracy\")\nplt.xlabel(\"Epoch #\")\nplt.ylabel(\"Loss/Accuracy\")\nplt.legend()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(H.history.keys())\nepochs_ran = len(H.history['loss'])\n# plot the training loss and accuracy\nplt.style.use(\"ggplot\")\nplt.figure()\nplt.plot(np.arange(0, epochs_ran), H.history[\"loss\"], label=\"train_loss\")\nplt.plot(np.arange(0, epochs_ran), H.history[\"val_loss\"], label=\"val_loss\")\nplt.plot(np.arange(0, epochs_ran), H.history[\"accuracy\"], label=\"accuracy\")\nplt.plot(np.arange(0, epochs_ran), H.history[\"val_accuracy\"], label=\"val_accuracy\")\nplt.title(\"Training Loss and Accuracy\")\nplt.xlabel(\"Epoch #\")\nplt.ylabel(\"Loss/Accuracy\")\nplt.legend()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Evaluate the model**","metadata":{}},{"cell_type":"code","source":"predictions = model.predict(test_generator, batch_size=batch_size)\nce, laa = predictions[:, 0], predictions[:, 1]\nprint(predictions)\n\n#print(classification_report(test_generator, predictions.argmax(axis=1), target_names=class_labels))\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save to csv\n\nprint(\"[INFO] predictions size: {}\".format(len(predictions)))\nprint(\"[INFO] predictions shape: {}\".format(predictions.shape))\n\npredicted_class_indices = np.argmax(predictions, axis=1)\nprint(\"[INFO] predicted_class_indices shape: {}\".format(predicted_class_indices.shape))\nprint(predicted_class_indices)\nlabels = (training_generator.class_indices)  # returns dict\nprint(labels)\nlabels = dict((v,k) for k,v in labels.items())\nprint(labels)\npredictions_label = [labels[k] for k in predicted_class_indices]\n\n# Match the filenames with the inference result\nfilenames = test_generator.filenames\nprint(\"[INFO] filenames size: {}\".format(len(filenames)))\nresults = pd.DataFrame({\"image_slice_id_path\":filenames, \"label\":predictions_label, 'ce': ce, 'laa': laa})\nprint(\"[INFO] store the result to the dataframe\")\nprint(results.head(14))\n\nprint(\"[INFO] merge the result with the sliced images dataset\")\nprediction_sliced_test_data = sliced_test_data.merge(results, on='image_slice_id_path')\nprint(prediction_sliced_test_data[[\"image_slice_id\", \"patient_id\", \"label\", \"ce\", \"laa\"]].head(14))\n\n## use mean for the meantime to compute the final prediction of patient's sample lab result\nprint(\"[INFO] compute the mean of images belong to the same patient_id\")\nprediction_result = prediction_sliced_test_data[['patient_id','ce', 'laa']].groupby(['patient_id']).mean()\nprint(prediction_result)\n\nprint(\"[INFO] read sample submission data\")\nsubmission = pd.read_csv(\"../input/mayo-clinic-strip-ai/sample_submission.csv\")\nprint(submission)\n\nsubmission.patient_id = prediction_result.index.to_list()\nsubmission.CE = prediction_result.ce.to_list()\nsubmission.LAA = prediction_result.laa.to_list()\n\nprint(\"[INFO] save submission data\")\nprint(submission)\nsubmission.to_csv(\"sample_submission.csv\", index = False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}