{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport tensorflow as tf\nimport cv2\nimport matplotlib.pyplot as plt\n\nfrom tensorflow import keras\nfrom keras.preprocessing.image import ImageDataGenerator\n\nimport os\nimport sys\nimport gc\nfrom tqdm import tqdm\n\nfrom time import sleep\n\nfrom openslide import OpenSlide\nfrom collections import defaultdict\nfrom PIL import Image\nimport tifffile","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# initialize variables here\nis_resize_image = True\nsave_image_size = (1024, 1024)\n\n# Image size must be of the same length with the save model. \n# If different from model, size MUST be greater than with the model size \n# AND multiple of model size \ntarget_image_size = (256, 256) \n\nbatch_size = 32\nmax_slices = 10\n\nresized_image_path = {'train': \"./train/\", 'test':\"./test/\"}\n\ninput_images_path = \"../input/mayo-clinic-strip-ai/train/\"\ntest_images_path = \"../input/minivggnet-training-on-stroke-blood-clot/test/\"\n\nmodel_dir = \"../input/minivggnet-training-on-stroke-blood-clot/model/\"\nfull_model_path = os.path.join(model_dir, \"minivggnet_strip_ai.hdf5\")\nweight_path = os.path.join(model_dir, \"weight_strip_ai.hdf5\")\nif is_resize_image:\n    #input_images_path = \"./train/\"\n    test_images_path = \"./test/\"\n\nis_debugging =  False","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Functions**","metadata":{}},{"cell_type":"code","source":"class CheckStdevPreprocessor:\n    def __init__(self, min_stdev=20, verbose=False):\n        self.min_stdev = min_stdev\n        self.verbose =  verbose\n        \n    # image in numpy array\n    def preprocess(self, image):     \n        std_dev = np.std(image.ravel())  # remove less than 20 std dev\n        if std_dev > self.min_stdev:\n            if self.verbose:\n                print(\"[INFO] STD: {:.02f}\".format(std_dev))\n            return True\n        else:\n            return False","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CheckContoursPreprocessor:\n    def __init__(self, min_contours=10, kernel_size=(11,11), verbose=False):\n        self.min_contours = min_contours\n        self.kernel_size = kernel_size\n        self.verbose =  verbose\n    \n    # image in numpy array\n    def preprocess(self, image):     \n        blurred = cv2.GaussianBlur(image, self.kernel_size, 0)\n        edged = cv2.Canny(blurred, 100, 200)\n        # version 3.x\n        cnts, hierarchy= cv2.findContours(edged.copy(), cv2.RETR_EXTERNAL, \n                                          cv2.CHAIN_APPROX_SIMPLE)\n        if cnts is not None and len(cnts) > self.min_contours:\n            if self.verbose:\n                print(\"[INFO] Edges {}\".format(len(cnts)))\n            return True\n        else:\n            return False","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def slice_image(slide, size=(1024,1024), slices=None, min_std=20, image_id=\"image_id\", path=\"\", \n                preprocessors=None, patient_id=\"patient_id\", other_info=None,\n                max_slices = 1e6, debug_resizing=False):\n    \n    hits = 0\n    \n    path = str(path).strip()\n    if len(path):\n        try:\n            os.mkdir(path)\n        except:\n            pass  \n    if slices is None:\n        sl = list()\n        sl.append(np.ceil(slide.dimensions[0] / size[0]).astype(int))\n        sl.append(np.ceil(slide.dimensions[1] / size[1]).astype(int))\n        slices = tuple(sl)\n    \n    if debug_resizing:\n        print(\"[DEBUG] Number of slices: x={}, y={}\".format(slices[0], slices[1]))\n        \n    image_prop = defaultdict(list)\n    for i in range(slices[0]):\n        \n        if len(image_prop['image_slice_id']) >= max_slices:\n            break\n            \n        for j in range(slices[1]):\n            \n            if len(image_prop['image_slice_id']) >= max_slices:\n                break\n                \n            region = slide.read_region((i*size[0], j*size[1]), 0, size).convert(\"RGB\")\n            region = np.array(region)  # convert PIL RGB to np array\n            grey = cv2.cvtColor(region, cv2.COLOR_RGB2GRAY) # convert RGB to gray\n            \n            # check to see if our preprocessors are not None\n            if preprocessors is not None:\n            # loop over the preprocessors and apply each to\n            # the image\n                is_valid = False\n                for p in preprocessors:\n                    is_valid = p.preprocess(grey)\n                    if is_valid is False:\n                        break\n                        \n                if is_valid:\n                    \n                    hits += 1\n                    \n                    image_slice_id = \"{}.{:03d}.{:03d}\".format(image_id,i,j)\n                    image_prop['image_id'].append(image_id)\n                    image_prop['image_slice_id'].append(image_slice_id)\n                    image_prop['patient_id'].append(patient_id)\n                    \n                    if other_info is not None:\n                        for desciption, value in other_info.items():\n                            image_prop[desciption].append(value)\n                    \n                    # save to disk  {image_id}.{x position}.{y position}.jpg\n                    image_slice_id_path = os.path.join(path, \"{}.jpg\".format(image_slice_id) )\n                    cv2.imwrite(image_slice_id_path, region)\n                    \n                    if debug_resizing:\n                        print(\"[DEBUG] With Preprocessor: saved {}-{}\".format(i, j))\n                    \n                if is_valid and debug_resizing: \n                    fig, axes = plt.subplots(1,2, figsize=(6,4))\n                    axes[0].imshow(region);\n                    sns.histplot(grey.ravel(), ax=axes[1])\n                    plt.show()\n            # preprocessors are None        \n            else:\n                image_slice_id = \"{}.{:03d}.{:03d}\".format(image_id,i,j)\n                image_prop['image_id'].append(image_id)\n                image_prop['image_slice_id'].append(image_slice_id)\n                image_prop['patient_id'].append(patient_id)\n                \n                if other_info is not None:\n                        for desciption, value in other_info.items():\n                            image_prop[desciption].append(value)\n                            \n                # save to disk  {image_id}.{x position}.{y position}.jpg\n                image_slice_id_path = os.path.join(path, \"{}.jpg\".format(image_slice_id) )\n                cv2.imwrite(image_slice_id_path, region)\n                \n                if debug_resizing:\n                    print(\"[DEBUG] No Preprocessor: saved {}-{}\".format(i, j))\n                    \n    if hits == 0 and preprocessors is not None:\n        if debug_resizing:\n            print(\"[DEBUG] No sliced image found for {}\".format(image_id))\n        image_slice_id = \"{}.non.non\".format(image_id)\n        image_prop['image_id'].append(image_id)\n        image_prop['image_slice_id'].append(image_slice_id)\n        image_prop['patient_id'].append(patient_id)\n\n        if other_info is not None:\n            for desciption, value in other_info.items():\n                image_prop[desciption].append(value)\n\n        # save to disk  {image_id}.non.non.jpg\n        region = np.zeros((size[0],size[1],3), np.uint8)\n        image_slice_id_path = os.path.join(path, \"{}.jpg\".format(image_slice_id) )\n        cv2.imwrite(image_slice_id_path, region)\n    return image_prop","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def resize_images(data, input_images_dir, output_path, size=None, \n              preprocessors=None, details=None, max_slices = 1e6, debug_resizing=False):\n\n    image_slices = defaultdict(list)\n    image_slices['image_id'] = []\n    image_slices['image_slice_id'] = []\n    image_slices['patient_id'] =[]\n    \n    image_prop = defaultdict(list)\n    \n    with tqdm(total=len(data.index)) as pbar: \n        for row in data.itertuples():\n            image_id = row.image_id\n            patient_id = row.patient_id\n            \n            other_info = {}\n            if details is not None:\n                for i in details:\n                    other_info[i] = getattr(row, i)\n            \n            image_path = os.path.join(input_images_dir, \"{}.tif\".format(image_id))\n            slide = OpenSlide(image_path)\n            \n            if size is not None:\n                image_prop = slice_image(slide, size=size, path=output_path, image_id=image_id, patient_id=patient_id,\n                                         other_info=other_info,\n                                         preprocessors=preprocessors, max_slices = max_slices, debug_resizing=debug_resizing)\n                \n                image_slices['image_id'].extend(image_prop['image_id'])\n                image_slices['image_slice_id'].extend(image_prop['image_slice_id'])\n                image_slices['patient_id'].extend(image_prop['patient_id'])\n                \n                for desciption, value in other_info.items():\n                    image_slices[desciption].extend(image_prop[desciption])\n                \n            elif size is None and slide.dimensions[0] > slide.dimensions[1]:  # width is greater than height \n                number_slice_w = np.ceil(slide.dimensions[0] / 8192.0).astype(int)\n                width = np.ceil(slide.dimensions[0]/ number_slice_w).astype(int)\n                height = np.ceil(width / (slide.dimensions[0] / slide.dimensions[1]) ).astype(int)\n                number_slice_h = np.ceil(slide.dimensions[1] / height).astype(int)\n                size = (width, height)\n                slices = (number_slice_w, number_slice_h)\n                image_prop = slice_image(slide, size=size, path=output_path, image_id=image_id, patient_id=patient_id,\n                                         other_info=other_info, slices = slices,\n                                         preprocessors=preprocessors, max_slices = max_slices, debug_resizing=debug_resizing)\n                \n                image_slices['image_id'].extend(image_prop['image_id'])\n                image_slices['image_slice_id'].extend(image_prop['image_slice_id'])\n                image_slices['patient_id'].extend(image_prop['patient_id'])\n                \n                for desciption, value in other_info.items():\n                    image_slices[desciption].extend(image_prop[desciption])\n                \n            elif size is None and slide.dimensions[1] > slide.dimensions[0] :\n                number_slice_h = np.ceil(slide.dimensions[1] / 8192.0).astype(int)\n                height = np.ceil(slide.dimensions[1] / number_slice_h).astype(int)\n                width = np.ceil(height / (slide.dimensions[1] / slide.dimensions[0])).astype(int)\n                number_slice_w = np.ceil(slide.dimensions[0]/ width).astype(int)\n                size = (width, height)\n                slices = (number_slice_w, number_slice_h)\n                image_prop = slice_image(slide, size=size, path=output_path, image_id=image_id, patient_id=patient_id,\n                                         other_info=other_info, slices = slices,\n                                         preprocessors=preprocessors, max_slices = max_slices, debug_resizing=debug_resizing)\n                \n                image_slices['image_id'].extend(image_prop['image_id'])\n                image_slices['image_slice_id'].extend(image_prop['image_slice_id'])\n                image_slices['patient_id'].extend(image_prop['patient_id'])\n                \n                for desciption, value in other_info.items():\n                    image_slices[desciption].extend(image_prop[desciption])\n           \n            pbar.update(1)\n            pbar.write('[INFO] processed: {} with slices of {}'.format(row.image_id, len(image_prop['image_slice_id'])))\n    return image_slices","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def resize_images_v1(data, size=(128,128), input_path=None, output_path=None):\n    # copy tif image to jpg in [512,512]\n    if input_path is None:\n        input_path = \"../input/mayo-clinic-strip-ai/test/\"\n    if output_path is None:\n        output_path = \"./train/\"\n    try:\n        os.mkdir(output_path)\n    except:\n        pass\n\n    with tqdm(total=len(data['image_id'])) as pbar:\n        for x in data['image_id']:       \n            img_id = x\n            img = cv2.resize(tifffile.imread(\n                os.path.join(input_path, \"{}.tif\".format(img_id))), size)\n            cv2.imwrite(os.path.join(output_path, \"{}.jpg\".format(img_id)), img)\n            del img\n            gc.collect()\n\n            pbar.write('processed: %s' %x)\n            pbar.update(1)\n            sleep(1)\n    return None","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Load test dataset**","metadata":{}},{"cell_type":"code","source":"# load the images and labels\ntrain_data = pd.read_csv(\"../input/mayo-clinic-strip-ai/train.csv\")\ntest_data = pd.read_csv(\"../input/mayo-clinic-strip-ai/test.csv\")\n\nclass_labels = [str(x) for x in np.unique(train_data['label'])]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test only the batch size and recompute the new batch size\nif False:\n    test_data = test_data.head(batch_size)\n    batch_size = batch_size // 8\n    input_images_path = \"./train/\"\n    test_images_path = \"./test/\"\n    is_resize_image = False\n\nif is_debugging:\n    print(train_data.head(4))\n    print(test_data.head(4))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Resize testing datasets**","metadata":{}},{"cell_type":"code","source":"# initialize the image processor\ncstdev = CheckStdevPreprocessor(verbose=False)\nccont = CheckContoursPreprocessor(verbose=False)\n\nif is_resize_image:\n    #resize_images_v1(test_data, size=save_image_size, \n    #              input_path=\"../input/mayo-clinic-strip-ai/test/\",\n    #              output_path=resized_image_path['test'])\n    \n    resized_image_slices = resize_images(test_data, \"../input/mayo-clinic-strip-ai/test/\", \"./test/\", \n                                         size=save_image_size, \n                                         preprocessors=[cstdev, ccont], max_slices = max_slices,\n                                         details=None, debug_resizing=False)\n\n    if is_debugging:\n        print(len(resized_image_slices['patient_id']))\n    \n    sliced_test_data = pd.DataFrame.from_dict(resized_image_slices) # load into dataframe\n    \n    if is_debugging:\n        print((sliced_test_data.head()))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Load the resized testing datasets**","metadata":{}},{"cell_type":"code","source":"def append_ext(fn):\n    return fn+\".jpg\"\n\ndef append_ext_tif(fn):\n    return fn+\".tif\"\n\ntrain_data.insert(1, \"image_id_path\", train_data[\"image_id\"].apply(append_ext_tif))\n#test_data.insert(1, \"image_id_path\", test_data[\"image_id\"].apply(append_ext))\n\nsliced_test_data.insert(2, \"image_slice_id_path\", sliced_test_data[\"image_slice_id\"].apply(append_ext))\n\nif is_debugging:\n    print((train_data.head()))\n    print((sliced_test_data.head()))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"[INFO] generators:\")\nimage_data_gen=ImageDataGenerator(rescale=1.0/255.0,validation_split=0.25)\ntest_data_gen=ImageDataGenerator(rescale=1.0/255.0)\n\ntraining_generator = image_data_gen.flow_from_dataframe(\n    dataframe=train_data, directory=input_images_path, x_col=\"image_id_path\", y_col=\"label\",\n    subset=\"training\", batch_size=batch_size, seed=42, shuffle=True, class_mode=\"categorical\",\n    target_size=target_image_size)\nvalidation_generator = image_data_gen.flow_from_dataframe(\n    dataframe=train_data, directory=input_images_path, x_col=\"image_id_path\", y_col=\"label\",\n    subset=\"validation\", batch_size=batch_size, seed=42, shuffle=True, class_mode=\"categorical\",\n    target_size=target_image_size)\n\ntest_generator = test_data_gen.flow_from_dataframe(\n    dataframe=sliced_test_data, directory=test_images_path, x_col=\"image_slice_id_path\", y_col=None,\n    batch_size=batch_size, seed=42, shuffle=False, class_mode=None,\n    target_size=target_image_size)\n\nif is_debugging:\n    print(training_generator)\n    print(validation_generator)\n    print(test_generator)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Load the model**","metadata":{}},{"cell_type":"code","source":"#tf.saved_model.load(\"my_model\")\nmodel = keras.models.load_model(full_model_path)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Evaluate the model**","metadata":{}},{"cell_type":"code","source":"predictions = model.predict(test_generator, batch_size=batch_size)\nce, laa = predictions[:, 0], predictions[:, 1]\n\n#print(classification_report(test_generator, predictions.argmax(axis=1), target_names=class_labels))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save to csv\n\nprint(\"[INFO] predictions size: {}\".format(len(predictions)))\nprint(\"[INFO] predictions shape: {}\".format(predictions.shape))\n\npredicted_class_indices = np.argmax(predictions, axis=1)\nprint(\"[INFO] predicted_class_indices shape: {}\".format(predicted_class_indices.shape))\n\nif is_debugging:\n    print(predicted_class_indices)\n\nlabels = (training_generator.class_indices)  # returns dict\n\nif is_debugging:\n    print(labels)\n\nlabels = dict((v,k) for k,v in labels.items())\n\nif is_debugging:\n    print(labels)\n\npredictions_label = [labels[k] for k in predicted_class_indices]\n\n# Match the filenames with the inference result\nfilenames = test_generator.filenames\nprint(\"[INFO] filenames size: {}\".format(len(filenames)))\nresults = pd.DataFrame({\"image_slice_id_path\":filenames, \"label\":predictions_label, 'ce': ce, 'laa': laa})\nprint(\"[INFO] store the result to the dataframe\")\nprint(results.head(20))\n\nprint(\"[INFO] merge the result with the sliced images dataset\")\nprediction_sliced_test_data = sliced_test_data.merge(results, on='image_slice_id_path')\nprint(prediction_sliced_test_data[[\"image_slice_id\", \"patient_id\", \"label\", \"ce\", \"laa\"]].head(20))\n\n## use mean for the meantime to compute the final prediction of patient's sample lab result\nprint(\"[INFO] compute the mean of images belong to the same patient_id\")\nprediction_result = prediction_sliced_test_data[['patient_id','ce', 'laa']].groupby(['patient_id']).mean()\nprint(prediction_result)\n\n\ndef maxprediction(data_frame):\n    max_ce = data_frame.loc[data_frame['ce'].idxmax()]\n    max_laa =  data_frame.loc[data_frame['laa'].idxmax()]\n    \n    if is_debugging:\n        print(\"===========\")\n        print(max_ce)\n        print(max_laa)\n    \n    if max_ce.ce >= max_laa.laa:\n        return max_ce[['ce', 'laa']]\n    else:\n        return  max_laa[['ce', 'laa']]\n\n## uget the highest predictions of images as final prediction of patient's sample lab result\nprint(\"[INFO] get the highest predictions of images belong to the same patient_id\")\nprediction_result = prediction_sliced_test_data[['patient_id','ce', 'laa']].groupby(['patient_id']).apply(maxprediction)\nprint(prediction_result)\n\nprint(\"[INFO] read sample submission data\")\nsubmission = pd.read_csv(\"../input/mayo-clinic-strip-ai/sample_submission.csv\")\nprint(submission)\n\nsubmission.patient_id = prediction_result.index.to_list()\nsubmission.CE = prediction_result.ce.to_list()\nsubmission.LAA = prediction_result.laa.to_list()\n\nprint(\"[INFO] save submission data\")\nprint(submission)\nsubmission.to_csv(\"submission.csv\", index = False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}