{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nimport tifffile\nfrom scipy.stats import mode \nimport time\nfrom tqdm import tqdm\nimport gc\nimport cv2\nfrom PIL import Image\n\ntrain_images_path = \"/kaggle/input/mayo-clinic-strip-ai/train/\"\ntest_images_path = \"/kaggle/input/mayo-clinic-strip-ai/test/\"\nalready_patched_path1 = \"/kaggle/input/mayoclinicpatchimages1/train_cropped/\"\nalready_patched_path2 = \"/kaggle/input/mayoclinic-patched-images/train_cropped\"\nalready_patched_path3 = \"/kaggle/input/mayoclinic-patched-images-3/train_cropped\"\nalready_patched_names1 = os.listdir(already_patched_path1)\nalready_patched_names2 = os.listdir(already_patched_path2)\nalready_patched_names3 = os.listdir(already_patched_path3)\n\ntrain_images_names = os.listdir(train_images_path)\ntest_images_names = os.listdir(test_images_path)\n\ntrain_output_path = \"./train_cropped/\"\nif not os.path.exists(train_output_path):\n    os.mkdir(train_output_path)\n    \nSIZE = 512\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-10T15:56:18.829877Z","iopub.execute_input":"2022-08-10T15:56:18.830274Z","iopub.status.idle":"2022-08-10T15:56:22.024258Z","shell.execute_reply.started":"2022-08-10T15:56:18.830242Z","shell.execute_reply":"2022-08-10T15:56:22.023196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_csv_path = \"/kaggle/input/mayo-clinic-strip-ai/train.csv\"\n# df_train = pd.read_csv(train_csv_path)\n# df_train.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df_ce = df_train.query(\"label=='CE'\")\n# df_laa = df_train.query(\"label=='LAA'\")\n# already_patched_names = set([name[:8] for name in already_patched_names1])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def select_n_random_patched_images(all_image_names,  patched_image_names, N=16):\n#     all_images_set = set(all_image_names)\n#     patched_images_set = set([name[:8] for name in patched_image_names])\n#     intersection = all_images_set.intersection(patched_images_set)\n#     n_images = np.random.choice(list(intersection), size = N)\n#     out = []\n#     for image_name in n_images:\n#         image_patches = [image_patch_path for image_patch_path in patched_image_names if image_name in image_patch_path]\n#         out.append(np.random.choice(image_patches, 1)[0])\n    \n#     return out\n    ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ce_names = select_n_random_patched_images(df_ce[\"image_id\"], already_patched_names1)\n# laa_names = select_n_random_patched_images(df_laa[\"image_id\"], already_patched_names1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import matplotlib.pyplot as plt\n# def plot_images(names, title):\n#     nrows, ncols = 4, 4\n#     fig, axs = plt.subplots(nrows, ncols, figsize=(20,20))\n#     fig.suptitle(title, fontsize=20)\n#     counter = 0\n#     for i in range(nrows):\n#         for j in range(ncols):\n#             image_name = names[counter]\n#             img = cv2.imread(already_patched_path1 + image_name)\n#             axs[i,j].imshow(img)\n#             counter +=1\n#     fig.tight_layout()\n#     plt.savefig(f\"./{title}.jpg\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot_images(laa_names, 'LAA')\n# plot_images(ce_names, 'CE')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_background_pixel(image):\n    x_size = image.shape[1]\n    y_size = image.shape[0]\n    pixels = []\n    for i in range(20):\n        x_random = np.random.randint(0, x_size)\n        y_random = np.random.randint(0, y_size)\n        pixel = image[y_random, x_random, :]\n        pixels.append(pixel)    \n    background_pixel = mode(pixels, axis=0)\n    return background_pixel[0][0]\n\ndef accept_image(cropped_image, background_pixel, threshold ):\n    image = cropped_image\n    n_pixels = image.shape[0]*image.shape[1]\n    n_background_pixels = (image == background_pixel).all(axis=2).sum()\n    out = False\n    try:\n        out = (n_background_pixels / n_pixels) < (1-threshold)\n    except:\n        out = False\n        print(f\"Image is not excepted, ValueError n_background_pixels = {n_background_pixels}, n_pixels = {n_pixels}\")\n    return out\n\ndef patch_image(image, size, image_name, threshold):\n    stride = size\n    counter = 0\n    background_pixel = get_background_pixel(image)\n    for y in range(0, image.shape[0]+1,stride):\n        for x in range(0, image.shape[1]+1, stride):\n            cropped_im = image[y:y+size, x:x+size,:]\n            if (accept_image(cropped_im, background_pixel, threshold)):\n                cv2.imwrite(train_output_path + f\"{image_name}_{counter}.jpg\", img=cropped_im)\n                counter +=1\n    print(f\"Image {image_name} was patched into {counter} images\")\n    return counter","metadata":{"execution":{"iopub.status.busy":"2022-08-10T15:56:27.424949Z","iopub.execute_input":"2022-08-10T15:56:27.425427Z","iopub.status.idle":"2022-08-10T15:56:27.447496Z","shell.execute_reply.started":"2022-08-10T15:56:27.42538Z","shell.execute_reply":"2022-08-10T15:56:27.445792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"already_patched_names1 = set([name[:8]+\".tif\" for name in already_patched_names1])\nalready_patched_names2 = set([name[:8]+\".tif\" for name in already_patched_names2])\nalready_patched_names3 = set([name[:8]+\".tif\" for name in already_patched_names3])\n\nalready_patched_names_1_2 = already_patched_names1.union(already_patched_names2)\nalready_patched_names = already_patched_names_1_2.union(already_patched_names3)\n\nprint(f\"In already_patched_names1 {len(already_patched_names1)} images\")\nprint(f\"In already_patched_names2 {len(already_patched_names2)} images\")\nprint(f\"In already_patched_names_1_2 {len(already_patched_names_1_2)} images\")\nprint(f\"In already_patched_names {len(already_patched_names)} images\")","metadata":{"execution":{"iopub.status.busy":"2022-08-10T15:56:31.25073Z","iopub.execute_input":"2022-08-10T15:56:31.251103Z","iopub.status.idle":"2022-08-10T15:56:31.380549Z","shell.execute_reply.started":"2022-08-10T15:56:31.251072Z","shell.execute_reply":"2022-08-10T15:56:31.379283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for train_image_name in tqdm(train_images_names):\n    if train_image_name not in already_patched_names:\n        retry = True\n        threshold = 0.95\n        while retry:\n            image = tifffile.imread(train_images_path + train_image_name)\n            counter = patch_image(image, size=SIZE, image_name = train_image_name, threshold=threshold)\n            del image\n            gc.collect()\n            if counter==0:\n                threshold = threshold - 0.05\n            else:\n                retry = False","metadata":{"execution":{"iopub.status.busy":"2022-08-09T17:11:32.175793Z","iopub.execute_input":"2022-08-09T17:11:32.176929Z","iopub.status.idle":"2022-08-09T17:14:27.641791Z","shell.execute_reply.started":"2022-08-09T17:11:32.176884Z","shell.execute_reply":"2022-08-09T17:14:27.639837Z"},"trusted":true},"execution_count":null,"outputs":[]}]}