{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Mayo Clinic Create Parsed Image Dataset\n\nMuch of this code was adapted from: https://www.kaggle.com/code/simsonimus/hubmap-training\n\n- Split into parts based on file size","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom PIL import Image\nimport matplotlib.pyplot as plt\nfrom skimage import io\nimport glob, os, json, cv2, gc, shutil\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nfrom tensorflow.keras import backend as K\nfrom tqdm.notebook import tqdm\n\nIMG_SIZE = 512  # 1024\nINPUT_PATH = \"../input/mayo-clinic-strip-ai\"\nPRINT_PLOTS = False\nFULL_RUN = True\n\ndef create_folder(folder):\n    if not os.path.exists(folder):\n        os.makedirs(folder)\n        \n# create_folder(\"./plots\")\ncreate_folder(\"./train\")\ncreate_folder(\"./test\")\n\ndf_train = pd.read_csv(os.path.join(INPUT_PATH, 'train.csv'))\ndisplay(df_train)\n\ndf_test = pd.read_csv(os.path.join(INPUT_PATH, 'test.csv'))\ndisplay(df_test)\n\nGROUP_TO_RUN = 7\nTOTAL_GROUPS = 7\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-09-12T15:01:28.31233Z","iopub.execute_input":"2022-09-12T15:01:28.312955Z","iopub.status.idle":"2022-09-12T15:01:41.600417Z","shell.execute_reply.started":"2022-09-12T15:01:28.312825Z","shell.execute_reply":"2022-09-12T15:01:41.598603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Get Filesizes\n- Split into equally sized groups based on these file sizes","metadata":{}},{"cell_type":"code","source":"for i, d in tqdm(df_train.iterrows(), total=len(df_train)):\n\n    image_id = d['image_id']\n    file_size = os.path.getsize(f'../input/mayo-clinic-strip-ai/train/{image_id}.tif')\n    df_train.loc[i, 'file_size'] = file_size\ntarget_group_size = int (df_train['file_size'].sum() / (TOTAL_GROUPS * 0.9))","metadata":{"execution":{"iopub.status.busy":"2022-09-12T15:01:41.60384Z","iopub.execute_input":"2022-09-12T15:01:41.604446Z","iopub.status.idle":"2022-09-12T15:01:42.807785Z","shell.execute_reply.started":"2022-09-12T15:01:41.604382Z","shell.execute_reply":"2022-09-12T15:01:42.80632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['file_size'].plot(kind='hist', bins=20)","metadata":{"execution":{"iopub.status.busy":"2022-09-12T15:01:42.810189Z","iopub.execute_input":"2022-09-12T15:01:42.810524Z","iopub.status.idle":"2022-09-12T15:01:43.067917Z","shell.execute_reply.started":"2022-09-12T15:01:42.810494Z","shell.execute_reply":"2022-09-12T15:01:43.066944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create 10 equal sized groups by file size\ngroup = 1\ngroup_size = 0\nfor i, d in tqdm(df_train.iterrows(), total=len(df_train)):\n    image_id = d['image_id']\n    file_size = os.path.getsize(f'../input/mayo-clinic-strip-ai/train/{image_id}.tif')\n    if group_size + file_size > target_group_size:\n        group += 1\n        group_size = file_size\n    else:\n        group_size += file_size\n    df_train.loc[i, 'file_size'] = file_size\n    df_train.loc[i, 'group'] = group\n    \ndf_train['group'] = df_train['group'].astype('int')\ndf_train.to_csv('train_with_groups.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-09-12T15:01:43.071434Z","iopub.execute_input":"2022-09-12T15:01:43.072372Z","iopub.status.idle":"2022-09-12T15:01:43.560844Z","shell.execute_reply.started":"2022-09-12T15:01:43.07231Z","shell.execute_reply":"2022-09-12T15:01:43.55982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.groupby('group')['file_size'].sum().plot(kind='barh', figsize=(10, 5))","metadata":{"execution":{"iopub.status.busy":"2022-09-12T15:01:43.562295Z","iopub.execute_input":"2022-09-12T15:01:43.563294Z","iopub.status.idle":"2022-09-12T15:01:43.75441Z","shell.execute_reply.started":"2022-09-12T15:01:43.563257Z","shell.execute_reply":"2022-09-12T15:01:43.75294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_train.query('group == @GROUP_TO_RUN')","metadata":{"execution":{"iopub.status.busy":"2022-09-12T15:01:43.756045Z","iopub.execute_input":"2022-09-12T15:01:43.756452Z","iopub.status.idle":"2022-09-12T15:01:43.769069Z","shell.execute_reply.started":"2022-09-12T15:01:43.756419Z","shell.execute_reply":"2022-09-12T15:01:43.767352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Functions","metadata":{}},{"cell_type":"code","source":"from skimage.color import rgb2hsv\nfrom skimage.exposure import is_low_contrast\nfrom skimage import img_as_ubyte\nimport openslide\n\ndef slice_images(image_id, image_path, mask=[], folder=\"\"):\n    slide = openslide.open_slide(str(image_path))\n    size = (IMG_SIZE, IMG_SIZE)\n    level = 0\n    print('Slicing Image ' + image_id + ' ...')\n    image_jd_path = os.path.join(folder, image_id)\n    print(\"image_jd_path: \", image_jd_path)\n    create_folder(image_jd_path)\n\n    \n#     img = read_tiff(image_path)\n#     print(img.shape)\n    w, h = slide.level_dimensions[0]  # 跟shape是反过来的\n#     print(w, h)\n    # possible_slices_x = image.shape[0] // IMG_SIZE\n    # possible_slices_y = image.shape[1] // IMG_SIZE\n    \n    possible_slices_x = h // IMG_SIZE\n    possible_slices_y = w // IMG_SIZE\n\n    for x in range(possible_slices_x):\n        for y in range(possible_slices_y):\n            \n            \n            # =========================  User Code Begin ==================================\n            # 左上角的位置叫做region\n            region = (y * IMG_SIZE, x * IMG_SIZE)\n            image = np.array(slide.read_region(region, level, size))\n            image = image[:, :, :-1]\n            \n#             image_slice = image[x * IMG_SIZE : (x+1) * IMG_SIZE, y * IMG_SIZE : (y+1) * IMG_SIZE]\n#             image_slice_big = image[int(max((x-0.2), 0) * IMG_SIZE) : int(min((x+1.2), possible_slices_x) * IMG_SIZE),\n#                                       int(max((y-0.2), 0) * IMG_SIZE) : int(min((y+1.2), possible_slices_y) * IMG_SIZE)]\n            \n            img_mean = np.mean(image)\n            if img_mean > 248 or img_mean < 10:\n                continue\n            \n            \n            \n            if image.shape[0] != IMG_SIZE or image.shape[1] != IMG_SIZE:\n                print(\"Error type!!!!!!\")\n                image = cv2.resize(image, (IMG_SIZE, IMG_SIZE))\n            \n#             print(\"before image2:\", image.shape)\n            image2 = rgb2hsv(image)\n            h, w, c = image2.shape\n            sat_img = image2[:, :, 1]\n            sat_img = img_as_ubyte(sat_img)\n            ave_sat = np.sum(sat_img) / (h * w)\n        \n            if ave_sat >= 8 or is_low_contrast(image):  # foreground-percent:域值 20倍的时候10 by汪的代码\n                \n            # =========================  User Code End ==================================\n            \n                #if np.any(image_slice) and not (image_slice > 200).all(): # only process non-black and non-gray images --> no background images\n\n                if not len(mask) == 0:\n                    mask_slice = mask[x * IMG_SIZE : (x+1) * IMG_SIZE, y * IMG_SIZE : (y+1) * IMG_SIZE] * 255\n                    if 255 in mask_slice:\n                        cv2.imwrite(f\"./{folder}/{image_id}-imgslice_{x}_{y}.jpg\", image)\n                        cv2.imwrite(f\"./{folder}/{image_id}-maskslice_{x}_{y}.png\", mask_slice.astype(int))\n                else:\n                    test_path = os.path.join(image_jd_path, f\"slice{IMG_SIZE}_{x}_{y}.png\")\n                    cv2.imwrite(test_path, image)\n                    \n#                     cv2.imwrite(f\"./{folder}/{image_id}-imgslice_{x}_{y}.png\", image)\n            else:\n#                 print(\"Error type!!!!!!这啥呀这是{}\".format(image_id))\n                print(\"Error: avg = {} ave_sat = {}\".format(img_mean, ave_sat))\n            \n#                 plt.figure(figsize=(16,16))  # 创建画布=\n#                 plt.subplot(111)  # 创建显示图片的格子\n#                 plt.imshow(image)","metadata":{"execution":{"iopub.status.busy":"2022-09-12T15:01:43.771824Z","iopub.execute_input":"2022-09-12T15:01:43.772309Z","iopub.status.idle":"2022-09-12T15:01:43.928181Z","shell.execute_reply.started":"2022-09-12T15:01:43.772272Z","shell.execute_reply":"2022-09-12T15:01:43.926947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"这啥呀这是\")","metadata":{"execution":{"iopub.status.busy":"2022-09-12T15:01:43.92973Z","iopub.execute_input":"2022-09-12T15:01:43.931056Z","iopub.status.idle":"2022-09-12T15:01:43.938454Z","shell.execute_reply.started":"2022-09-12T15:01:43.931012Z","shell.execute_reply":"2022-09-12T15:01:43.936898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_tiff(image_path):\n    image = io.imread(image_path)\n    image = np.squeeze(image) # some images have unnecessary axes with shape 1 --> remove\n    if image.shape[0] == 3: # some images have color as first axis -> swap axes\n        image = image.swapaxes(0,1)\n        image = image.swapaxes(1,2)\n    return image\n\ndef read_mask(image, encoded_mask):\n    mask = rle_decode(encoded_mask, (image.shape[1], image.shape[0])) # with inverted axes\n    mask = mask.swapaxes(0,1) # swap back axes\n    mask = np.expand_dims(mask, -1) # add one axis to have same shape as images\n    return mask\n\ndef delete_directory_contents(dir):\n    for file in os.scandir(dir):\n        os.remove(file.path)\n        \ndef plot_masked_image(image, mask, name):\n    plt.imshow(image, interpolation='none')\n    plt.imshow(mask, cmap='jet', alpha=0.3, interpolation='none')\n    \n    plt.savefig(f\"./plots/{name}.png\", dpi = 1000)\n    plt.show()\n    \n\n# ref.: https://www.kaggle.com/stainsby/fast-tested-rle\ndef rle_decode(mask_rle, shape):\n    '''\n    mask_rle: run-length as string formated (start length)\n    shape: (height,width) of array to return \n    Returns numpy array, 1 - mask, 0 - background\n\n    '''\n    s = mask_rle.split()\n    starts, lengths = [np.asarray(x, dtype=int) for x in (s[0:][::2], s[1:][::2])]\n    starts -= 1\n    ends = starts + lengths\n    img = np.zeros(shape[0]*shape[1], dtype=np.uint8)\n    for lo, hi in zip(starts, ends):\n        img[lo:hi] = 1\n    return img.reshape(shape)\n\n## ref.: https://www.kaggle.com/bguberfain/memory-aware-rle-encoding\ndef rle_encode_less_memory(img):\n    pixels = img.T.flatten()\n    \n    # This simplified method requires first and last pixel to be zero\n    pixels[0] = 0\n    pixels[-1] = 0\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 2\n    runs[1::2] -= runs[::2]\n    \n    return ' '.join(str(x) for x in runs)","metadata":{"execution":{"iopub.status.busy":"2022-09-12T15:01:43.940499Z","iopub.execute_input":"2022-09-12T15:01:43.940899Z","iopub.status.idle":"2022-09-12T15:01:43.958184Z","shell.execute_reply.started":"2022-09-12T15:01:43.940863Z","shell.execute_reply":"2022-09-12T15:01:43.95712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Slice Training Images","metadata":{}},{"cell_type":"code","source":"def slice_training_images(df_train):\n    if not FULL_RUN:\n        df_train = df_train.iloc[0:1, :]  # only use one training image for quicker debug runs\n    else:\n        df_train = df_train.iloc[0:, :]\n    for index, train_sample in tqdm(df_train.iterrows(), total=len(df_train)):\n        image_id = train_sample['image_id']\n\n        image_path = os.path.join(INPUT_PATH, f\"train/{image_id}.tif\")\n#         image = read_tiff(image_path)\n\n#         slice_images(image_id, image, [], \"train\")\n        slice_images(image_id, image_path, [], \"train\")","metadata":{"execution":{"iopub.status.busy":"2022-09-12T15:01:43.96142Z","iopub.execute_input":"2022-09-12T15:01:43.96268Z","iopub.status.idle":"2022-09-12T15:01:43.975873Z","shell.execute_reply.started":"2022-09-12T15:01:43.962634Z","shell.execute_reply":"2022-09-12T15:01:43.974852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"slice_training_images(df_train)","metadata":{"execution":{"iopub.status.busy":"2022-09-12T15:01:43.977477Z","iopub.execute_input":"2022-09-12T15:01:43.978806Z","iopub.status.idle":"2022-09-12T15:09:06.012412Z","shell.execute_reply.started":"2022-09-12T15:01:43.97877Z","shell.execute_reply":"2022-09-12T15:09:06.010679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Slice Test Images","metadata":{}},{"cell_type":"code","source":"def slice_test_images(df_test):\n    if not FULL_RUN:\n        df_test = df_test.iloc[0:1, :]  # only use one training image for quicker debug runs\n    else:\n        df_test = df_test.iloc[0:, :]\n    for index, test_sample in tqdm(df_test.iterrows(), total=len(df_test)):\n        image_id = test_sample['image_id']\n\n        image_path = os.path.join(INPUT_PATH, f\"test/{image_id}.tif\")\n#         image = read_tiff(image_path)\n\n        slice_images(image_id, image_path, [], \"test\")","metadata":{"execution":{"iopub.status.busy":"2022-09-12T15:09:06.01467Z","iopub.execute_input":"2022-09-12T15:09:06.015092Z","iopub.status.idle":"2022-09-12T15:09:06.023935Z","shell.execute_reply.started":"2022-09-12T15:09:06.015056Z","shell.execute_reply":"2022-09-12T15:09:06.022231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# slice_test_images(df_test)","metadata":{"execution":{"iopub.status.busy":"2022-09-12T15:09:06.025766Z","iopub.execute_input":"2022-09-12T15:09:06.026155Z","iopub.status.idle":"2022-09-12T15:09:06.034504Z","shell.execute_reply.started":"2022-09-12T15:09:06.026108Z","shell.execute_reply":"2022-09-12T15:09:06.033587Z"},"trusted":true},"execution_count":null,"outputs":[]}]}