{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# STEP 1: Creating Tiles","metadata":{}},{"cell_type":"markdown","source":"For a more detailed exposition of some of the thoughts behind this notebook, please refer\n\nhttps://www.kaggle.com/code/saurabhsawhney/handling-large-images-tile-resize-and-stitch\n\nhttps://www.kaggle.com/code/saurabhsawhney/good-quality-tiles-that-respect-the-cpu-limits","metadata":{}},{"cell_type":"markdown","source":"# Adjustables, served straight up","metadata":{}},{"cell_type":"code","source":"# Adjustables -------------------------------------------------------------------------\nout_tile_size=300\nn_horizontal_tiles_per_image=5\nhorizontal_size=out_tile_size*n_horizontal_tiles_per_image \ntiles_per_image=17   # an upper limit to the number of tiles that should be generated. They'll still have to be of acceptable quality, so this number is flexible\ndrop_proportion=0.65 # use this to drop some tiles from the majority class\ncoar=1.0\ntiles_per_side=50 # 50 for the tile-resize-stitch bit\ncutoff_size=3500000000  #  # Cutoff size to decide if its a big image\nprint('Adjustables set ------------------ ')\nprint(f'horizontal_size: {horizontal_size}')\nprint(f'tiles_per_image: {tiles_per_image}')\nprint(f'drop_proportion: {drop_proportion}')\nprint(f'out_tile_size  : {out_tile_size}')\nprint(f'Cut-off aspect ratio to decide flips: {coar}')\nprint(f'tiles_per_side (for the tile-resize-stitch routine only): {tiles_per_side}')\nprint()\nprint('Suggested folder name for downloading data produced:')\nfolder_name=f'tiles_{n_horizontal_tiles_per_image}_{tiles_per_image}_{out_tile_size}'\nprint(folder_name)\n\n# Reverting threshold back to original from 0 for version 41.\n# Preprocessing functions line 292","metadata":{"execution":{"iopub.status.busy":"2022-10-03T11:07:48.463006Z","iopub.execute_input":"2022-10-03T11:07:48.4642Z","iopub.status.idle":"2022-10-03T11:07:48.498502Z","shell.execute_reply.started":"2022-10-03T11:07:48.464067Z","shell.execute_reply":"2022-10-03T11:07:48.497127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Libraries","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nfrom pathlib import Path\nimport numpy as np\nimport skimage.io as io\nfrom skimage import transform as skimagetransform\nfrom openslide import OpenSlide\nimport cv2\nimport matplotlib.pyplot as plt\nfrom tqdm.notebook import tqdm\nfrom PIL import Image\nImage.MAX_IMAGE_PIXELS = None\nimport pickle\nimport gc\nimport time\nprint('libraries imported')","metadata":{"execution":{"iopub.status.busy":"2022-10-03T03:25:20.310293Z","iopub.execute_input":"2022-10-03T03:25:20.311066Z","iopub.status.idle":"2022-10-03T03:25:22.775521Z","shell.execute_reply.started":"2022-10-03T03:25:20.311015Z","shell.execute_reply":"2022-10-03T03:25:22.774137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Read in the data","metadata":{}},{"cell_type":"code","source":"# Reading in the train data\ntrain_csv = pd.read_csv('../input/mayo-clinic-strip-ai/train.csv')\nprint(f'train_data read in, BIG image cutoff size set at {cutoff_size} pixels')","metadata":{"execution":{"iopub.status.busy":"2022-10-03T03:25:22.777326Z","iopub.execute_input":"2022-10-03T03:25:22.777809Z","iopub.status.idle":"2022-10-03T03:25:22.801953Z","shell.execute_reply.started":"2022-10-03T03:25:22.777769Z","shell.execute_reply":"2022-10-03T03:25:22.80069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Utility Functions - enhance_df, check_class_balance, get_labels","metadata":{}},{"cell_type":"code","source":"def enhance_df(df): # for train data\n    df['image_path']     = df.image_id.apply(lambda x: os.path.join(\"../input/mayo-clinic-strip-ai/train/\", x+\".tif\"))    \n    df[\"image_size\"]     = df.image_path.apply(lambda x: Image.open(x).size)\n    df[\"image_pixels\"]   = df[\"image_size\"].apply(lambda x: int(x[0]*int(x[1])))    \n    df[\"image_width\"]    = df[\"image_size\"].apply(lambda x: int(x[0]))\n    df[\"image_height\"]   = df[\"image_size\"].apply(lambda x: int(x[1]))\n    df[\"aspect_ratio\"]   = df[\"image_width\"]/df[\"image_height\"]\n    df[\"too_big\"]        = df[\"image_pixels\"].apply(lambda x: 1 if x>cutoff_size else 0)\n    df[\"majority_class\"] = df[\"label\"].apply(lambda x: 1 if x==\"CE\" else 0)  # Booleany to indicate if the row belongs to the majority class \n    return df\n\ndef check_class_balance(labels):\n    print(f'Minority Class (LAA) percentage representation: {100*sum(labels)/len(labels):.2f}%')\n    return sum(labels)/len(labels) # CE is 0  # LAA is 1\n\ndef get_label(s):\n    if s=='CE': return 0\n    if s=='LAA': return 1 # strictly alphabetically\n    return -1\n\ndef wipe_slate():\n    sure = input('are you sure?')\n    if sure.lower()=='y':\n        ! rm './fast/train_labels.npy'\n        ! rm './fast/last_idx.pkl'\n        ! rm './fast/train_tiles.npy'\n        ! rmdir './fast'\n        print('train_labels.npy, last_idx.pkl, train_tiles.npy WIPED')       \n    else: return\n\nprint('\"enhance_df\", \"get_labels\" and \"wipe_slate\" util functions ready')    ","metadata":{"execution":{"iopub.status.busy":"2022-10-03T03:25:22.804169Z","iopub.execute_input":"2022-10-03T03:25:22.80535Z","iopub.status.idle":"2022-10-03T03:25:22.836645Z","shell.execute_reply.started":"2022-10-03T03:25:22.805239Z","shell.execute_reply":"2022-10-03T03:25:22.835206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Enhancing the data with image metadata","metadata":{}},{"cell_type":"code","source":"# Getting image paths and other metadata into the df\ntrain_csv=enhance_df(train_csv)\n\n# Outcomes\ndisplay(train_csv.head())\nprint('enhanced train dataframe ready')\n# Checking class balance at the outset\ncheck_class_balance(train_csv.label.apply(get_label))","metadata":{"execution":{"iopub.status.busy":"2022-10-03T03:25:22.838121Z","iopub.execute_input":"2022-10-03T03:25:22.838579Z","iopub.status.idle":"2022-10-03T03:25:40.932175Z","shell.execute_reply.started":"2022-10-03T03:25:22.838535Z","shell.execute_reply":"2022-10-03T03:25:40.930966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Portraits vs landscapes","metadata":{}},{"cell_type":"code","source":"portraits  = len(train_csv[train_csv[\"aspect_ratio\"]<coar])\nlandscapes = len(train_csv[train_csv[\"aspect_ratio\"]>coar])\nsquares    = len(train_csv[train_csv[\"aspect_ratio\"]==coar])\n# Cut Off aspect ratio - coar\nbig_landscapes = train_csv[(train_csv[\"aspect_ratio\"]>coar) & (train_csv[\"too_big\"]==1)]\nbig_portraits  = train_csv[(train_csv[\"aspect_ratio\"]<=coar) & (train_csv[\"too_big\"]==1)]\nprint(f'At a COAR of {coar}')\nprint(f\"There are {(portraits)} portraits, {(landscapes)} landscapes that'll need flipping, and {(squares)} 'deemed' squares.\")\nprint(f'There are {len(big_landscapes)} big landscapes, with aspect ratio more than {coar}, that we should do something about')\nprint(f'There are {len(big_portraits)} big portraits, with aspect ratio less than {coar}, that we can safely ignore')","metadata":{"execution":{"iopub.status.busy":"2022-10-03T03:25:40.933945Z","iopub.execute_input":"2022-10-03T03:25:40.934748Z","iopub.status.idle":"2022-10-03T03:25:40.953797Z","shell.execute_reply.started":"2022-10-03T03:25:40.934701Z","shell.execute_reply":"2022-10-03T03:25:40.952428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# How many images have aspect ratio less than 0.5?\n# len(train_csv[train_csv[\"aspect_ratio\"]<0.5])\n# These images can potentially produce more than a single tile when setting n_horizontal_tiles_per_image=1","metadata":{"execution":{"iopub.status.busy":"2022-10-03T03:25:40.955378Z","iopub.execute_input":"2022-10-03T03:25:40.956275Z","iopub.status.idle":"2022-10-03T03:25:40.963334Z","shell.execute_reply.started":"2022-10-03T03:25:40.956237Z","shell.execute_reply":"2022-10-03T03:25:40.961549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocessing functions","metadata":{}},{"cell_type":"code","source":"################################### WASTE PIXEL REMOVER ###################################\ndef waste_pixel_remover(img, debug=False):\n    '''\n    Go row by row, evaluating if the row is \"worthy\", keeping it if so\n    Then turn around 90 degrees, do it again.\n    Seems to work reasonably well\n    '''\n    if debug: print('Entering the waste_pixel_remover ....')\n    new_image=[]\n    img_level_sd=img.std()\n    for pixel_row in img[:]:\n        if pixel_row.std()>0.80*img_level_sd: # consider a flexible cutoff\n            new_image.append(pixel_row)\n    img=np.array(new_image).astype(np.uint8)\n    img = skimagetransform.rotate(img,angle=90,resize=True, mode='reflect')\n    img = (img*255).astype(np.uint8)\n\n    # we now flip it and repeat\n    new_image=[]\n    for pixel_row in img[:]:\n        if pixel_row.std()>0.8*img_level_sd:\n            new_image.append(pixel_row)\n    img=np.array(new_image).astype(np.uint8)\n    # then we return the crop\n    if debug: print('Exiting the waste_pixel_remover ....')    \n    return img\n\n################################### TILE - RESIZE - STITCH  ###############################\ndef tile_resize_stich(row, horizontal_size=2240, cutoff_size=3500000000, tiles_per_side=5, flip=False, debug=False):\n    '''\n    Break up the large image, resize individual tiles, put them back together\n    Keep horizontal size at the same size specified for general resizing.\n    implemented a flip switch so we can turn the image 90 degrees before sending it back\n    \n    Also, the flip argument is no longer relevant, so we need to refactor it out of circulation\n    '''\n    def demarc(): print('='*100)\n    # get the metadata --------------------------------------------------------------------\n    image_path=row.image_path\n    image_width=row.image_width\n    image_height=row.image_height\n    # Calculate final image size, based on flipping requirements --> final_h_size and final_v_size ----------------    \n    # Planned dimensions if no flipping is required\n    final_h_size=horizontal_size*3                        # multiplying by 3 to ensure downsampling on resize after waste_pixel_removal\n    final_v_size=int(horizontal_size/row.aspect_ratio)*3  # multiplying by 3 to ensure downsampling on resize after waste_pixel_removal\n    dsize=(final_h_size, final_v_size)\n    # Planned dimensions if flipping will be done -----------------------------------------\n    if flip:\n        final_v_size=int(horizontal_size*row.aspect_ratio)*3\n        dsize=(final_v_size, final_h_size)\n    if debug: print(f'Inside tile-resize-stitch function. Planned dimensions : W {final_h_size}. H {final_v_size}')\n    # What should be the tile size? -------------------------------------------------------   \n    # Before  resizing -----------\n    tile_size=(int(image_width/tiles_per_side),int(image_height/tiles_per_side))\n    # After resizing -------------\n    final_tile_h_size=int(dsize[0]/tiles_per_side)        # target horizontal size of each tile after resizing. Using dsize takes care of flip\n    final_tile_v_size=int((dsize[0]/tiles_per_side)/row.aspect_ratio)        # target vertical size of each tile after resizing\n    # Let's make tiles from this ----------------------------------------------------------\n    slide=OpenSlide(image_path) # using OpenSlide to get access to the image    \n    if debug:\n        print(f'Original image_width {image_width}   image_height {image_height}')\n        print(f'individual tile_size before resizing: {tile_size}')\n        print(f'individual tile_size after resizing:  {final_tile_h_size,final_tile_v_size}')\n    tiles=[]\n    big_tile_number=1\n    for v in tqdm(range(0,image_height-tile_size[1]+1,tile_size[1])): # The +1 is just to manage the last step in the range\n        for h in range(0,image_width-tile_size[0]+1,tile_size[0]):\n            if debug: print('processing big tile', big_tile_number)\n            image = slide.read_region((h,v),0, tile_size)  # reading a tile_size area of the image, starting at h,v,position.\n            image = np.array(image)\n            if debug: print('Tile shape as read in with openslide:',image.shape)\n            image = np.array(image)[:,:,:3] # we need only 3 channels, let's save resizing computation\n            if debug: print('Tile shape before sending to resize:',image.shape)    \n            image = cv2.resize(image, dsize=(final_tile_h_size,final_tile_v_size), interpolation=cv2.INTER_NEAREST) # INTER_CUBIC takes far longer\n            tiles.append(image)\n            big_tile_number+=1\n            if debug: print('Tile shape:', image.shape)              \n    # Stitching it all up -----------------------------------------------------------------\n    stitched = np.array(Image.new('RGB', (final_tile_h_size*tiles_per_side, final_tile_v_size*tiles_per_side))) # since our tiles have 3 channels\n    if debug:\n        print('Beginning the stitching process...')\n        print('First, a placeholder image of shape', stitched.shape)\n    for pos, individual_tile in enumerate(tiles):\n            x_grid=pos%tiles_per_side\n            y_grid=int(pos/tiles_per_side)\n            if debug: print(f'GRID POSITIONS: {x_grid}, {y_grid}')\n            stitched[\n                     y_grid*final_tile_v_size:y_grid*final_tile_v_size+final_tile_v_size,\n                     x_grid*final_tile_h_size:x_grid*final_tile_h_size+final_tile_h_size,                \n                    :] = individual_tile\n    # Showing off the stitching process --------------------------------------------------                \n            if debug: \n                plt.imshow(stitched)\n                plt.show()\n                demarc()\n    if debug: print('Pre waste removal stats: stitched image of size', stitched.shape)\n            \n    # Let's get rid of the waste pixels now ----------------------------------------------\n    if debug:\n        print('pre-waste removal')\n        plt.imshow(stitched); plt.show();\n    stitched=waste_pixel_remover(stitched, debug=debug)\n    if debug: print('post-waste removal')\n    if debug: plt.imshow(stitched); plt.show();\n    \n    # This will change the image size, so let's get that back in perspective   \n    new_aspect_ratio=stitched.shape[1]/stitched.shape[0]\n    if debug: \n        print('new_aspect_ratio',new_aspect_ratio)\n        print('old_aspect_ratio',row.aspect_ratio)\n    # Planned dimensions if no flipping is required\n    final_h_size=horizontal_size\n    final_v_size=int(horizontal_size/new_aspect_ratio)\n    dsize=(final_h_size, final_v_size)\n    # Planned dimensions if flipping will be done -----------------------------------------\n    if new_aspect_ratio>coar: # global variable alert - work this in later\n        if debug: print(f'FLIPPING at a post waste a.r of {new_aspect_ratio}')\n        final_v_size=int(horizontal_size*new_aspect_ratio)\n        dsize=(final_v_size, final_h_size)\n    # Now, first we resize\n    stitched=cv2.resize(stitched, dsize=dsize, interpolation=cv2.INTER_NEAREST)\n    # and then we flip, as required    \n    if new_aspect_ratio>coar: # global variable alert - work this in later        \n        stitched = skimagetransform.rotate(stitched,angle=90,resize=True, mode='reflect')\n        stitched = (stitched*255).astype(np.uint8)\n    if debug:\n        print('post-waste removal and sos flip and resizing')\n        print('returning a stitched image of size', stitched.shape)\n        plt.imshow(stitched); plt.show(); \n        print('Exiting the tile-resize-stitch function now...')\n    # All done ---------------------------------------------------------------------------              \n    return stitched\n### END OF tile_resize_stich FUNCTION ####################################################\n############################ LOAD IMAGE ##################################################\ndef load_image(row, horizontal_size=horizontal_size, coar=1.25, debug=False):\n    '''\n    row = dataframe row\n    coar = cut-off aspect ratio. If the aspect ratio (width/height) is greater, the image is rotated to portrait orientation.\n    '''\n    # Let's answer the flip question first -----------------------------------------------\n    needs_flip=False # by default, since most don't require flipping\n    if row.aspect_ratio>coar:\n        needs_flip=True        \n    # Get the image path -----------------------------------------------------------------\n    image_path=row.image_path\n    if debug:\n        print('Too Big?', bool(row.too_big))\n        print(f'Original dimensions: W {row.image_width}, H {row.image_height}' )\n    # Calculate final sizes, based on flipping requirements ------------------------------    \n    # Planned dimensions if no flipping is required\n    final_h_size=horizontal_size\n    final_v_size=int(horizontal_size/row.aspect_ratio)\n    dsize=(final_h_size, final_v_size)\n    # Planned dimensions if flipping will be done\n    if needs_flip:\n        final_v_size=int(horizontal_size*row.aspect_ratio)\n        dsize=(final_v_size, final_h_size)\n    if debug: print(f'Planned dimensions : W {final_h_size}. H {final_v_size}')\n    # Handling a SMALL image -------------------------------------------------------------    \n    if not row.too_big:  # small image, we can handle it here\n        img=io.imread(row.image_path)\n        # Prelim resizing required?\n        scale_factor = 0.05\n        if row.image_pixels>scale_factor*cutoff_size:\n        # This is to take care of an OOM error for some large images that are not beyond the cutoff\n        # I'd rather not send them to tile-resize-stitch as that takes up time\n            if debug: print(f'Entering the big small-image area with initial size of {row.image_pixels}')\n            width = int(img.shape[1] * scale_factor)\n            height = int(img.shape[0] * scale_factor)\n            img = cv2.resize(img, (width, height))\n            if debug:\n                print('The big small-image before we remove wastes')\n                plt.imshow(img); plt.show();\n        else:\n            if debug:\n                print('The small-image before we remove wastes')\n                plt.imshow(img); plt.show();\n\n        # get rid of wastes now ----------------------------------------------------------\n        img=waste_pixel_remover(img, debug=debug)        \n        if debug:\n            print('The image AFTER we removed wastes')\n            plt.imshow(img); plt.show();\n\n        # This will change the image size, so let's get that back in perspective---------------   \n        new_aspect_ratio=img.shape[1]/img.shape[0]\n        if debug: \n            print('new_aspect_ratio',new_aspect_ratio)\n            print('old_aspect_ratio',row.aspect_ratio)\n        # Planned dimensions if no flipping is required\n        final_h_size=horizontal_size\n        final_v_size=int(horizontal_size/new_aspect_ratio)\n        dsize=(final_h_size, final_v_size)\n        # Planned dimensions if flipping will be done -----------------------------------------\n        if new_aspect_ratio>coar: # global variable alert - work this in later\n            if debug: print(f'FLIPPING at a post waste a.r of {new_aspect_ratio}')\n            final_v_size=int(horizontal_size*new_aspect_ratio)\n            dsize=(final_v_size, final_h_size)\n        # Now, first we resize\n        img=cv2.resize(img, dsize=dsize, interpolation=cv2.INTER_NEAREST)\n        # and then we flip, as required    \n        if new_aspect_ratio>coar:\n            if debug:\n                print(f'flipping this SMALL image at an aspect ratio of {new_aspect_ratio}')\n            img = skimagetransform.rotate(img,angle=90,resize=True, mode='reflect')\n            img = (img*255).astype(np.uint8)\n        # That's about it for the small image -------------------------------------------------\n        if debug:\n            print(f'Returning an image of final dimensions Width:{img.shape[1]} Height:{img.shape[0]}')\n            print(f'Returning an image of final shape:{img.shape} and type {img.dtype}')            \n            plt.imshow(img)\n            plt.show()\n        return img            \n    else:  # handling a big image by outsourcing------------------------------------------\n        if debug:\n            print(f'Sending the big image to the tile-resize-stitch function. It will{\" not\" if not needs_flip else \"\"} need flipping.')\n        img= tile_resize_stich(row, horizontal_size=horizontal_size, cutoff_size=cutoff_size, tiles_per_side=tiles_per_side, flip=needs_flip, debug=debug)\n        return img\n### END OF load_image FUNCTION ###########################################################\n############################ MAKE TILES ##################################################\ndef make_tiles(img, tile_size=out_tile_size, debug=False):\n    if debug:\n        print('>'*80)\n        print(f'INSIDE THE make_tiles FUNCTION')\n        print(f'this is the image received. Shape: {img.shape}')\n        plt.imshow(img)\n        plt.show()\n        print('>'*80)                \n    how_many_tiles_horizontally=img.shape[1]//tile_size  #-------------------------------- \n    how_many_tiles_vertically  =img.shape[0]//tile_size  #--------------------------------     \n    if debug:\n        print('how_many_tiles_horizontally:',how_many_tiles_horizontally)\n        print('how_many_tiles_vertically  :',how_many_tiles_vertically)\n    # Making the tiles -------------------------------------------------------------------    \n    tiles=[]\n    counter=0\n    for gridrow in range(how_many_tiles_vertically):\n        for gridcol in range(how_many_tiles_horizontally):\n            row_start=gridrow*tile_size\n            row_end=gridrow*tile_size+tile_size\n            col_start=gridcol*tile_size\n            col_end=gridcol*tile_size+tile_size\n            counter=counter+1\n            tile=img[row_start:row_end,\n                     col_start:col_end,\n                     :]\n            tiles.append(tile)\n            if debug:   # showing each tile hot off the press ----------------------------\n                print(f'GRIDROW: {gridrow}, GRIDCOL:{gridcol}')            \n                print(f'row start: {row_start} --> row end: {row_end}')\n                print(f'col start: {col_start} --> col end: {col_end}')\n                print('counter',counter)\n                print(tile.shape)\n                plt.imshow(tile)\n                plt.show()\n    return np.array(tiles)\n### END OF make_tiles FUNCTION ###########################################################\n############################ GRAY CONVERSION #############################################\ndef grayConversion(image):\n    '''\n    We'll use greyscale images to determine a color-independent image standard deviation, for use as a tile selection criterion.\n    '''\n    grayValue = 0.07 * image[:,:,2] + 0.72 * image[:,:,1] + 0.21 * image[:,:,0]\n    gray_img = grayValue.astype(np.uint8)\n    return gray_img\n### END OF grayConversion FUNCTION #######################################################\n############################ pick n tiles ################################################\ndef pick_n_tiles(tiles, n=tiles_per_image, show=False, debug=False):\n    ''' Pick CoV based best n tiles from the tiles generated'''\n    if debug:\n        print(f\"Received {len(tiles)} tiles for selection\")\n        print('Nature of tiles received for picking')\n        for tile in tiles:\n            print(tile.dtype)\n            break        \n    # Calculating CoV for all tiles sent our way -----------------------------------------        \n    coeffs_of_variation=[]    \n    for idx in range(tiles.shape[0]):\n        grey_img=grayConversion(tiles[idx])\n        sd = np.std(grey_img)\n        if debug: print(f'For this tile, calculated sd {sd}')\n        if sd==0: cov=0 # sd=0 will generate a NaN value for cov, so we best not go there\n        else:     cov = sd/np.mean(grey_img)\n        coeffs_of_variation.append(cov)\n    # What if all tiles are plain white (or some other colour) sheets? -------------------\n    if max(coeffs_of_variation)==0: # If all covs are zero, then we have nothing to show\n        return [tiles[0]],[1e-8]  # =========================================== RETURN ===\n    # Finding out image level covariance in the presence of multiple zero cov values------\n    non_zero_cov_values=[x for x in coeffs_of_variation if x>0]\n    image_level_cov= np.mean(non_zero_cov_values)\n    reasonable_cov = 0.5\n    threshold_score = 0.9 + reasonable_cov/(image_level_cov+reasonable_cov) # as the image level sd falls, the background is more, so we need stricter criterion\n    kept_tiles = []\n    scores=[]\n    # Time to choose wisely --------------------------------------------------------------\n    for tile, cov in zip(tiles, coeffs_of_variation):\n        score=cov/(image_level_cov+1e-8)\n        if debug: print(f'Tile selection process score: {score}, threshold score {threshold_score}, image_level cov {image_level_cov}')\n        tile_status=1 if score >= threshold_score else 0\n        if tile_status:\n            kept_tiles.append(tile)\n            scores.append(score)            \n    # What if all tiles are bad? --------------------------------------------------------- \n    if len(kept_tiles)==0:\n        return [tiles[0]],[1e-8]  # =========================================== RETURN ===\n    # Sorting -----( psst...I want Gryffindor)--------------------------------------------        \n    score_dict={k:v for v,k in enumerate(scores) }\n    sorted_scores=list(score_dict.keys())\n    sorted_scores.sort(reverse=True)\n    sorted_tiles=[]\n    for k in sorted_scores[:n]:\n        sorted_tiles.append(kept_tiles[score_dict[k]])\n        if debug: print(score_dict[k], k)\n    if show: # show and tell ------------------------------------------------------------\n        for tile in sorted_tiles:\n            plt.imshow(tile)\n            plt.show()\n    return sorted_tiles,sorted_scores[:n]\n### END OF pick_n_tiles FUNCTION #########################################################\n############################ TILE AND PROCESS ############################################\ndef tile_and_process(row, coar=1.25, limit=10, debug=False):\n    if debug: print('Loading image...')\n    image=load_image(row, coar=coar, debug=debug)\n    if debug: print('loaded image size:', image.shape)\n        \n    if debug: print('Making tiles...', end=' ')    \n    tiles=make_tiles(image, tile_size=out_tile_size, debug=debug)\n    if debug: print(tiles.shape, type(tiles))\n\n    if debug: print(f'Selecting top {limit} tiles...')        \n    tiles,scores = pick_n_tiles(tiles, n=limit, debug=debug)\n    if debug: print('scores:', scores)\n\n    return tiles,scores\n### END OF tile_and_process FUNCTION #####################################################\n##########################################################################################\nprint('Preprocessing functions ready')","metadata":{"execution":{"iopub.status.busy":"2022-10-03T03:25:40.967498Z","iopub.execute_input":"2022-10-03T03:25:40.967928Z","iopub.status.idle":"2022-10-03T03:25:41.031035Z","shell.execute_reply.started":"2022-10-03T03:25:40.967894Z","shell.execute_reply":"2022-10-03T03:25:41.030165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Test on a large image\n# row=train_csv.loc[12]\n# img=load_image(row, coar=1)\n# plt.imshow(img)\n# plt.show()\n# print(img.shape)\n# del img","metadata":{"execution":{"iopub.status.busy":"2022-10-03T03:25:41.03284Z","iopub.execute_input":"2022-10-03T03:25:41.03322Z","iopub.status.idle":"2022-10-03T03:25:41.051899Z","shell.execute_reply.started":"2022-10-03T03:25:41.033188Z","shell.execute_reply":"2022-10-03T03:25:41.051088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Test on a small image\n# row=train_csv.loc[522]\n# img=load_image(row, debug=True)\n# plt.imshow(img)\n# plt.show()\n# print(img.shape)\n# del img","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocessing Run","metadata":{}},{"cell_type":"code","source":"%%time\ntiles_per_side=50 # for the tile-resize-stitch bit\n\ntrain_tiles=[]\ntrain_labels=[]\nlast_idx=0                   # 3 # 522 #329  # 319 # 13\nshow_as_you_process=False    #True #False\ndebug=False                  #True #False\nsanity_check_step=100\nfor idx in tqdm(train_csv.index[last_idx:]):   # last_idx+1\n#     if idx not in [109]: continue              # 514 requires a flip\n    if idx in range(0,800,50):\n        np.save(f\"./train_tiles.npy\", np.array(train_tiles).astype(np.uint8))\n        np.save(f\"./train_labels.npy\", np.array(train_labels))\n        print('<=>'*40); print(f'Checkpointing at index position {idx}'); print('<=>'*40)\n    lbl = get_label(train_csv.loc[idx,'label'])\n    print(f'Working with index position {idx}. The image belongs to class [{lbl}]')\n    tiles,scores=tile_and_process(train_csv.loc[idx], coar=coar, limit=tiles_per_image, debug=debug)\n    if debug:\n        print('nature of tiles', type(tiles))\n        print('nature of individual tile', tiles[0].dtype)        \n    # Balance it out - If the label is majority class [0] and there are more than one images, drop a few of the bad ones\n    if lbl==0 and len(tiles)>1:\n        n_tiles_to_drop=int(drop_proportion*len(tiles)+0.5)\n        n_tiles_to_keep=len(tiles)-int(drop_proportion*len(tiles))\n        print(f'Keeping {n_tiles_to_keep}, dropping {n_tiles_to_drop} of {len(tiles)} tiles for the sake of balance')\n        tiles=tiles[:n_tiles_to_keep]\n    print(f'This image produced {len(tiles)} tiles of the class {lbl}.')        \n    train_tiles.extend(tiles)\n    if debug:\n        print(f'Len of train tiles now is {len(train_tiles)}')\n        print(f'Shape of train tiles as an nparray now is {np.array(train_tiles).shape}') \n        print(f'>>>>>>>>>>>>>>>>>>>>>>>>-----------------  MEMORY SIZE of tiles in this lot: {np.array(tiles).nbytes/(1024*1024)} MB and per tile size: {np.array(tiles).nbytes/(len(tiles)*1024*1024)} ')\n    this_image_labelset=[get_label(train_csv.loc[idx,'label'])]*len(tiles)\n    train_labels.extend(this_image_labelset)\n    if debug: \n        print(f'Len of train labels now is {len(train_labels)}')  \n        print(f'Shape of train labels as an nparray now is {np.array(train_labels).shape}')\n        print('The labels are', train_labels)\n        print('The np array labels are', np.array(train_labels))    \n    if show_as_you_process:\n        for ii, tt, ss in zip(range(len(tiles)),tiles,scores):\n            print('='*100)\n            print(f'TILE NUMBER {ii+1}, score: {ss}')\n            plt.imshow(tt)\n            plt.show()\n    if idx%sanity_check_step==0:\n        print('ENROUTE QUALITY CHECKS. AT INDEX POSITION',idx)\n        print(f'Memory footprint so far: {np.array(train_tiles).nbytes/(1024*1024):.0f} MB at an average of {np.array(train_tiles).nbytes/(len(train_tiles)*1024*1024):.4f} MB per tile.')\n        plt.imshow(np.array(train_tiles[-1]))\n        plt.show()\n        \n    print(f'IDX {idx} processed --------------------------------------------------------')\n    if idx%100==0:\n        print('garbage collection time')\n        time.sleep(2)\n        gc.collect()\nprint(f'Done. There are now {len(train_tiles)} tiles and {len(train_labels)} labels. SAVING WORK ... ', end='')\n# Saving the work --------------------------------------------------------------------------------------------\nnp.save(f\"./train_tiles.npy\", np.array(train_tiles).astype(np.uint8))\nnp.save(f\"./train_labels.npy\", np.array(train_labels))\nprint(f\"... and SAVED\")\nprint()\nprint('Shape of train_tiles as a numpy array:', np.array(train_tiles).shape)\nprint(f'ALL TILES MEMORY SIZE: {np.array(train_tiles).nbytes/(1024*1024):.0f} MB at an average of {np.array(train_tiles).nbytes/(len(train_tiles)*1024*1024):.4f} MB per tile.')\nprint(f'ALL LABELS MEMORY SIZE: {np.array(train_labels).nbytes/(1024*1024):.4f} MB')\nprint(f'Class balance: {check_class_balance(train_labels)}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Suggested folder name for downloading data produced:', folder_name)","metadata":{"execution":{"iopub.status.busy":"2022-10-03T03:29:00.649601Z","iopub.execute_input":"2022-10-03T03:29:00.65011Z","iopub.status.idle":"2022-10-03T03:29:00.655915Z","shell.execute_reply.started":"2022-10-03T03:29:00.650061Z","shell.execute_reply":"2022-10-03T03:29:00.654953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i, t in enumerate(train_tiles): # just taking a look at the output\n    if i%1000==0:\n        print(f'At INDEX POSITION {i}, nature of image: {t.dtype}')\n        plt.imshow(t)\n        plt.show()\n        print(t.max())","metadata":{"execution":{"iopub.status.busy":"2022-10-03T03:29:00.657039Z","iopub.execute_input":"2022-10-03T03:29:00.658007Z","iopub.status.idle":"2022-10-03T03:29:00.892456Z","shell.execute_reply.started":"2022-10-03T03:29:00.657969Z","shell.execute_reply":"2022-10-03T03:29:00.891246Z"},"trusted":true},"execution_count":null,"outputs":[]}]}