{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Published on October 09, 2023. By Marília Prata, mpwolke","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\n# import skimage.io as io\nfrom openslide import OpenSlide\nimport cv2\nimport matplotlib.pyplot as plt\nfrom tqdm.notebook import tqdm\nfrom PIL import Image\nImage.MAX_IMAGE_PIXELS = 10000000000\nprint('libraries imported')","metadata":{"execution":{"iopub.status.busy":"2023-10-10T00:02:00.755863Z","iopub.execute_input":"2023-10-10T00:02:00.756208Z","iopub.status.idle":"2023-10-10T00:02:01.497512Z","shell.execute_reply.started":"2023-10-10T00:02:00.756177Z","shell.execute_reply":"2023-10-10T00:02:01.496473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#By Saurabh Sawhney https://www.kaggle.com/code/saurabhsawhney/handling-large-images-tile-resize-and-stitch\n\n# Reading in the train and test data\ntrain_csv = pd.read_csv('../input/UBC-OCEAN/train.csv')\ncols = train_csv.columns\n# Getting image paths into the df and Apply Lambda adapted by Amar Deep adeep028\ntrain_csv['image_path'] = train_csv.image_id.apply(lambda x: os.path.join(\"../input/UBC-OCEAN/train_images/\" + str(x)+\".png\"))#Original was /train/\", x+\".tif\"\ndef enhance_df(df):\n    df[\"image_size\"]   = df.image_path.apply(lambda x: Image.open(x).size)\n    df[\"image_pixels\"] = df[\"image_size\"].apply(lambda x: int(x[0]*int(x[1])))    \n    df[\"image_width\"]  = df[\"image_size\"].apply(lambda x: int(x[0]))\n    df[\"image_height\"] = df[\"image_size\"].apply(lambda x: int(x[1]))\n    df[\"aspect_ratio\"] = df[\"image_width\"]/df[\"image_height\"]\n    return df\ntrain_csv=enhance_df(train_csv)\nprint('enhanced train dataframe ready')\n\n# The largest and smallest images\ndisplay(train_csv[train_csv.image_pixels==train_csv.image_pixels.max()])\ndisplay(train_csv[train_csv.image_pixels==train_csv.image_pixels.min()])","metadata":{"execution":{"iopub.status.busy":"2023-10-10T00:02:11.252142Z","iopub.execute_input":"2023-10-10T00:02:11.252781Z","iopub.status.idle":"2023-10-10T00:02:18.923987Z","shell.execute_reply.started":"2023-10-10T00:02:11.252714Z","shell.execute_reply":"2023-10-10T00:02:18.923269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Wow number 429 image_pixels (4083771720)","metadata":{}},{"cell_type":"markdown","source":"#Lilo and Stitch. Ooops! Tile and Stitch\n\n![](https://y.yarn.co/5cffb49f-2448-49d1-b52a-ddc0eb3f8cc0_text.gif)GetYarn","metadata":{}},{"cell_type":"markdown","source":"#The tile_resize_stitch function\n\nI added a zero to cutoff_size though it didn't work.","metadata":{}},{"cell_type":"code","source":"%%time\ndef tile_resize_stich(row, horizontal_size=4000, cutoff_size=3500000000, tiles_per_side=5, show=False, debug=False):\n    '''\n    Break up the large image, resize individual tiles, put them back together\n    Keep horizontal size at the same size specified for general resizing.\n    '''\n    def demarc(): print('='*100)\n    \n    # get the metadata -----------------------------------------------------------        \n    image_path=row.image_path\n    image_width=row.image_width\n    image_height=row.image_height\n    \n    # Show the original image, if req'd ------------------------------------------\n    if show and row.image_pixels<cutoff_size:\n        orig_image = np.array(cv2.imread(row.image_path))\n        orig_image = cv2.cvtColor(orig_image, cv2.COLOR_RGB2BGR)        \n        print('Original Image')\n        plt.imshow(orig_image); plt.show()\n        demarc()\n        \n    # What should be the tile size? ----------------------------------------------    \n    # Before  resizing -----------\n    tile_size=(int(image_width/tiles_per_side),int(image_height/tiles_per_side))\n    # After resizing -------------\n    h_size=int(horizontal_size/tiles_per_side) # target horizontal size of each tile after resizing\n    v_size=int(h_size/row.aspect_ratio)        # target vertical size of each tile after resizing, to maintain aspect ratio\n\n    # Let's make tiles from this -------------------------------------------------\n    slide=OpenSlide(image_path) # using OpenSlide to get access to the image    \n    if debug:\n        print(f'Original image_width {image_width}   image_height {image_height}')\n        print(f'individual tile_size before resizing: {tile_size}')\n    tiles=[]\n    big_tile_number=1\n    for v in tqdm(range(0,image_height-tile_size[1]+1,tile_size[1])): # The +1 is just to manage the last step in the range\n        for h in range(0,image_width-tile_size[0]+1,tile_size[0]):\n            if debug: print('processing big tile', big_tile_number)\n            image = slide.read_region((h,v),0, tile_size)  # reading a tile_size area of the image, starting at h,v,position.\n            image = np.array(image)\n            image = cv2.resize(image, dsize=(h_size,v_size), interpolation=cv2.INTER_NEAREST) # INTER_CUBIC takes far longer\n            tiles.append(image)\n            big_tile_number+=1\n            if debug: print('Tile shape:', image.shape)\n            \n    # Showing off the tiles in a grid structure ----------------------------------        \n    if show:\n        print('showing tiles')\n        fig, ax = plt.subplots(nrows = tiles_per_side,ncols = tiles_per_side, figsize = (6,6/row.aspect_ratio))\n        for i,t in enumerate(tiles):\n            x_grid=int(i/tiles_per_side)            \n            y_grid=i%tiles_per_side\n            ax[x_grid,y_grid].imshow(t)\n            ax[x_grid,y_grid].axis('off')\n        fig.tight_layout()\n        plt.show()\n        \n    # Stitching it all up --------------------------------------------------------\n    stitched = np.array(Image.new('RGBA', (h_size*tiles_per_side, v_size*tiles_per_side)))\n    if debug:\n        print('Beginning the stitching process...')\n        print('First, a placeholder image of shape', stitched.shape)\n    for pos, individual_tile in enumerate(tiles):\n            x_grid=pos%tiles_per_side\n            y_grid=int(pos/tiles_per_side)\n            if debug: print(f'GRID POSITIONS: {x_grid}, {y_grid}')\n            stitched[\n                     y_grid*v_size:y_grid*v_size+v_size,\n                     x_grid*h_size:x_grid*h_size+h_size,                \n                    :] = individual_tile\n    # Showing off the stitching process -----------------------------------------                \n            if show: \n                plt.imshow(stitched)\n                plt.show()\n                demarc()\n    # All done ------------------------------------------------------------------              \n    return stitched\n\n################################################\n# The largest image is 429 - tackling the beast now. See output 3 to get that number 429\nrow=train_csv.loc[329]#It did't opened anything with the Beast 429 \nprint('='*100)\nstitched = tile_resize_stich(row, tiles_per_side=5, show=True, debug=True)\nprint('='*100)\nplt.imshow(stitched)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-10T00:05:23.795524Z","iopub.execute_input":"2023-10-10T00:05:23.79597Z","iopub.status.idle":"2023-10-10T00:07:06.770537Z","shell.execute_reply.started":"2023-10-10T00:05:23.795932Z","shell.execute_reply":"2023-10-10T00:07:06.769512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Trying to open a \"Little\" one image_pixels.min","metadata":{}},{"cell_type":"code","source":"#By Saurabh Sawhney https://www.kaggle.com/code/saurabhsawhney/handling-large-images-tile-resize-and-stitch\n\nrow=train_csv.loc[423]#That was the last number on output 3 (image_pixels.min)\nprint('='*100)\nstitched = tile_resize_stich(row, tiles_per_side=5, show=True, debug=True)\nprint('='*100)\nplt.imshow(stitched)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-10T00:09:52.335815Z","iopub.execute_input":"2023-10-10T00:09:52.336206Z","iopub.status.idle":"2023-10-10T00:09:54.30822Z","shell.execute_reply.started":"2023-10-10T00:09:52.33617Z","shell.execute_reply":"2023-10-10T00:09:54.306833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Another \"little\" image_pixels.min","metadata":{}},{"cell_type":"code","source":"#By Saurabh Sawhney https://www.kaggle.com/code/saurabhsawhney/handling-large-images-tile-resize-and-stitch\n\nrow=train_csv.loc[37]#That was the last number on output 3 (image_pixels.min)\nprint('='*100)\nstitched = tile_resize_stich(row, tiles_per_side=5, show=True, debug=True)\nprint('='*100)\nplt.imshow(stitched)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-10T00:15:14.20487Z","iopub.execute_input":"2023-10-10T00:15:14.206012Z","iopub.status.idle":"2023-10-10T00:15:15.911529Z","shell.execute_reply.started":"2023-10-10T00:15:14.205972Z","shell.execute_reply":"2023-10-10T00:15:15.909997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Any tip to fix that is welcome.\n\nEven with Image.MAX_IMAGE_PIXELS = 10000000000 I got that \"Unsupported or missing image file\" Error.","metadata":{}},{"cell_type":"markdown","source":"![](https://encrypted-tbn0.gstatic.com/images?q=tbn:ANd9GcTAryEPAVzB1j3X9MrrYtnTMNOybWDf-5lC_Q&usqp=CAU)GetYarn","metadata":{}},{"cell_type":"markdown","source":"#Acknowledgements\n\nSaurabh Sawhney https://www.kaggle.com/code/saurabhsawhney/handling-large-images-tile-resize-and-stitch\n\nAmar Deep adeep028 https://www.kaggle.com/competitions/UBC-OCEAN/discussion/445771","metadata":{}}]}