{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# UBC Images slicing","metadata":{}},{"cell_type":"markdown","source":"# 1. Intro","metadata":{}},{"cell_type":"markdown","source":"<p style=\"font:18px Verdana; text-indent: 5%\">\nThe purpose of this notebook is to slice images from UBC Ovarian cancer classification base on background. \n</p>\n\n<p style=\"font:18px Verdana; text-indent: 5%; color: red\">\nIf you find this notebook useful, feel free to upvote ;-).\n</p>","metadata":{}},{"cell_type":"markdown","source":"# 2. Librairies import","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport pandas as pd\nimport numpy as np\nimport cv2 # for resize\n\nimport os, gc\nfrom tqdm import tqdm\nfrom PIL import Image, ImageChops","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-10-31T09:17:24.799296Z","iopub.execute_input":"2023-10-31T09:17:24.799705Z","iopub.status.idle":"2023-10-31T09:17:25.509185Z","shell.execute_reply.started":"2023-10-31T09:17:24.799673Z","shell.execute_reply":"2023-10-31T09:17:25.507881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. Data loading","metadata":{}},{"cell_type":"markdown","source":"<p style=\"font:18px Verdana; text-indent: 0%\">\nWe first define working directories :\n</p>","metadata":{}},{"cell_type":"code","source":"repBase        = '/kaggle/input/UBC-OCEAN'\nrepTrain       = 'train_images/'\nrepTrain_thumb = 'train_thumbnails/'","metadata":{"execution":{"iopub.status.busy":"2023-10-31T09:17:32.601821Z","iopub.execute_input":"2023-10-31T09:17:32.602519Z","iopub.status.idle":"2023-10-31T09:17:32.608445Z","shell.execute_reply.started":"2023-10-31T09:17:32.602471Z","shell.execute_reply":"2023-10-31T09:17:32.60757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p style=\"font:18px Verdana; text-indent: 0%\">\nThen we load training data in dataframe. For images marked as TMA, we will switch directory to thumbnails one :\n</p>","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/UBC-OCEAN/train.csv')\ntrain['path'] = train.apply(lambda row: (os.path.join(repTrain, str(row['image_id'])+'.png')) if row['is_tma'] else (os.path.join(repTrain_thumb, str(row['image_id'])+'_thumbnail.png')), axis=1)\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-10-31T09:17:35.90417Z","iopub.execute_input":"2023-10-31T09:17:35.904699Z","iopub.status.idle":"2023-10-31T09:17:35.963759Z","shell.execute_reply.started":"2023-10-31T09:17:35.90466Z","shell.execute_reply":"2023-10-31T09:17:35.962343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p style=\"font:18px Verdana; text-indent: 0%\">\nFinally, we create an index of couple image_id / path.\n</p>","metadata":{}},{"cell_type":"code","source":"index_paths  = {image:codage for image, codage in zip(train['image_id'], train['path'])}","metadata":{"execution":{"iopub.status.busy":"2023-10-31T09:17:38.65271Z","iopub.execute_input":"2023-10-31T09:17:38.653399Z","iopub.status.idle":"2023-10-31T09:17:38.665861Z","shell.execute_reply.started":"2023-10-31T09:17:38.653346Z","shell.execute_reply":"2023-10-31T09:17:38.664722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p style=\"font:18px Verdana; text-indent: 0%\">\nFor information, here is the number of observations of each class.\n</p>","metadata":{}},{"cell_type":"code","source":"train.label.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-10-31T09:17:40.910231Z","iopub.execute_input":"2023-10-31T09:17:40.910621Z","iopub.status.idle":"2023-10-31T09:17:40.925953Z","shell.execute_reply.started":"2023-10-31T09:17:40.910592Z","shell.execute_reply":"2023-10-31T09:17:40.924794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 4. Image visualization","metadata":{}},{"cell_type":"markdown","source":"<p style=\"font:18px Verdana; text-indent: 0%\">\nHere are some examples of cases that can be met in dataset :\n</p>","metadata":{}},{"cell_type":"code","source":"fig = plt.figure(figsize=(14, 20)) \n  \n# setting values to rows and column variables \nrows = 1\ncolumns = 4\n\nfor position, image_id in enumerate([3997,23523,21260,5251]):\n    img = Image.open(os.path.join(repBase, index_paths[image_id]))\n    fig.add_subplot(rows, columns, position+1)   \n    plt.imshow(img) \n    plt.axis('off') \n    plt.title(image_id) ","metadata":{"execution":{"iopub.status.busy":"2023-10-31T10:35:52.915705Z","iopub.execute_input":"2023-10-31T10:35:52.91612Z","iopub.status.idle":"2023-10-31T10:35:56.635633Z","shell.execute_reply.started":"2023-10-31T10:35:52.91609Z","shell.execute_reply":"2023-10-31T10:35:56.634533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 4. Boxes determination","metadata":{}},{"cell_type":"markdown","source":"<p style=\"font:18px Verdana; text-indent: 0%\">\nFirst of all, we create a function that will identify form within image base on pixels color threshold :\n</p>","metadata":{}},{"cell_type":"code","source":"def boxesWithinImage(image):\n\n    # define border color\n    lower = (0, 0, 0)\n    upper = (0, 0, 0)\n\n    # threshold on border color\n    mask = cv2.inRange(img, lower, upper)\n\n    # dilate threshold\n    kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (15, 15))\n    mask = cv2.morphologyEx(mask, cv2.MORPH_DILATE, kernel)\n\n    # recolor border to white\n    img[mask==255] = (255,255,255)\n\n    # convert img to grayscale\n    gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n\n    # otsu threshold\n    thresh = cv2.threshold(gray, 0, 255, cv2.THRESH_OTSU )[1] \n\n    # apply morphology open\n    kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (17,17))\n    morph = cv2.morphologyEx(thresh, cv2.MORPH_OPEN, kernel)\n    morph = 255 - morph\n\n    # find contours and bounding boxes\n    bboxes = []\n    contours = cv2.findContours(morph, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)\n    contours = contours[0] if len(contours) == 2 else contours[1]\n    for cntr in contours:\n        x,y,w,h = cv2.boundingRect(cntr)\n        bboxes.append((x,y,w,h))\n\n    return bboxes","metadata":{"execution":{"iopub.status.busy":"2023-10-31T10:50:05.754154Z","iopub.execute_input":"2023-10-31T10:50:05.754603Z","iopub.status.idle":"2023-10-31T10:50:05.766218Z","shell.execute_reply.started":"2023-10-31T10:50:05.754567Z","shell.execute_reply":"2023-10-31T10:50:05.764698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p style=\"font:18px Verdana; text-indent: 0%\">\nLet's test this function on image 26644 for example :\n</p>","metadata":{}},{"cell_type":"code","source":"image_id = 26644\n\nimagePath = index_paths[image_id] \nfullFromPath = os.path.join(repBase, imagePath)\nimg = cv2.imread(fullFromPath)\nbboxes = boxesWithinImage(img)\n\nprint(\"Size of originale image : \", img.shape)\nplt.imshow(img) \n\nprint(\"Number of identified boxes : \"+str(len(bboxes)))","metadata":{"execution":{"iopub.status.busy":"2023-10-31T10:53:17.792482Z","iopub.execute_input":"2023-10-31T10:53:17.792931Z","iopub.status.idle":"2023-10-31T10:53:18.748435Z","shell.execute_reply.started":"2023-10-31T10:53:17.792897Z","shell.execute_reply":"2023-10-31T10:53:18.747253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p style=\"font:18px Verdana; text-indent: 0%\">\nAs we can see, it seems there are 3 slices but we found 3 boxes. We will have to remove noisy boxes. For this we will base on the surface covered by each box.\n</p>","metadata":{}},{"cell_type":"code","source":"surfaceImage = img.shape[0] * img.shape[1]\ncleanedBoxes = []\nfor boxe in bboxes:\n    surfaceBoxe = boxe[2] * boxe[3]\n    partBoxe = surfaceBoxe*100/surfaceImage\n    print(\"Boxe : \", boxe, \" surface : \", surfaceBoxe, \" -> \", partBoxe, \"%\")\n    if(partBoxe>5):\n        cleanedBoxes.append(boxe)","metadata":{"execution":{"iopub.status.busy":"2023-10-31T10:56:56.308786Z","iopub.execute_input":"2023-10-31T10:56:56.309227Z","iopub.status.idle":"2023-10-31T10:56:56.318442Z","shell.execute_reply.started":"2023-10-31T10:56:56.309179Z","shell.execute_reply":"2023-10-31T10:56:56.316817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p style=\"font:18px Verdana; text-indent: 5%\">\nThe first box in the list above frames a noise in the image. We are going to set an arbitrary rule to overcome this type of case: we are going to delete all boxes covering less than 5% of the image surface.<br>\nHere is the result :\n</p>","metadata":{}},{"cell_type":"code","source":"fig = plt.figure(figsize=(14, 7)) \n  \nrows = 2\ncolumns = 3\nprint(cleanedBoxes)\nfor position, boxe in enumerate(cleanedBoxes):\n    newImage = img[boxe[1]:boxe[1]+boxe[3],boxe[0]:boxe[0]+boxe[2]]\n    fig.add_subplot(rows, columns, position+1)   \n    plt.imshow(newImage) \n    plt.axis('off') ","metadata":{"execution":{"iopub.status.busy":"2023-10-31T11:01:00.683438Z","iopub.execute_input":"2023-10-31T11:01:00.683884Z","iopub.status.idle":"2023-10-31T11:01:01.120442Z","shell.execute_reply.started":"2023-10-31T11:01:00.683849Z","shell.execute_reply":"2023-10-31T11:01:01.118902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 5. Application on whole dataset","metadata":{}},{"cell_type":"code","source":"processedTrain = []\n\nfor image_id in list(train['image_id']):\n    \n    #Read Image\n    imagePath = index_paths[image_id] \n    fullFromPath = os.path.join(repBase, imagePath)\n    img = cv2.imread(fullFromPath)\n    \n    #Compute boxes\n    bboxes = boxesWithinImage(img)\n    \n    #Remove false ones\n    surfaceImage = img.shape[0] * img.shape[1]\n    cleanedBoxes = []\n    for boxe in bboxes:\n        surfaceBoxe = boxe[2] * boxe[3]\n        partBoxe = surfaceBoxe*100/surfaceImage\n        if(partBoxe>5):\n            cleanedBoxes.append(boxe)\n    \n    #Build dataset\n    for i, boxe in enumerate(cleanedBoxes):\n        newRow = [image_id, i+1, boxe]\n        processedTrain.append(newRow)\n        ","metadata":{"execution":{"iopub.status.busy":"2023-10-31T11:10:40.698189Z","iopub.execute_input":"2023-10-31T11:10:40.698723Z","iopub.status.idle":"2023-10-31T11:13:41.884692Z","shell.execute_reply.started":"2023-10-31T11:10:40.69868Z","shell.execute_reply":"2023-10-31T11:13:41.883552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"processedTrain = pd.DataFrame(processedTrain)  \nprocessedTrain.columns = ['image_id', 'slice', 'boxe']\nprocessedTrain","metadata":{"execution":{"iopub.status.busy":"2023-10-31T11:19:04.199043Z","iopub.execute_input":"2023-10-31T11:19:04.199512Z","iopub.status.idle":"2023-10-31T11:19:04.221158Z","shell.execute_reply.started":"2023-10-31T11:19:04.199478Z","shell.execute_reply":"2023-10-31T11:19:04.219647Z"},"trusted":true},"execution_count":null,"outputs":[]}]}