{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport sys\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\n\n\nimport seaborn as sns\nimport cv2\nfrom glob import glob\nfrom pprint import pprint\nfrom collections import defaultdict\nimport gc\nfrom openslide import OpenSlide\nfrom PIL import Image\n\n\nImage.MAX_IMAGE_PIXELS = None\ntrain_images = glob(\"/kaggle/input/mayo-clinic-strip-ai/train/*\")\ntest_images = glob(\"/kaggle/input/mayo-clinic-strip-ai/test/*\")\nother_images = glob(\"/kaggle/input/mayo-clinic-strip-ai/other/*\")\n\n\n\n","metadata":{"execution":{"iopub.status.busy":"2022-10-09T15:53:50.481222Z","iopub.execute_input":"2022-10-09T15:53:50.481596Z","iopub.status.idle":"2022-10-09T15:53:51.788138Z","shell.execute_reply.started":"2022-10-09T15:53:50.48151Z","shell.execute_reply":"2022-10-09T15:53:51.786987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.rcParams.update({'font.size': 6})\nplt.style.use('default')\nimg_path = train_images[0]\nimg = Image.open(img_path) \nwidth,height = img.size\nprint(img.size)\nimg.thumbnail((50,50), Image.Resampling.LANCZOS)\nplt.xticks(list(range(1,51,2)))\nplt.yticks(list(range(1,51,2)))\nplt.imshow(X = img)\nprint(\"ok\")","metadata":{"execution":{"iopub.status.busy":"2022-10-09T15:57:24.973715Z","iopub.execute_input":"2022-10-09T15:57:24.974169Z","iopub.status.idle":"2022-10-09T15:57:38.222157Z","shell.execute_reply.started":"2022-10-09T15:57:24.974131Z","shell.execute_reply":"2022-10-09T15:57:38.218919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.style.use('default')\n# left, up, right, down\n\nimg_cut = img.crop((3,31,25,45))\ndel img\n\nprint(img_cut.size)\nimg_cut.thumbnail((50,50), Image.Resampling.LANCZOS)\nplt.imshow(img_cut)\nprint(\"ok\")","metadata":{"execution":{"iopub.status.busy":"2022-10-09T16:00:26.277391Z","iopub.execute_input":"2022-10-09T16:00:26.2778Z","iopub.status.idle":"2022-10-09T16:00:26.486624Z","shell.execute_reply.started":"2022-10-09T16:00:26.277767Z","shell.execute_reply":"2022-10-09T16:00:26.485696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(1,5, figsize=(16,16))\ntrain_images\nfor ax in axes.reshape(-1):\n    img_path = np.random.choice(LAA_imgs)\n    img = Image.open(img_path)   \n    img.thumbnail((300,300), Image.Resampling.LANCZOS)\n    ax.imshow(img), ax.set_title(\"target: LAA\")\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-10-09T08:26:26.976415Z","iopub.status.idle":"2022-10-09T08:26:26.97836Z","shell.execute_reply.started":"2022-10-09T08:26:26.978076Z","shell.execute_reply":"2022-10-09T08:26:26.978104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"type(img)","metadata":{"execution":{"iopub.status.busy":"2022-10-07T11:50:24.912985Z","iopub.execute_input":"2022-10-07T11:50:24.914162Z","iopub.status.idle":"2022-10-07T11:50:24.920851Z","shell.execute_reply.started":"2022-10-07T11:50:24.91412Z","shell.execute_reply":"2022-10-07T11:50:24.919606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Note that PIL stores an image as a PIL object. This is different to CV2 and skimage which store images as numpy arrays - what eases significantly the postprocessing.\n\n**First impressions:**\n1. Images sizes are from small ones to a high-resolution ones\n2. Images have different aspect ratios\n3. A significant amount of images is a background\n4. Backgrounds have different colours\n5. Cloths are usually in the form of multiple small pieces\n6. Blood cloths have different colours","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:43:16.533119Z","iopub.execute_input":"2022-07-09T10:43:16.533551Z","iopub.status.idle":"2022-07-09T10:43:16.721438Z","shell.execute_reply.started":"2022-07-09T10:43:16.533516Z","shell.execute_reply":"2022-07-09T10:43:16.72019Z"}}},{"cell_type":"markdown","source":"Let's look at the biggest image metadata. Also, if you try to open it using PIL, CV2 or Matplotlib without scaling, on a standard Kaggle notebook you'll run out of memory.","metadata":{}},{"cell_type":"code","source":"biggest_img_path = image_data.iloc[image_data[\"size\"].idxmax(),:]['path']\n\nimg = Image.open(biggest_img_path)\nimg.thumbnail((400,400), Image.Resampling.LANCZOS)\nplt.title(\"The biggest image\")\nplt.imshow(img)\nplt.axis('off'); plt.show()\ndel img","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-10-07T11:50:24.922288Z","iopub.execute_input":"2022-10-07T11:50:24.922686Z","iopub.status.idle":"2022-10-07T11:51:55.56227Z","shell.execute_reply.started":"2022-10-07T11:50:24.922649Z","shell.execute_reply":"2022-10-07T11:51:55.56102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now let's look at 3 smallest images. I'll use here, for demonstration, matplotlib library. Documentation and tutorial regarding image reading is [here](https://matplotlib.org/stable/tutorials/introductory/images.html).","metadata":{}},{"cell_type":"code","source":"smallest_3img_data = image_data.sort_values(by=\"size\")[:3]\nsmallest_3img_paths = smallest_3img_data['path'].values","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-10-07T11:51:55.565793Z","iopub.execute_input":"2022-10-07T11:51:55.566321Z","iopub.status.idle":"2022-10-07T11:51:55.596714Z","shell.execute_reply.started":"2022-10-07T11:51:55.56627Z","shell.execute_reply":"2022-10-07T11:51:55.59584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.image as mpimg\n\nfig, axes = plt.subplots(1,3, figsize=(14,14))\nfor i, ax in enumerate(axes.reshape(-1)):\n    img = mpimg.imread(smallest_3img_paths[i])   \n    ax.imshow(img), plt.axis('off')\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-10-07T11:51:55.598239Z","iopub.execute_input":"2022-10-07T11:51:55.599298Z","iopub.status.idle":"2022-10-07T11:53:12.323288Z","shell.execute_reply.started":"2022-10-07T11:51:55.599257Z","shell.execute_reply":"2022-10-07T11:53:12.322167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2.2.2 Displaying, resizing and manipulation using CV2 and skimage\n\nHere we're going to open the full image using *Computer Vision* library CV2. We can use this library also for image manipulations what I'll show you as well. You can also detect edges bu using the Canny edge detector.  \nNOTE: CV2 library has an image size limit and will give you an error with some of the images from our dataset. There's a workaround by changing the environmental variable (see section 0) but even then some images will not fit into memory. That's why we are going also to use an other library later.\n\nWe can manipulate this image in couple of ways. For this I've chosen an exemplary image.","metadata":{}},{"cell_type":"code","source":"img_path = train_images[261]\nimg_name = img_path[-12:-4]\nsel_img = image_data[image_data['image_id']==img_name]\nsel_img","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-10-07T11:53:12.32495Z","iopub.execute_input":"2022-10-07T11:53:12.32536Z","iopub.status.idle":"2022-10-07T11:53:12.369398Z","shell.execute_reply.started":"2022-10-07T11:53:12.325321Z","shell.execute_reply":"2022-10-07T11:53:12.368398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's look how does it look using matplotlib (why not CV2? you'll find something peculiar about CV2 later).","metadata":{}},{"cell_type":"code","source":"img = mpimg.imread(img_path)   \nplt.imshow(img), plt.axis('off')\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-10-07T11:53:12.370724Z","iopub.execute_input":"2022-10-07T11:53:12.371479Z","iopub.status.idle":"2022-10-07T11:53:40.862278Z","shell.execute_reply.started":"2022-10-07T11:53:12.371429Z","shell.execute_reply":"2022-10-07T11:53:40.861259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" Now I'll read this image in CV2 and I'll rescale it. The size will be reduces to 10% of the original scale - you can confirm it works by looking at the right axes or simply plotting a shape of a second image.","metadata":{}},{"cell_type":"code","source":"img = cv2.imread(img_path, cv2.IMREAD_COLOR)\nimage_resized = cv2.resize(img, (0,0), fx=0.05, fy=0.05)\n\nfig, ax = plt.subplots(1,2, figsize=(12,12))\nax[0].imshow(img)\nax[1].imshow(image_resized)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-07T11:53:40.863578Z","iopub.execute_input":"2022-10-07T11:53:40.864234Z","iopub.status.idle":"2022-10-07T11:54:06.128297Z","shell.execute_reply.started":"2022-10-07T11:53:40.864196Z","shell.execute_reply":"2022-10-07T11:54:06.127101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In order to open it in RGB we have to do the following operation:","metadata":{}},{"cell_type":"code","source":"img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)","metadata":{"execution":{"iopub.status.busy":"2022-10-07T11:54:06.130045Z","iopub.execute_input":"2022-10-07T11:54:06.130415Z","iopub.status.idle":"2022-10-07T11:54:06.267482Z","shell.execute_reply.started":"2022-10-07T11:54:06.13038Z","shell.execute_reply":"2022-10-07T11:54:06.266126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_resized = cv2.resize(img, (0,0), fx=0.05, fy=0.05)\n\nfig, ax = plt.subplots(1,2, figsize=(12,12))\nax[0].imshow(img)\nax[1].imshow(image_resized)\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-10-07T11:54:06.26905Z","iopub.execute_input":"2022-10-07T11:54:06.270104Z","iopub.status.idle":"2022-10-07T11:54:27.851469Z","shell.execute_reply.started":"2022-10-07T11:54:06.270057Z","shell.execute_reply":"2022-10-07T11:54:27.850364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"You may notice later that images opened in CV2 have different colors than the ones opened in other libraries. This is because CV2 by defoult reads color channgels as BGR (blue-gree-red) not RGB.","metadata":{}},{"cell_type":"code","source":"print(f\"Type of CV2 image: {type(img)}\")\nprint(f\"Memory occupied by CV2 image: {sys.getsizeof(img)/1e6:.2f}MB\")\nprint(f\"Memory occupied by reduced CV2 image: {sys.getsizeof(image_resized)/1e6:.2f}MB\")\nprint(f'A shape of a resized image: ({image_resized.shape[0]}, {image_resized.shape[1]}). Number of color channels: {image_resized.shape[2]}')","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-10-07T11:54:27.8528Z","iopub.execute_input":"2022-10-07T11:54:27.853219Z","iopub.status.idle":"2022-10-07T11:54:27.859218Z","shell.execute_reply.started":"2022-10-07T11:54:27.853173Z","shell.execute_reply":"2022-10-07T11:54:27.85843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Note that CV2 stores image as a numpy array. \n\nBelow are 3 exemplary transformations using CV2:\n* Increasing contract. For this we have to use `addWeighted` function\n* Detecting edges. For this we use Canny method.\n* Reading in a grayscale. Here we use `cv2.COLOR_RGB2GRAY` and we tell `.imshow()` that we are passing a single gray channel.","metadata":{}},{"cell_type":"code","source":"contrast_img = cv2.addWeighted(image_resized, 2.5, np.zeros(image_resized.shape, image_resized.dtype), 0, 0)\nimg_gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n\nmin_intensity_grad, max_intensity_grad = 100, 200\nedge_img = cv2.Canny(image_resized, min_intensity_grad, max_intensity_grad)","metadata":{"execution":{"iopub.status.busy":"2022-10-07T11:54:27.860667Z","iopub.execute_input":"2022-10-07T11:54:27.861668Z","iopub.status.idle":"2022-10-07T11:54:27.984586Z","shell.execute_reply.started":"2022-10-07T11:54:27.861577Z","shell.execute_reply":"2022-10-07T11:54:27.983197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1,3, figsize=(15,20))\nax[0].imshow(contrast_img); ax[0].set_title('increased contrast')\nax[1].imshow(img_gray, cmap='gray', vmin = 0, vmax = 255); ax[1].set_title('grayscale')\nax[2].imshow(edge_img); ax[2].set_title('detected edges')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-07T11:54:27.987262Z","iopub.execute_input":"2022-10-07T11:54:27.988381Z","iopub.status.idle":"2022-10-07T11:54:35.672158Z","shell.execute_reply.started":"2022-10-07T11:54:27.988327Z","shell.execute_reply":"2022-10-07T11:54:35.670419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()\ndel contrast_img, edge_img, img_gray, image_resized","metadata":{"execution":{"iopub.status.busy":"2022-10-07T11:54:35.673752Z","iopub.execute_input":"2022-10-07T11:54:35.674124Z","iopub.status.idle":"2022-10-07T11:54:35.866517Z","shell.execute_reply.started":"2022-10-07T11:54:35.674092Z","shell.execute_reply":"2022-10-07T11:54:35.865378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2.2.3 Transformations using scikit-image\n\nNow the same transformations will be performed using Scikit-Image. This is an another popular python package containing set image processing funcions. Official page: https://scikit-image.org/ but the most interesting section is this: https://scikit-image.org/docs/stable/user_guide/transforming_image_data.html. Skimage stores images as numpy arrays.\n","metadata":{}},{"cell_type":"code","source":"from skimage import io, exposure, feature\nfrom skimage.transform import rescale\nfrom skimage.color import rgb2gray","metadata":{"execution":{"iopub.status.busy":"2022-10-07T11:54:35.868046Z","iopub.execute_input":"2022-10-07T11:54:35.868428Z","iopub.status.idle":"2022-10-07T11:54:37.559565Z","shell.execute_reply.started":"2022-10-07T11:54:35.868396Z","shell.execute_reply":"2022-10-07T11:54:37.558316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img = io.imread(img_path)\nimage_resized = rescale(img, 0.2, multichannel=True, anti_aliasing=True) # creating a scaled image\nincreased_contrast = exposure.adjust_gamma(image_resized, 2)\nimg_gray = rgb2gray(image_resized)\nedge_img = feature.canny(img_gray, sigma=3) # canny works only with grayscale images\n\nfig, ax = plt.subplots(1,4, figsize=(15,20))\nax[0].imshow(image_resized); ax[0].set_title('image resized')\nax[1].imshow(increased_contrast); ax[1].set_title('increased contrast')\nax[2].imshow(edge_img); ax[2].set_title('detected edges')\nax[3].imshow(img_gray, cmap='gray'); ax[3].set_title('grayscale')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-07T11:54:37.561337Z","iopub.execute_input":"2022-10-07T11:54:37.561708Z","iopub.status.idle":"2022-10-07T11:55:47.244628Z","shell.execute_reply.started":"2022-10-07T11:54:37.561675Z","shell.execute_reply":"2022-10-07T11:55:47.243288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2.2.4 Using OpenSlide\n\nHere we're going to open a slide (image) and plot only a zoomed region. It is achieved by using `slide.read_region()` function. OpenSlide stores images in memory as Openslide objects.","metadata":{}},{"cell_type":"code","source":"slide = OpenSlide(img_path) # opening a full slide\n\nregion = (2500, 2000) # location of the top left pixel\nlevel = 0 # level of the picture (we have only 0)\nsize = (3500, 3500) # region size in pixels\n\nregion = slide.read_region(region, level, size)\n\nplt.figure(figsize=(15, 15))\nplt.imshow(region)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-07T11:55:47.246441Z","iopub.execute_input":"2022-10-07T11:55:47.247891Z","iopub.status.idle":"2022-10-07T11:55:50.478143Z","shell.execute_reply.started":"2022-10-07T11:55:47.247838Z","shell.execute_reply":"2022-10-07T11:55:50.47582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"What do we actually see here? I'm not a doctor so I'll use a picture from the wikipedia: \n\n<img src=\"https://upload.wikimedia.org/wikipedia/commons/0/06/Composition_of_a_fresh_thrombus.jpg\" width=\"648\" />","metadata":{}},{"cell_type":"code","source":"type(slide)","metadata":{"execution":{"iopub.status.busy":"2022-10-07T11:55:50.480499Z","iopub.execute_input":"2022-10-07T11:55:50.480984Z","iopub.status.idle":"2022-10-07T11:55:50.489626Z","shell.execute_reply.started":"2022-10-07T11:55:50.480942Z","shell.execute_reply":"2022-10-07T11:55:50.488241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3 Sliced version of the database\n\nList od the datasets: https://www.kaggle.com/competitions/mayo-clinic-strip-ai/discussion/335755. In order to add thia database to the notebook you have to use \"+ Add data\" option in the upper right corner. I'm going to use skimage library here.","metadata":{}},{"cell_type":"code","source":"sliced_train_images = glob(\"../input/mayo-clinic-1024-jpg-part1/train/*\")","metadata":{"execution":{"iopub.status.busy":"2022-10-07T11:55:50.491827Z","iopub.execute_input":"2022-10-07T11:55:50.492368Z","iopub.status.idle":"2022-10-07T11:55:53.768747Z","shell.execute_reply.started":"2022-10-07T11:55:50.492324Z","shell.execute_reply":"2022-10-07T11:55:53.767302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list(axes.flat)","metadata":{"execution":{"iopub.status.busy":"2022-10-07T12:02:31.052473Z","iopub.execute_input":"2022-10-07T12:02:31.052903Z","iopub.status.idle":"2022-10-07T12:02:31.061851Z","shell.execute_reply.started":"2022-10-07T12:02:31.052867Z","shell.execute_reply":"2022-10-07T12:02:31.060406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(6,6, figsize=(20,20))\nfor i, ax in enumerate(axes.flat):\n    img = io.imread(sliced_train_images[i])\n    ax.imshow(img); ax.set_title(i)\nplt.tight_layout()\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-10-07T11:55:53.770247Z","iopub.execute_input":"2022-10-07T11:55:53.770652Z","iopub.status.idle":"2022-10-07T11:56:05.121371Z","shell.execute_reply.started":"2022-10-07T11:55:53.770614Z","shell.execute_reply":"2022-10-07T11:56:05.119955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As you can see a lot of these pictures contains just a background. Dealing with this problem will be another step to perform. Below some more detailed introduction into image processing using skimage. Stay tuned and if you liked what you've seen so far don't hesitate to upvote!","metadata":{}},{"cell_type":"code","source":"img_obj_ids = [1, 2, 4, 11, 12, 14]\nimg_bck_ids = [0, 3, 5, 6, 7, 8]\n\nimgs_object = []\nimgs_backgr = []\n\nfor i in img_obj_ids:\n    imgs_object.append(io.imread(sliced_train_images[i]))\n    \nfor i in img_bck_ids:\n    imgs_backgr.append(io.imread(sliced_train_images[i]))","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-10-07T11:56:05.123001Z","iopub.execute_input":"2022-10-07T11:56:05.123436Z","iopub.status.idle":"2022-10-07T11:56:05.416907Z","shell.execute_reply.started":"2022-10-07T11:56:05.123396Z","shell.execute_reply":"2022-10-07T11:56:05.415491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Analysis of color channels\nEach image has 3 color channels (RGB): Red, Green, Blue. Below I'll display these channels. As each channel represents only intensity of a given single color I'll display these images in a grayscale to avoid any confusion (by displayed colors otherwise).\n\nNOTE: Values of each colour channel are in range from 0 to 255.","metadata":{}},{"cell_type":"code","source":"imgs_object[0].shape","metadata":{"execution":{"iopub.status.busy":"2022-10-07T11:56:05.418648Z","iopub.execute_input":"2022-10-07T11:56:05.419567Z","iopub.status.idle":"2022-10-07T11:56:05.430564Z","shell.execute_reply.started":"2022-10-07T11:56:05.419514Z","shell.execute_reply":"2022-10-07T11:56:05.428863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(1,3, figsize=(12,12))\ncolors = ['red', 'green', 'blue']\nfor i, ax in enumerate(axes):\n    ax.imshow(imgs_object[0][:,:,i], cmap='gray')\n    ax.set_title(colors[i]+' channel'), ax.axis('off')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-07T11:56:05.432742Z","iopub.execute_input":"2022-10-07T11:56:05.433336Z","iopub.status.idle":"2022-10-07T11:56:05.98984Z","shell.execute_reply.started":"2022-10-07T11:56:05.433284Z","shell.execute_reply":"2022-10-07T11:56:05.988206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_obj = imgs_object[0]\nimg_bck = imgs_backgr[0]\n\nimg_obj_channels = [img_obj[0], img_obj[1], img_obj[2]]\nimg_bck_channels = [img_bck[0], img_bck[1], img_bck[2]]","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-10-07T11:56:05.991537Z","iopub.execute_input":"2022-10-07T11:56:05.992037Z","iopub.status.idle":"2022-10-07T11:56:06.000278Z","shell.execute_reply.started":"2022-10-07T11:56:05.991987Z","shell.execute_reply":"2022-10-07T11:56:05.998914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Images in skimage are stored as Numpy arrays and all Numpy functions can be applied to it.","metadata":{}},{"cell_type":"code","source":"# 1024*1024\ntype(img_obj_channels[0])","metadata":{"execution":{"iopub.status.busy":"2022-10-07T11:56:06.0019Z","iopub.execute_input":"2022-10-07T11:56:06.002338Z","iopub.status.idle":"2022-10-07T11:56:06.014745Z","shell.execute_reply.started":"2022-10-07T11:56:06.002278Z","shell.execute_reply":"2022-10-07T11:56:06.013483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Images with objects\")\nfor i, channel in enumerate(img_obj_channels):\n    print(f\"Channel {i} values: min: {channel.min()}, max: {channel.max()}, average: {channel.mean()}, std: {channel.std()}\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-10-07T11:56:06.024968Z","iopub.execute_input":"2022-10-07T11:56:06.025424Z","iopub.status.idle":"2022-10-07T11:56:06.036519Z","shell.execute_reply.started":"2022-10-07T11:56:06.025383Z","shell.execute_reply":"2022-10-07T11:56:06.035127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(1,3, figsize=(18,6))\nfor i, ax in enumerate(axes):\n    df = img_obj_channels[i].ravel()\n    sns.histplot(df, bins=np.arange(0,255), ax=ax)\n    ax.set_title(colors[i]+' channel')\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-10-07T11:56:06.037593Z","iopub.execute_input":"2022-10-07T11:56:06.037928Z","iopub.status.idle":"2022-10-07T11:56:07.873173Z","shell.execute_reply.started":"2022-10-07T11:56:06.037897Z","shell.execute_reply":"2022-10-07T11:56:07.871848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Images with background only\")\nfor i, channel in enumerate(img_bck_channels):\n    print(f\"Channel {i} values: min: {channel.min()}, max: {channel.max()}, average: {channel.mean()}\")","metadata":{"execution":{"iopub.status.busy":"2022-10-07T11:56:07.875013Z","iopub.execute_input":"2022-10-07T11:56:07.875811Z","iopub.status.idle":"2022-10-07T11:56:07.882761Z","shell.execute_reply.started":"2022-10-07T11:56:07.875772Z","shell.execute_reply":"2022-10-07T11:56:07.881428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(1,3, figsize=(18,6))\nfor i, ax in enumerate(axes):\n    df = img_bck_channels[i].ravel()\n    sns.histplot(df, bins=np.arange(0,255), ax=ax)\n    ax.set_title(colors[i]+' channel')\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-10-07T11:56:07.884394Z","iopub.execute_input":"2022-10-07T11:56:07.885938Z","iopub.status.idle":"2022-10-07T11:56:10.096574Z","shell.execute_reply.started":"2022-10-07T11:56:07.885875Z","shell.execute_reply":"2022-10-07T11:56:10.09526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"centers.plot.bar(rot=0)\nplt.show()Assuming that images with object have a wide range of values (high std. error) in color channels and background ones have a narrow, this property can be used to distinguish images between them. Also we don't have to use 3 color channels - as we see the difference is significant in all three channels. So reading images in grayscale should do the work as well.\n\nNOTE: When working with grayscale images each pixel is now a value of intensity only, which is in range from 0 to 1. ","metadata":{}},{"cell_type":"code","source":"imgs_object = []\nimgs_backgr = []\n\nfor i in img_obj_ids:\n    img = io.imread(sliced_train_images[i])\n    imgs_object.append(rgb2gray(img))\n    \nfor i in img_bck_ids:\n    img = io.imread(sliced_train_images[i])\n    imgs_backgr.append(rgb2gray(img))","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-10-07T11:56:10.098335Z","iopub.execute_input":"2022-10-07T11:56:10.099319Z","iopub.status.idle":"2022-10-07T11:56:10.515927Z","shell.execute_reply.started":"2022-10-07T11:56:10.099265Z","shell.execute_reply":"2022-10-07T11:56:10.514251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"std_dev_obj = []\nfig, axes = plt.subplots(2,3, figsize=(18,6))\nfor i, ax in enumerate(axes.flat):\n    img_matrix = imgs_object[i].ravel()\n    std_dev = np.std(img_matrix)\n    std_dev_obj.append(std_dev)\n    sns.histplot(img_matrix, bins=np.arange(0,1, 0.01), ax=ax)\n    ax.text(0.05,0.85, f'std dev: {std_dev:.2f}', size=15, transform = ax.transAxes)\n    ax.set_title(f'img {i}') \nplt.suptitle('Histograms for images w. object')\nplt.tight_layout()\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-10-07T11:56:10.518466Z","iopub.execute_input":"2022-10-07T11:56:10.519004Z","iopub.status.idle":"2022-10-07T11:56:17.798889Z","shell.execute_reply.started":"2022-10-07T11:56:10.518953Z","shell.execute_reply":"2022-10-07T11:56:17.797539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(2,3, figsize=(18,6))\nfor i, ax in enumerate(axes.flat):\n    img_matrix = imgs_backgr[i].ravel()\n    std_dev = np.std(img_matrix)\n    std_dev_obj.append(std_dev)\n    sns.histplot(img_matrix, bins=np.arange(0,1, 0.01), ax=ax)\n    ax.text(0.05,0.85, f'std_dev: {std_dev:.2f}', size=15, transform = ax.transAxes)\n    ax.set_title(f'img {i}')  \nplt.suptitle('Histogram for images wo. object')\nplt.tight_layout()\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-10-07T11:56:17.800699Z","iopub.execute_input":"2022-10-07T11:56:17.801106Z","iopub.status.idle":"2022-10-07T11:56:24.907661Z","shell.execute_reply.started":"2022-10-07T11:56:17.801071Z","shell.execute_reply":"2022-10-07T11:56:24.906276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}