{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import time\nimport cv2 as cv\nimport pandas as pd\nimport os\nimport matplotlib.pyplot as plt\nimport numpy as np\n\nfrom sklearn.preprocessing import LabelEncoder\n\nimport tensorflow as tf\nfrom tensorflow.keras.layers import Dense, MaxPooling2D, Dropout, Conv2D, Flatten\nfrom tensorflow.keras.callbacks import EarlyStopping\nfrom keras.models import Sequential","metadata":{"execution":{"iopub.status.busy":"2023-11-09T10:11:30.981659Z","iopub.execute_input":"2023-11-09T10:11:30.981911Z","iopub.status.idle":"2023-11-09T10:11:39.660644Z","shell.execute_reply.started":"2023-11-09T10:11:30.981888Z","shell.execute_reply":"2023-11-09T10:11:39.659802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Read the Data","metadata":{}},{"cell_type":"code","source":"# Load the Data\ntrain = pd.read_csv(\"/kaggle/input/UBC-OCEAN/train.csv\")\n\n# is_ma = True\nistma_False = train[train[\"is_tma\"]==False]\nistma_False.tail()","metadata":{"execution":{"iopub.status.busy":"2023-11-09T10:11:39.662283Z","iopub.execute_input":"2023-11-09T10:11:39.66284Z","iopub.status.idle":"2023-11-09T10:11:39.694123Z","shell.execute_reply.started":"2023-11-09T10:11:39.662811Z","shell.execute_reply":"2023-11-09T10:11:39.693138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I will compare binary thresholding using CLAHE and without CLAHE","metadata":{}},{"cell_type":"code","source":"# Sample images for plotting\npath = \"../input/UBC-OCEAN/test_thumbnails/41_thumbnail.png\"\nimage = cv.imread(path, 0)\nimage = cv.resize(image, (224, 224), interpolation=cv.INTER_AREA)\n\n# CLAHE (Contrast Limited Adaptive Histogram Equalization)\nclahe = cv.createCLAHE(clipLimit=40.0, tileGridSize=(2, 2))\nclahe_img = clahe.apply(image)\nret, thresh = cv.threshold(image, 150, 255, cv.THRESH_BINARY_INV)\nret, thresh1 = cv.threshold(clahe_img, 150, 255, cv.THRESH_BINARY_INV)\n\n#  Plotting\nfig, ax = plt.subplots(2,2, figsize=(8,5))\nax[0,0].imshow(image, cmap='gray'),ax[0,0].set_title('img+thresh'), ax[0,0].axis('off')\nax[0,1].imshow(clahe_img, cmap='gray'), ax[0,1].set_title('img+CLAHE+thresh'), ax[0,1].axis('off')\nax[1,0].imshow(thresh, cmap='gray'), ax[1,0].axis('off')\nax[1,1].imshow(thresh1, cmap='gray'), ax[1,1].axis('off')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-09T10:11:39.69525Z","iopub.execute_input":"2023-11-09T10:11:39.695523Z","iopub.status.idle":"2023-11-09T10:11:40.320956Z","shell.execute_reply.started":"2023-11-09T10:11:39.6955Z","shell.execute_reply":"2023-11-09T10:11:40.320051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Preprocessing","metadata":{}},{"cell_type":"markdown","source":"### Slicing the Images","metadata":{}},{"cell_type":"code","source":"def get_slices(img_path, lower_limit, higher_limit=9999999, zoom_out=0, crop_square=0):\n    #Reading image\n    if '.png' in img_path: \n        img = cv.imread(img_path, 0)\n    else:\n        img = cv.imread(img_path, 0)\n    \n    width = img.shape[1]\n    height = img.shape[0]\n    \n    # CLAHE + binary thresholding\n    clahe = cv.createCLAHE(clipLimit=40.0, tileGridSize=(2,2))\n    clahe_img = clahe.apply(img)\n    ret, thresh = cv.threshold(clahe_img, 150, 255, cv.THRESH_BINARY_INV)\n\n    # detect the contours on the binary image using cv.CHAIN_APPROX_NONE\n    contours, hierarchy = cv.findContours(image=thresh, mode=cv.RETR_EXTERNAL, \n                                           method=cv.CHAIN_APPROX_NONE)\n    contours = sorted(contours, key=cv.contourArea, reverse=True)\n\n    #----------------------------------------------------------------------------#\n    zoom_out_x1  = zoom_out\n    zoom_out_x0  = zoom_out\n    zoom_out_y0  = zoom_out\n    zoom_out_y1  = zoom_out\n\n    temps = []\n    # Find bounding box \n    for c in contours:\n        x,y,w,h = cv.boundingRect(c)\n        area = cv.contourArea(c)\n        \n        if area>lower_limit and area<higher_limit:\n            # Crop the image so that it is the same size\n            if crop_square>0:\n                c_height = (y+h)-(y)\n                c_witdth = (x+w)-(x)\n                \n                center_y = int(c_height/2)\n                center_x = int(c_witdth/2)\n                \n                center_y = (center_y + y)\n                center_x = (center_x + x)\n                zoom_out = int(crop_square/2)\n                zoom_out_x1  = zoom_out\n                zoom_out_x0  = zoom_out\n                zoom_out_y0  = zoom_out\n                zoom_out_y1  = zoom_out\n                \n                if center_y - zoom_out_y0 < 0: \n                    zoom_out_y1 = zoom_out_y1 - (center_y-zoom_out_y0)\n                    zoom_out_y0 = center_y\n                else:pass\n                \n                if center_y + zoom_out_y1 > height:\n                    zoom_out_y1 = height - (center_y)\n                    zoom_out_y0 = zoom_out_y0 + (zoom_out-zoom_out_y1)\n                else:pass\n                \n                if center_x - zoom_out_x0<0:\n                    zoom_out_x1 = zoom_out_x1 - (center_x-zoom_out_x0)\n                    zoom_out_x0 = center_x\n                else:pass\n                \n                if center_x+zoom_out > width:\n                    zoom_out_x1 = width - (center_x)\n                    zoom_out_x0 = zoom_out_x0 + (zoom_out-zoom_out_x1)\n                else:pass\n                \n                \n                temp = img[center_y-zoom_out_y0:center_y+zoom_out_y1, center_x-zoom_out_x0:center_x+zoom_out_x1]\n                \n            else:\n                \n                if y-zoom_out_y0 < 0: \n                    zoom_out_y1 = zoom_out_y1 - (y-zoom_out_y0)\n                    zoom_out_y0 = y\n                elif y+h+zoom_out_y1 > height:\n                    zoom_out_y1 = height - (y+h)\n                    zoom_out_y0 = zoom_out_y0 + (zoom_out-zoom_out_y1)\n\n                elif x-zoom_out_x0<0:\n                    zoom_out_x1 = zoom_out_x1 - (x-zoom_out_x0)\n                    zoom_out_x0 = x\n                elif x+w+zoom_out > width:\n                    zoom_out_x1 = width - (x+w)\n                    zoom_out_x0 = zoom_out_x0 + (zoom_out-zoom_out_x1)\n\n                temp = img[y-zoom_out_y0:y+h+zoom_out_y1, x-zoom_out_x0:x+w+zoom_out_x1]\n            \n            \n            if temp.size==0:\n                temp = img[y:y+h, x:x+w]\n                \n        else: continue\n            \n        temps.append(temp)\n            \n    return temps","metadata":{"execution":{"iopub.status.busy":"2023-11-09T10:11:40.323815Z","iopub.execute_input":"2023-11-09T10:11:40.324298Z","iopub.status.idle":"2023-11-09T10:11:40.343168Z","shell.execute_reply.started":"2023-11-09T10:11:40.324266Z","shell.execute_reply":"2023-11-09T10:11:40.341763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def data_img(dataframe, path):\n    classes = []\n    images = []\n    for data in dataframe.iterrows():\n        img_path = os.path.join(path,f\"{str(data[1][0])}_thumbnail.png\")\n        contours = get_slices(img_path, 5000, zoom_out=100, crop_square=1000)\n\n        for lol in contours:\n            img_resize = cv.resize(lol, (224, 224), interpolation=cv.INTER_AREA)\n            w,h = img_resize.shape\n            rgb_img = np.zeros((w,h,3), dtype=np.uint8)\n            rgb_img[:,:,0] = img_resize\n            rgb_img[:,:,1] = img_resize\n            rgb_img[:,:,2] = img_resize\n\n            images.append(rgb_img)\n            classes.append(data[1][1])\n        \n    return images, classes","metadata":{"execution":{"iopub.status.busy":"2023-11-09T10:11:40.344355Z","iopub.execute_input":"2023-11-09T10:11:40.344705Z","iopub.status.idle":"2023-11-09T10:11:40.356598Z","shell.execute_reply.started":"2023-11-09T10:11:40.344679Z","shell.execute_reply":"2023-11-09T10:11:40.355661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = \"/kaggle/input/UBC-OCEAN/train_thumbnails\"\nimages, classes = data_img(istma_False, path)\n\nimages = np.array(images)\nprint(images.shape)\nprint(len(classes))","metadata":{"execution":{"iopub.status.busy":"2023-11-09T10:11:40.357933Z","iopub.execute_input":"2023-11-09T10:11:40.358199Z","iopub.status.idle":"2023-11-09T10:15:08.917486Z","shell.execute_reply.started":"2023-11-09T10:11:40.358176Z","shell.execute_reply":"2023-11-09T10:15:08.916423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Encoded","metadata":{}},{"cell_type":"code","source":"# Encode classes\nle = LabelEncoder()\nlabel = le.fit_transform(classes)\nlabel_list = list(le.classes_)\n\nprint(len(label))\nprint(label_list)","metadata":{"execution":{"iopub.status.busy":"2023-11-09T10:15:08.918697Z","iopub.execute_input":"2023-11-09T10:15:08.918987Z","iopub.status.idle":"2023-11-09T10:15:08.927684Z","shell.execute_reply.started":"2023-11-09T10:15:08.918962Z","shell.execute_reply":"2023-11-09T10:15:08.926658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Sample\nrandom_n = np.random.randint(0,512,3)\n\nfig, ax = plt.subplots(1,3, figsize=(12, 8))\nj=0\nfor i in random_n:\n    ax[j].imshow(images[i], cmap='gray'), ax[j].set_title(classes[i])\n    j+=1\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-09T10:15:08.929073Z","iopub.execute_input":"2023-11-09T10:15:08.929701Z","iopub.status.idle":"2023-11-09T10:15:09.592066Z","shell.execute_reply.started":"2023-11-09T10:15:08.929668Z","shell.execute_reply":"2023-11-09T10:15:09.591129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.applications.efficientnet import EfficientNetB0\nbase_model = EfficientNetB0(\n    include_top=False,\n    input_shape=(224, 224, 3),\n    weights='imagenet'\n)\n\nfor layer in base_model.layers:\n    layer.trainable = False","metadata":{"execution":{"iopub.status.busy":"2023-11-09T10:15:09.593288Z","iopub.execute_input":"2023-11-09T10:15:09.593928Z","iopub.status.idle":"2023-11-09T10:15:15.481616Z","shell.execute_reply.started":"2023-11-09T10:15:09.593893Z","shell.execute_reply":"2023-11-09T10:15:15.480456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = tf.keras.layers.Flatten()(base_model.output)\nx = tf.keras.layers.Dense(64, activation='relu')(x)\nx = tf.keras.layers.Dense(5, activation='softmax')(x)\n\nmodel = tf.keras.Model(inputs=base_model.input, outputs=x)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-11-09T10:15:15.48498Z","iopub.execute_input":"2023-11-09T10:15:15.48529Z","iopub.status.idle":"2023-11-09T10:15:16.050524Z","shell.execute_reply.started":"2023-11-09T10:15:15.485265Z","shell.execute_reply":"2023-11-09T10:15:16.047982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(\n    optimizer=tf.keras.optimizers.Adam(learning_rate=0.001),\n    loss='sparse_categorical_crossentropy',\n    metrics= [\"accuracy\"]\n)","metadata":{"execution":{"iopub.status.busy":"2023-11-09T10:15:16.051558Z","iopub.execute_input":"2023-11-09T10:15:16.051858Z","iopub.status.idle":"2023-11-09T10:15:16.096461Z","shell.execute_reply.started":"2023-11-09T10:15:16.051834Z","shell.execute_reply":"2023-11-09T10:15:16.09515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_callback = EarlyStopping(\n    monitor = 'val_loss', mode = 'min',\n    patience = 3\n)\n\nhistory = model.fit(\n    images, label,\n    batch_size = 32,\n    validation_split = 0.2,\n    epochs = 100,\n    callbacks = [model_callback]\n)","metadata":{"execution":{"iopub.status.busy":"2023-11-09T10:15:16.102682Z","iopub.execute_input":"2023-11-09T10:15:16.102951Z","iopub.status.idle":"2023-11-09T10:16:16.700336Z","shell.execute_reply.started":"2023-11-09T10:15:16.102928Z","shell.execute_reply":"2023-11-09T10:16:16.699444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nplt.figure(figsize=(8, 3))\nplt.plot(history.epoch, history.history['loss'])\n#plt.plot(history.epoch, history.history['val_loss'])\nplt.legend(['train loss', 'val loss'])\nplt.title('Loss Diagram')\nplt.xlabel('Epoch(s)')\nplt.ylabel('Loss')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-09T10:16:16.701925Z","iopub.execute_input":"2023-11-09T10:16:16.702654Z","iopub.status.idle":"2023-11-09T10:16:46.385499Z","shell.execute_reply.started":"2023-11-09T10:16:16.702614Z","shell.execute_reply":"2023-11-09T10:16:46.384562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Testing","metadata":{}},{"cell_type":"markdown","source":"I will explain step by step how images are preprocessed starting from Read image -> Slicing the images (because the image has very high dimensions) -> Take several image slices using \"contours & contoursArea opencv\" -> Modeling using Efficientnet","metadata":{}},{"cell_type":"code","source":"# Read the image\ntest_path = \"/kaggle/input/UBC-OCEAN/test_thumbnails/41_thumbnail.png\"\nimg = cv.imread(test_path, 0)\nplt.imshow(img,cmap='gray')\nimg.shape","metadata":{"execution":{"iopub.status.busy":"2023-11-09T10:16:46.387373Z","iopub.execute_input":"2023-11-09T10:16:46.38807Z","iopub.status.idle":"2023-11-09T10:16:47.164003Z","shell.execute_reply.started":"2023-11-09T10:16:46.388035Z","shell.execute_reply":"2023-11-09T10:16:47.163063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Slicing the Images\nimg_test = []\ncontours = get_slices(test_path, 5000, zoom_out=100, crop_square=1000)\nfor lol in contours:\n    img_resize = cv.resize(lol, (224, 224), interpolation=cv.INTER_AREA)\n    w,h = img_resize.shape\n    rgb_img = np.zeros((w,h,3), dtype=np.uint8)\n    rgb_img[:,:,0] = img_resize\n    rgb_img[:,:,1] = img_resize\n    rgb_img[:,:,2] = img_resize\n    \n    img_test.append(rgb_img)\n    \nimg_test = np.array(img_test)\nprint(img_test.shape)\n\nplt.figure(figsize=(32,32))\nfor i, n in enumerate(img_test):\n    try:\n        plt.subplot(4,4,i+1)\n        plt.grid(False)\n        plt.imshow(n)\n    except:pass","metadata":{"execution":{"iopub.status.busy":"2023-11-09T10:16:47.165137Z","iopub.execute_input":"2023-11-09T10:16:47.165414Z","iopub.status.idle":"2023-11-09T10:16:50.23999Z","shell.execute_reply.started":"2023-11-09T10:16:47.165389Z","shell.execute_reply":"2023-11-09T10:16:50.238996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Model Predict\nclasses = model.predict(img_test)\n\n# Plotting the Result\nplt.figure(figsize=(10,10))\nfor i, n in enumerate(img_test):\n    try:\n        plt.subplot(4,4,i+1)\n        #plt.grid(False)\n        plt.imshow(n, cmap='gray')\n        plt.title(label_list[np.argmax(classes[i])])\n        plt.axis('off')\n    except:pass","metadata":{"execution":{"iopub.status.busy":"2023-11-09T10:16:50.241152Z","iopub.execute_input":"2023-11-09T10:16:50.241452Z","iopub.status.idle":"2023-11-09T10:16:53.178976Z","shell.execute_reply.started":"2023-11-09T10:16:50.241426Z","shell.execute_reply":"2023-11-09T10:16:53.178049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From one image --> 8 images (results from sliced images), then predicted. From 8 classes of prediction results:\n- 5/8 -> HGSC\n- 3/8 -> EC","metadata":{}},{"cell_type":"code","source":"v_result = []\nfor i in classes:\n    v_result.append(np.argmax(i))\n\nprint(f\"The result: {v_result}\")\nprint(f\"Voting: {np.argmax(np.bincount(v_result))}\")","metadata":{"execution":{"iopub.status.busy":"2023-11-09T10:28:08.117639Z","iopub.execute_input":"2023-11-09T10:28:08.117985Z","iopub.status.idle":"2023-11-09T10:28:08.124164Z","shell.execute_reply.started":"2023-11-09T10:28:08.11796Z","shell.execute_reply":"2023-11-09T10:28:08.123098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission = pd.read_csv(\"/kaggle/input/UBC-OCEAN/sample_submission.csv\")\nsample_submission['label'] = label_list[np.argmax(np.bincount(v_result))]\n\n# Save the updated DataFrame to a CSV file\nsample_submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-11-09T10:29:35.012064Z","iopub.execute_input":"2023-11-09T10:29:35.012865Z","iopub.status.idle":"2023-11-09T10:29:35.022875Z","shell.execute_reply.started":"2023-11-09T10:29:35.012828Z","shell.execute_reply":"2023-11-09T10:29:35.022009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# If you like, upvote pls  :)","metadata":{}},{"cell_type":"markdown","source":"## Acknowledgement:\n","metadata":{}},{"cell_type":"markdown","source":"Iurick Santos https://www.kaggle.com/code/iuryck/quick-blood-clot-slicing/notebook","metadata":{}}]}