{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **In This Notebook**\nWe went through MayoClinic - StripAI dataset with analyzing and visualizing the data through different types of visualizations, then we ran 2 models with different approaches to solve data unbalancing issue.\n\n**Team:**\n\n1- Omar Khaled\n\n2- Nour Ehab\n\n3- Abdalrahman Al-jammal\n\n4- Philopateer Magdy","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport os\nimport keras.backend as K \nimport matplotlib.pyplot as plt\nimport seaborn as sns\n%matplotlib inline\nimport warnings\nwarnings.filterwarnings('ignore')\nfrom pprint import pprint\nfrom collections import defaultdict\nimport openslide\nfrom openslide import OpenSlide\nfrom glob import glob\nfrom sklearn.model_selection import train_test_split\nfrom tqdm import tqdm\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Dropout, Flatten, Dense\nfrom tensorflow.keras.layers import GlobalMaxPooling2D\nfrom keras.models import load_model","metadata":{"execution":{"iopub.status.busy":"2023-02-19T11:13:24.669941Z","iopub.execute_input":"2023-02-19T11:13:24.670701Z","iopub.status.idle":"2023-02-19T11:13:30.874985Z","shell.execute_reply.started":"2023-02-19T11:13:24.670604Z","shell.execute_reply":"2023-02-19T11:13:30.870529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('../input/mayo-clinic-strip-ai/train.csv')\ntest_df  = pd.read_csv('../input/mayo-clinic-strip-ai/test.csv')\ntrain_df.head()","metadata":{"papermill":{"duration":0.031325,"end_time":"2022-12-13T12:03:24.419596","exception":false,"start_time":"2022-12-13T12:03:24.388271","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-19T11:13:30.880083Z","iopub.execute_input":"2023-02-19T11:13:30.880779Z","iopub.status.idle":"2023-02-19T11:13:30.93119Z","shell.execute_reply.started":"2023-02-19T11:13:30.880742Z","shell.execute_reply":"2023-02-19T11:13:30.929947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data viusalization","metadata":{}},{"cell_type":"code","source":"plt.style.use('Solarize_Light2')\nlabels = train_df.groupby('label')['label'].count().div(len(train_df)).mul(100)\ncenters = train_df.groupby(\"center_id\")['center_id'].count().div(len(train_df)).mul(100)\nfig, ax = plt.subplots(1,2, figsize=(16,5))\nsns.barplot(x=labels.index, y=labels.values, ax=ax[0])\nax[0].set_title(\"Distribution of a target variable\"), ax[0].set_ylabel(\"%\")\nsns.barplot(x=centers.index, y=centers.values, ax=ax[1])\nax[1].set_title(\"Images per clinic center\"), ax[1].set_ylabel(\"%\")\nplt.show()\nprint('Train Size = {}'.format(len(train_df)))\nprint('Test Size = {}'.format(len(test_df)))","metadata":{"execution":{"iopub.status.busy":"2023-02-19T11:13:30.935516Z","iopub.execute_input":"2023-02-19T11:13:30.935866Z","iopub.status.idle":"2023-02-19T11:13:31.347845Z","shell.execute_reply.started":"2023-02-19T11:13:30.935833Z","shell.execute_reply":"2023-02-19T11:13:31.34694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## From this plot we notice that there is a class imbalance which we will deal with later in this notebook","metadata":{}},{"cell_type":"code","source":"train_images = glob(\"/kaggle/input/mayo-clinic-strip-ai/train/*\")\ntest_images = glob(\"/kaggle/input/mayo-clinic-strip-ai/test/*\")\nother_images = glob(\"/kaggle/input/mayo-clinic-strip-ai/other/*\")\nprint(f\"Number of images in a training set: {len(train_images)}\")\nprint(f\"Number of images in a training set: {len(test_images)}\")\nprint(f\"Number of other: {len(other_images)}\")","metadata":{"execution":{"iopub.status.busy":"2023-02-19T11:13:34.456335Z","iopub.execute_input":"2023-02-19T11:13:34.456688Z","iopub.status.idle":"2023-02-19T11:13:35.052827Z","shell.execute_reply.started":"2023-02-19T11:13:34.456658Z","shell.execute_reply":"2023-02-19T11:13:35.05177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_prop = defaultdict(list)\nfor i, path in enumerate(train_images):\n    img_path = train_images[i]\n    slide = OpenSlide(img_path)    \n    img_prop['image_id'].append(img_path[-12:-4])\n    img_prop['width'].append(slide.dimensions[0])\n    img_prop['height'].append(slide.dimensions[1])\n    img_prop['size'].append(round(os.path.getsize(img_path) / 1e6, 2))\n    img_prop['path'].append(img_path)\nimage_data = pd.DataFrame(img_prop)\nimage_data['img_aspect_ratio'] = image_data['width']/image_data['height']\nimage_data.sort_values(by='image_id', inplace=True)\nimage_data.reset_index(inplace=True, drop=True)\nimage_data = image_data.merge(train_df, on='image_id')\nimage_data.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-19T11:13:36.096116Z","iopub.execute_input":"2023-02-19T11:13:36.097029Z","iopub.status.idle":"2023-02-19T11:14:01.933376Z","shell.execute_reply.started":"2023-02-19T11:13:36.096978Z","shell.execute_reply":"2023-02-19T11:14:01.931925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.style.use('Solarize_Light2')\nfig, ax = plt.subplots(1,2, figsize=(16,5))\nsns.histplot(x='size', data = image_data, bins=100, ax=ax[0])\nax[0].set_title(\"Distribution of size\"), ax[0].set_ylabel(\"%\")\nsns.histplot(x='img_aspect_ratio', data = image_data, bins=100, ax=ax[1])\nax[1].set_title(\"Image aspect ratio\"), ax[1].set_ylabel(\"%\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-19T11:14:01.935229Z","iopub.execute_input":"2023-02-19T11:14:01.936072Z","iopub.status.idle":"2023-02-19T11:14:02.581607Z","shell.execute_reply.started":"2023-02-19T11:14:01.936035Z","shell.execute_reply":"2023-02-19T11:14:02.580572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Some Image Displaying on the data**","metadata":{}},{"cell_type":"code","source":"def reading_tiff(image):\n    Reading_Image = cv2.cvtColor(cv2.imread(image),cv2.COLOR_BGR2RGB)\n    return Reading_Image","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def single_display_tiff(image):\n    figure = plt.figure(figsize=(20,20))\n    Reading_Image = cv2.cvtColor(cv2.imread(image),cv2.COLOR_BGR2RGB)\n    plt.xlabel(Reading_Image.shape)\n    plt.ylabel(Reading_Image.size)\n    plt.title(\"TIFF\")\n    plt.imshow(Reading_Image)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def threshold_display(image):\n    figure = plt.figure(figsize=(8,8))\n    Reading_Image = cv2.cvtColor(cv2.imread(image),cv2.COLOR_BGR2RGB)\n    _,Threshold_Image = cv2.threshold(Reading_Image,30,255,cv2.THRESH_BINARY)\n    plt.xlabel(Threshold_Image.shape)\n    plt.ylabel(Threshold_Image.size)\n    plt.title(\"THRESHOLD\")\n    plt.imshow(Threshold_Image)\ndef threshold_reading(image):\n    Reading_Image = cv2.cvtColor(cv2.imread(image),cv2.COLOR_BGR2RGB)\n    _,Threshold_Image = cv2.threshold(Reading_Image,30,255,cv2.THRESH_BINARY)\n    return Threshold_Image","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def adaptive_threshold_display(image):\n    figure = plt.figure(figsize=(8,8))\n    Reading_Image = cv2.imread(image,0)\n    Adaptive_Image = cv2.adaptiveThreshold(Reading_Image,20,cv2.ADAPTIVE_THRESH_GAUSSIAN_C,cv2.THRESH_BINARY_INV,9,2)\n    plt.xlabel(Adaptive_Image.shape)\n    plt.ylabel(Adaptive_Image.size)\n    plt.title(\"ADAPTIVE_THRESHOLD\")\n    plt.imshow(Adaptive_Image)\ndef adaptive_threshold_reading(image):\n    Reading_Image = cv2.imread(image,0)\n    Adaptive_Image = cv2.adaptiveThreshold(Reading_Image,20,cv2.ADAPTIVE_THRESH_GAUSSIAN_C,cv2.THRESH_BINARY_INV,9,2)\n    return Adaptive_Image","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def canny_reading(image):\n    Reading_Image = cv2.cvtColor(cv2.imread(image),cv2.COLOR_BGR2RGB)\n    Canny_Image = cv2.Canny(Reading_Image,5,100)\n    return Canny_Image","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def bitwise_and_display(image):\n    figure = plt.figure(figsize=(8,8))\n    Reading_Image = cv2.cvtColor(cv2.imread(image),cv2.COLOR_BGR2RGB)\n    Reading_Image = cv2.resize(Reading_Image,(180,180))\n    _,Threshold_Image = cv2.threshold(Reading_Image,30,255,cv2.THRESH_BINARY)\n    Threshold_Image = cv2.resize(Threshold_Image,(180,180))\n    Mask_For_Image = cv2.inRange(Reading_Image,Reading_Image,Threshold_Image)\n    Bitwise_Image = cv2.bitwise_and(Reading_Image,Reading_Image,mask=Mask_For_Image)\n    plt.xlabel(Bitwise_Image.shape)\n    plt.ylabel(Bitwise_Image.size)\n    plt.title(\"Bitwise_Image\")\n    plt.imshow(Bitwise_Image)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def different_type_display(image_one,image_two,image_three,image_four):\n    figure,axis = plt.subplots(1,4,figsize=(10,12))\n    axis[0].imshow(image_one)\n    axis[0].set_xlabel(image_one.shape)\n    axis[0].set_ylabel(image_one.size)\n    axis[0].set_title(\"image_one\")\n    axis[1].imshow(image_two)\n    axis[1].set_xlabel(image_two.shape)\n    axis[1].set_ylabel(image_two.size)\n    axis[1].set_title(\"image_two\")\n    axis[2].imshow(image_three)\n    axis[2].set_xlabel(image_three.shape)\n    axis[2].set_ylabel(image_three.size)\n    axis[2].set_title(\"image_three\")\n    axis[3].imshow(image_four)\n    axis[3].set_xlabel(image_four.shape)\n    axis[3].set_ylabel(image_four.size)\n    axis[3].set_title(\"image_four\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Dispalying TIFF images**","metadata":{}},{"cell_type":"code","source":"plt.style.use(\"dark_background\")\nsingle_display_tiff('../input/mayo-clinic-strip-ai/train/0cc0bc_0.tif')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.style.use(\"dark_background\")\nthreshold_display('../input/mayo-clinic-strip-ai/train/0cc0bc_0.tif')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.style.use(\"dark_background\")\nadaptive_threshold_display('../input/mayo-clinic-strip-ai/train/0cc0bc_0.tif')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"figure = plt.figure(figsize=(8,8))\nCanny_Image = canny_reading('../input/mayo-clinic-strip-ai/train/0cc0bc_0.tif')\nplt.imshow(Canny_Image)\nplt.axis(\"off\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bitwise_and_display('../input/mayo-clinic-strip-ai/train/0cc0bc_0.tif')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Analyzing CE class**","metadata":{}},{"cell_type":"code","source":"Reading_Image=reading_tiff('../input/mayo-clinic-strip-ai/train/31adaa_0.tif')\nfigure = plt.figure(figsize=(20,20))\nplt.imshow(Reading_Image)\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"different_type_display(reading_tiff('../input/mayo-clinic-strip-ai/train/31adaa_0.tif'),\n                      threshold_reading('../input/mayo-clinic-strip-ai/train/31adaa_0.tif'),\n                      adaptive_threshold_reading('../input/mayo-clinic-strip-ai/train/31adaa_0.tif'),\n                      canny_reading('../input/mayo-clinic-strip-ai/train/31adaa_0.tif'))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.style.use(\"dark_background\")\nsingle_display_tiff('../input/mayo-clinic-strip-ai/train/1b86c5_0.tif')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"different_type_display(reading_tiff('../input/mayo-clinic-strip-ai/train/1b86c5_0.tif'),\n                      threshold_reading('../input/mayo-clinic-strip-ai/train/1b86c5_0.tif'),\n                      adaptive_threshold_reading('../input/mayo-clinic-strip-ai/train/1b86c5_0.tif'),\n                      canny_reading('../input/mayo-clinic-strip-ai/train/1b86c5_0.tif'))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Analyzing LAA class**","metadata":{}},{"cell_type":"code","source":"plt.style.use(\"dark_background\")\nsingle_display_tiff('../input/mayo-clinic-strip-ai/train/6f6e0c_0.tif')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"different_type_display(reading_tiff('../input/mayo-clinic-strip-ai/train/6f6e0c_0.tif'),\n                      threshold_reading('../input/mayo-clinic-strip-ai/train/6f6e0c_0.tif'),\n                      adaptive_threshold_reading('../input/mayo-clinic-strip-ai/train/6f6e0c_0.tif'),\n                      canny_reading('../input/mayo-clinic-strip-ai/train/6f6e0c_0.tif'))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.style.use(\"dark_background\")\nsingle_display_tiff('../input/mayo-clinic-strip-ai/train/ed5006_0.tif')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.style.use(\"dark_background\")\nsingle_display_tiff('../input/mayo-clinic-strip-ai/train/a2c497_0.tif')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from PIL import Image\nImage.MAX_IMAGE_PIXELS = None \n\nCE_imgs = image_data.loc[image_data['label']=='CE','path']\nLAA_imgs = image_data.loc[image_data['label']=='LAA','path']\n\n\nplt.style.use('default')\nfig, axes = plt.subplots(1,5, figsize=(16,16))\ntrain_images\nfor ax in axes.reshape(-1):\n    img_path = np.random.choice(CE_imgs)\n    img = Image.open(img_path)   \n    img.thumbnail((300,300), Image.Resampling.LANCZOS)\n    ax.imshow(img), ax.set_title(\"target: CE\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-18T10:36:36.716525Z","iopub.execute_input":"2023-02-18T10:36:36.71766Z","iopub.status.idle":"2023-02-18T10:37:23.665254Z","shell.execute_reply.started":"2023-02-18T10:36:36.717614Z","shell.execute_reply":"2023-02-18T10:37:23.664374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(1,5, figsize=(16,16))\ntrain_images\nfor ax in axes.reshape(-1):\n    img_path = np.random.choice(LAA_imgs)\n    img = Image.open(img_path)   \n    img.thumbnail((300,300), Image.Resampling.LANCZOS)\n    ax.imshow(img), ax.set_title(\"target: LAA\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-18T10:37:23.666608Z","iopub.execute_input":"2023-02-18T10:37:23.667292Z","iopub.status.idle":"2023-02-18T10:38:36.785701Z","shell.execute_reply.started":"2023-02-18T10:37:23.667241Z","shell.execute_reply":"2023-02-18T10:38:36.784714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"slide = OpenSlide('/kaggle/input/mayo-clinic-strip-ai/train/026c97_0.tif') # opening a full slide\n\nregion = (2500, 2000) # location of the top left pixel\nlevel = 0 # level of the picture (we have only 0)\nsize = (3500, 3500) # region size in pixels\n\nregion = slide.read_region(region, level, size)\nimage = region.resize((512, 512))\nplt.figure(figsize=(10, 10))\nplt.imshow(image)\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-02-18T10:38:36.787992Z","iopub.execute_input":"2023-02-18T10:38:36.788762Z","iopub.status.idle":"2023-02-18T10:38:39.696291Z","shell.execute_reply.started":"2023-02-18T10:38:36.788724Z","shell.execute_reply":"2023-02-18T10:38:39.691852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Solution 1:**\nBalancing the weights","metadata":{}},{"cell_type":"code","source":"train_df[\"file_path\"] = train_df[\"image_id\"].apply(lambda x: \"../input/mayo-clinic-strip-ai/train/\" + x + \".tif\")\ntest_df[\"file_path\"]  = test_df[\"image_id\"].apply(lambda x: \"../input/mayo-clinic-strip-ai/test/\" + x + \".tif\")","metadata":{"papermill":{"duration":0.028968,"end_time":"2022-12-13T12:03:24.495672","exception":false,"start_time":"2022-12-13T12:03:24.466704","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-19T11:14:02.583343Z","iopub.execute_input":"2023-02-19T11:14:02.583718Z","iopub.status.idle":"2023-02-19T11:14:02.593343Z","shell.execute_reply.started":"2023-02-19T11:14:02.58368Z","shell.execute_reply":"2023-02-19T11:14:02.59058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[\"target\"] = train_df[\"label\"].apply(lambda x : 1 if x==\"CE\" else 0)\ntrain_df.head()","metadata":{"papermill":{"duration":0.017639,"end_time":"2022-12-13T12:03:24.519397","exception":false,"start_time":"2022-12-13T12:03:24.501758","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-19T11:14:02.595708Z","iopub.execute_input":"2023-02-19T11:14:02.596201Z","iopub.status.idle":"2023-02-19T11:14:02.613139Z","shell.execute_reply.started":"2023-02-19T11:14:02.596057Z","shell.execute_reply":"2023-02-19T11:14:02.61196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Preprocessing","metadata":{}},{"cell_type":"code","source":"%%time\ndef preprocess(image_path):\n    slide=OpenSlide(image_path)\n    region= (1000,1000)    \n    size  = (5000, 5000)\n    image = slide.read_region(region, 0, size)\n    image = image.resize((512, 512))\n    image = np.array(image)    \n    return image\n\nX_train=[]\nfor i in tqdm(train_df['file_path']):\n    x1=preprocess(i)\n    X_train.append(x1)\n\nY_train=[]    \nY_train=train_df['target']","metadata":{"papermill":{"duration":2496.090688,"end_time":"2022-12-13T12:45:00.650948","exception":false,"start_time":"2022-12-13T12:03:24.56026","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-19T11:14:06.441228Z","iopub.execute_input":"2023-02-19T11:14:06.441587Z","iopub.status.idle":"2023-02-19T12:01:20.275852Z","shell.execute_reply.started":"2023-02-19T11:14:06.441557Z","shell.execute_reply":"2023-02-19T12:01:20.274836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train=np.array(X_train)\nX_train=X_train/255.0\nY_train = np.array(Y_train)","metadata":{"papermill":{"duration":0.296197,"end_time":"2022-12-13T12:45:01.078179","exception":false,"start_time":"2022-12-13T12:45:00.781982","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-19T12:03:00.269245Z","iopub.execute_input":"2023-02-19T12:03:00.269792Z","iopub.status.idle":"2023-02-19T12:03:09.809055Z","shell.execute_reply.started":"2023-02-19T12:03:00.269748Z","shell.execute_reply":"2023-02-19T12:03:09.807949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Splitting data\nx_train,x_test,y_train,y_test=train_test_split(X_train,Y_train, test_size=0.2)","metadata":{"papermill":{"duration":0.049801,"end_time":"2022-12-13T12:45:01.170206","exception":false,"start_time":"2022-12-13T12:45:01.120405","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-19T12:06:18.423308Z","iopub.execute_input":"2023-02-19T12:06:18.423681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(x_train.shape)","metadata":{"papermill":{"duration":0.050654,"end_time":"2022-12-13T12:45:01.261913","exception":false,"start_time":"2022-12-13T12:45:01.211259","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-19T12:04:57.75059Z","iopub.execute_input":"2023-02-19T12:04:57.751012Z","iopub.status.idle":"2023-02-19T12:04:57.757607Z","shell.execute_reply.started":"2023-02-19T12:04:57.750978Z","shell.execute_reply":"2023-02-19T12:04:57.756352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(x_train[0])","metadata":{"execution":{"iopub.status.busy":"2023-02-19T12:05:07.562681Z","iopub.execute_input":"2023-02-19T12:05:07.563058Z","iopub.status.idle":"2023-02-19T12:05:07.851863Z","shell.execute_reply.started":"2023-02-19T12:05:07.563028Z","shell.execute_reply":"2023-02-19T12:05:07.850575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## CNN Model","metadata":{"papermill":{"duration":0.040808,"end_time":"2022-12-13T12:45:01.435232","exception":false,"start_time":"2022-12-13T12:45:01.394424","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def precision(y_true, y_pred):\n        true_positives = K.sum(K.round(K.clip(y_true * y_pred, 0, 1)))\n        predicted_positives = K.sum(K.round(K.clip(y_pred, 0, 1)))\n        precision = true_positives / (predicted_positives + K.epsilon())\n        return precision","metadata":{"execution":{"iopub.status.busy":"2023-02-19T12:05:13.013158Z","iopub.execute_input":"2023-02-19T12:05:13.013528Z","iopub.status.idle":"2023-02-19T12:05:13.019484Z","shell.execute_reply.started":"2023-02-19T12:05:13.0135Z","shell.execute_reply":"2023-02-19T12:05:13.018255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def f1_score(y_true, y_pred): \n    true_positives = K.sum(K.round(K.clip(y_true * y_pred, 0, 1)))\n    possible_positives = K.sum(K.round(K.clip(y_true, 0, 1)))\n    predicted_positives = K.sum(K.round(K.clip(y_pred, 0, 1)))\n    precision = true_positives / (predicted_positives + K.epsilon())\n    recall = true_positives / (possible_positives + K.epsilon())\n    f1_val = 2*(precision*recall)/(precision+recall+K.epsilon())\n    return f1_val","metadata":{"execution":{"iopub.status.busy":"2023-02-19T12:05:15.095909Z","iopub.execute_input":"2023-02-19T12:05:15.096265Z","iopub.status.idle":"2023-02-19T12:05:15.103195Z","shell.execute_reply.started":"2023-02-19T12:05:15.096236Z","shell.execute_reply":"2023-02-19T12:05:15.101812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Sequential()\ninput_shape = (512, 512, 4)\nmodel.add(Conv2D(filters=32, kernel_size = (3,3), strides =2, padding = 'same', activation = 'relu', input_shape = input_shape))\nmodel.add(Conv2D(filters=64, kernel_size = (3,3), strides =2, padding = 'same', activation = 'relu'))\nmodel.add(Conv2D(filters=32, kernel_size = (3,3), strides =2, padding = 'same', activation = 'relu'))\nmodel.add(Flatten())\nmodel.add(Dense(128, activation = 'relu',kernel_regularizer=tf.keras.regularizers.l2(0.1)))\nmodel.add(Dropout(0.50))\nmodel.add(Dense(1))\nmodel.compile(\n    loss = tf.keras.losses.MeanSquaredError(),    \n    metrics=[tf.keras.metrics.RootMeanSquaredError(name=\"rmse\"), \n             tf.keras.metrics.BinaryAccuracy(name=\"accuracy\"),f1_score,precision],\n    optimizer = tf.keras.optimizers.Adam(1e-4))","metadata":{"papermill":{"duration":0.213269,"end_time":"2022-12-13T12:45:01.68956","exception":false,"start_time":"2022-12-13T12:45:01.476291","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-19T12:05:18.470383Z","iopub.execute_input":"2023-02-19T12:05:18.47074Z","iopub.status.idle":"2023-02-19T12:05:21.410832Z","shell.execute_reply.started":"2023-02-19T12:05:18.470704Z","shell.execute_reply":"2023-02-19T12:05:21.409906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dot_img_file = 'model.png'\ntf.keras.utils.plot_model(model, to_file=dot_img_file, show_shapes=True)","metadata":{"execution":{"iopub.status.busy":"2023-02-19T12:05:26.840385Z","iopub.execute_input":"2023-02-19T12:05:26.840744Z","iopub.status.idle":"2023-02-19T12:05:28.18502Z","shell.execute_reply.started":"2023-02-19T12:05:26.840707Z","shell.execute_reply":"2023-02-19T12:05:28.183812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#This is where we balance class weights\nfrom sklearn.utils import compute_class_weight\ntrain_classes = Y_train\nclass_weights = compute_class_weight(\n                                        class_weight = \"balanced\",\n                                        classes = np.unique(train_classes),\n                                        y = train_classes                                                    \n                                    )\nclass_weights = dict(zip(np.unique(train_classes), class_weights))\nclass_weights","metadata":{"papermill":{"duration":0.054636,"end_time":"2022-12-13T12:45:01.787016","exception":false,"start_time":"2022-12-13T12:45:01.73238","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-19T12:05:31.299383Z","iopub.execute_input":"2023-02-19T12:05:31.299791Z","iopub.status.idle":"2023-02-19T12:05:31.31176Z","shell.execute_reply.started":"2023-02-19T12:05:31.299755Z","shell.execute_reply":"2023-02-19T12:05:31.310382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training the model","metadata":{}},{"cell_type":"code","source":"callback = tf.keras.callbacks.ModelCheckpoint(\n    filepath='our_cnn_best.h5',\n    monitor='val_binary_accuracy',\n    mode='max',\n    save_best_only=True, verbose=1)","metadata":{"execution":{"iopub.status.busy":"2023-02-19T12:05:34.709795Z","iopub.execute_input":"2023-02-19T12:05:34.710208Z","iopub.status.idle":"2023-02-19T12:05:34.716024Z","shell.execute_reply.started":"2023-02-19T12:05:34.710177Z","shell.execute_reply":"2023-02-19T12:05:34.714235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(\n    X_train,\n    y_train,\n    epochs = 1000,\n    batch_size=32,\n    shuffle=True,\n    validation_data = (x_test,y_test),\n    class_weight= class_weights,\n    callbacks = callback\n)","metadata":{"papermill":{"duration":13.303133,"end_time":"2022-12-13T12:45:15.131496","exception":false,"start_time":"2022-12-13T12:45:01.828363","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-16T15:56:30.736452Z","iopub.execute_input":"2023-02-16T15:56:30.736845Z","iopub.status.idle":"2023-02-16T15:57:12.899002Z","shell.execute_reply.started":"2023-02-16T15:56:30.736807Z","shell.execute_reply":"2023-02-16T15:57:12.897821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_cnn = load_model('/kaggle/working/our_cnn_best.h5', custom_objects={\"f1_score\": f1_score })\np=model.evaluate(x_test,y_test)\np","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:57:12.900812Z","iopub.execute_input":"2023-02-16T15:57:12.901134Z","iopub.status.idle":"2023-02-16T15:57:13.295649Z","shell.execute_reply.started":"2023-02-16T15:57:12.901107Z","shell.execute_reply":"2023-02-16T15:57:13.294685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\nimport seaborn as sns\n\nplt.figure(figsize=(15, 5))\n\npreds = best_cnn.predict(x_test)\npreds = (preds >= 0.5).astype(np.int32)\n\ncm = confusion_matrix(y_test, preds)\ndf_cm = pd.DataFrame(cm, index=['LAA', 'CE'], columns=['LAA', 'CE'])\nplt.subplot(121)\nplt.title(\"Confusion matrix for our model\\n\")\nsns.heatmap(df_cm, annot=True, fmt=\"d\", cmap=\"YlGnBu\")\nplt.ylabel(\"Predicted\")\nplt.xlabel(\"Actual\")","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:49:59.79141Z","iopub.execute_input":"2023-02-16T15:49:59.791793Z","iopub.status.idle":"2023-02-16T15:50:00.303367Z","shell.execute_reply.started":"2023-02-16T15:49:59.791759Z","shell.execute_reply":"2023-02-16T15:50:00.302376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Submission for kaggle competition","metadata":{}},{"cell_type":"code","source":"test1=[]\nfor i in test_df['file_path']:\n    x1=preprocess(i)\n    test1.append(x1)\n    print(i)\n    \ntest1=np.array(test1)\n","metadata":{"papermill":{"duration":18.955522,"end_time":"2022-12-13T12:45:34.133527","exception":false,"start_time":"2022-12-13T12:45:15.178005","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-01-03T22:06:04.538989Z","iopub.execute_input":"2023-01-03T22:06:04.539374Z","iopub.status.idle":"2023-01-03T22:06:13.792079Z","shell.execute_reply.started":"2023-01-03T22:06:04.539342Z","shell.execute_reply":"2023-01-03T22:06:13.790979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cnn_pred=model.predict(test1)\ncnn_pred","metadata":{"papermill":{"duration":0.212752,"end_time":"2022-12-13T12:45:34.392434","exception":false,"start_time":"2022-12-13T12:45:34.179682","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-01-03T22:06:13.795322Z","iopub.execute_input":"2023-01-03T22:06:13.796232Z","iopub.status.idle":"2023-01-03T22:06:13.927119Z","shell.execute_reply.started":"2023-01-03T22:06:13.796192Z","shell.execute_reply":"2023-01-03T22:06:13.926141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.DataFrame(test_df[\"patient_id\"].copy())\nsub[\"CE\"] = cnn_pred\nsub[\"LAA\"] = 1- sub[\"CE\"]\n\nsub = sub.groupby(\"patient_id\").mean()\nsub = sub[[\"CE\", \"LAA\"]].round(6).reset_index()\nsub","metadata":{"papermill":{"duration":0.083787,"end_time":"2022-12-13T12:45:34.524093","exception":false,"start_time":"2022-12-13T12:45:34.440306","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-01-03T22:06:17.875828Z","iopub.execute_input":"2023-01-03T22:06:17.876219Z","iopub.status.idle":"2023-01-03T22:06:17.897902Z","shell.execute_reply.started":"2023-01-03T22:06:17.87616Z","shell.execute_reply":"2023-01-03T22:06:17.8967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv(\"submission.csv\", index = False)\n!head submission.csv","metadata":{"papermill":{"duration":1.063784,"end_time":"2022-12-13T12:45:35.635044","exception":false,"start_time":"2022-12-13T12:45:34.57126","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-01-03T22:06:22.041214Z","iopub.execute_input":"2023-01-03T22:06:22.04237Z","iopub.status.idle":"2023-01-03T22:06:23.051516Z","shell.execute_reply.started":"2023-01-03T22:06:22.042319Z","shell.execute_reply":"2023-01-03T22:06:23.050317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Solution 2\nUsing DenseNET","metadata":{}},{"cell_type":"code","source":"del X_train\ndel Y_train","metadata":{"execution":{"iopub.status.busy":"2023-01-04T13:14:44.874156Z","iopub.execute_input":"2023-01-04T13:14:44.874534Z","iopub.status.idle":"2023-01-04T13:14:44.908974Z","shell.execute_reply.started":"2023-01-04T13:14:44.874501Z","shell.execute_reply":"2023-01-04T13:14:44.906821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-04T13:14:58.771159Z","iopub.execute_input":"2023-01-04T13:14:58.771748Z","iopub.status.idle":"2023-01-04T13:14:58.790057Z","shell.execute_reply.started":"2023-01-04T13:14:58.771698Z","shell.execute_reply":"2023-01-04T13:14:58.788927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndef preprocess_RGB(image_path):\n    slide=OpenSlide(image_path)\n    region= (2500,2500)    \n    size  = (5000, 5000)\n    image = slide.read_region(region, 0, size).convert('RGB')\n    image = image.resize((128, 128))\n    image = np.array(image)    \n    return image\n\nX_train=[]\nfor i in tqdm(train_df['file_path']):\n    #print(i)\n    x1=preprocess_RGB(i)\n    X_train.append(x1)\n    \nY_train=train_df['target']","metadata":{"execution":{"iopub.status.busy":"2023-01-05T15:40:01.255637Z","iopub.execute_input":"2023-01-05T15:40:01.256013Z","iopub.status.idle":"2023-01-05T15:45:55.548723Z","shell.execute_reply.started":"2023-01-05T15:40:01.255981Z","shell.execute_reply":"2023-01-05T15:45:55.54774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train=np.array(X_train)\nX_train=X_train/255.0\nY_train = np.array(Y_train)","metadata":{"execution":{"iopub.status.busy":"2023-01-05T15:46:08.860437Z","iopub.execute_input":"2023-01-05T15:46:08.861144Z","iopub.status.idle":"2023-01-05T15:46:08.882872Z","shell.execute_reply.started":"2023-01-05T15:46:08.861107Z","shell.execute_reply":"2023-01-05T15:46:08.881891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train,x_test,y_train,y_test=train_test_split(X_train,Y_train, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-01-05T15:46:51.536634Z","iopub.execute_input":"2023-01-05T15:46:51.537027Z","iopub.status.idle":"2023-01-05T15:46:51.551915Z","shell.execute_reply.started":"2023-01-05T15:46:51.536993Z","shell.execute_reply":"2023-01-05T15:46:51.550669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CE = 0\nLAA = 0\nfor i in y_test:\n    if i ==1:\n        CE = CE+1\n    else:\n        LAA = LAA + 1\nprint(CE)\nprint(LAA)","metadata":{"execution":{"iopub.status.busy":"2023-01-05T15:46:51.906429Z","iopub.execute_input":"2023-01-05T15:46:51.906789Z","iopub.status.idle":"2023-01-05T15:46:51.913629Z","shell.execute_reply.started":"2023-01-05T15:46:51.906757Z","shell.execute_reply":"2023-01-05T15:46:51.912524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(x_test[0])","metadata":{"execution":{"iopub.status.busy":"2023-01-05T15:46:54.879208Z","iopub.execute_input":"2023-01-05T15:46:54.879878Z","iopub.status.idle":"2023-01-05T15:46:55.099099Z","shell.execute_reply.started":"2023-01-05T15:46:54.879841Z","shell.execute_reply":"2023-01-05T15:46:55.098093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def f1_score(y_true, y_pred): \n    true_positives = K.sum(K.round(K.clip(y_true * y_pred, 0, 1)))\n    possible_positives = K.sum(K.round(K.clip(y_true, 0, 1)))\n    predicted_positives = K.sum(K.round(K.clip(y_pred, 0, 1)))\n    precision = true_positives / (predicted_positives + K.epsilon())\n    recall = true_positives / (possible_positives + K.epsilon())\n    f1_val = 2*(precision*recall)/(precision+recall+K.epsilon())\n    return f1_val","metadata":{"execution":{"iopub.status.busy":"2023-01-05T15:47:02.34076Z","iopub.execute_input":"2023-01-05T15:47:02.341122Z","iopub.status.idle":"2023-01-05T15:47:02.348417Z","shell.execute_reply.started":"2023-01-05T15:47:02.34109Z","shell.execute_reply":"2023-01-05T15:47:02.347077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.utils import compute_class_weight\ntrain_classes = Y_train\nclass_weights = compute_class_weight(\n                                        class_weight = \"balanced\",\n                                        classes = np.unique(train_classes),\n                                        y = train_classes                                                    \n                                    )\nclass_weights = dict(zip(np.unique(train_classes), class_weights))\nclass_weights","metadata":{"execution":{"iopub.status.busy":"2023-01-05T15:47:03.024233Z","iopub.execute_input":"2023-01-05T15:47:03.025309Z","iopub.status.idle":"2023-01-05T15:47:03.036134Z","shell.execute_reply.started":"2023-01-05T15:47:03.025258Z","shell.execute_reply":"2023-01-05T15:47:03.034944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.optimizers import Adam\nfrom keras.models import Sequential\nfrom tensorflow.keras.applications import DenseNet121\nfrom keras import layers\nfrom keras import metrics\n\neffnet = DenseNet121(\n        weights='imagenet',\n        include_top=False,\n        input_shape=(128,128,3)\n)\nprint(len(effnet.layers))\nfor layer in effnet.layers[0:200]:\n    layer.trainable = True\nfor layer in effnet.layers[201::]:\n    layer.trainable = False\nDense_model = Sequential()\nDense_model.add(effnet)\nDense_model.add(layers.GlobalAveragePooling2D())\nDense_model.add(layers.Dropout(0.5))\nDense_model.add(layers.Dense(512,activation='relu'))\nDense_model.add(layers.Dense(64,activation='relu'))\nDense_model.add(layers.Dense(1, activation='sigmoid'))\nDense_model.compile(\n    loss = tf.keras.losses.BinaryCrossentropy(),\n    metrics=[metrics.binary_accuracy,f1_score],\n    optimizer = tf.keras.optimizers.Adam(1e-3))\nDense_model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-01-05T15:47:04.923955Z","iopub.execute_input":"2023-01-05T15:47:04.924333Z","iopub.status.idle":"2023-01-05T15:47:08.39497Z","shell.execute_reply.started":"2023-01-05T15:47:04.924301Z","shell.execute_reply":"2023-01-05T15:47:08.393898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dot_img_file = 'model.png'\ntf.keras.utils.plot_model(Dense_model, to_file=dot_img_file, show_shapes=True)","metadata":{"execution":{"iopub.status.busy":"2023-01-05T15:47:08.39708Z","iopub.execute_input":"2023-01-05T15:47:08.397449Z","iopub.status.idle":"2023-01-05T15:47:08.591317Z","shell.execute_reply.started":"2023-01-05T15:47:08.397412Z","shell.execute_reply":"2023-01-05T15:47:08.590158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"callback2 = tf.keras.callbacks.ModelCheckpoint(\n    filepath='Dense_best.h5',\n    monitor='val_binary_accuracy',\n    mode='max',\n    save_best_only=True, verbose=1)","metadata":{"execution":{"iopub.status.busy":"2023-01-05T15:47:08.593068Z","iopub.execute_input":"2023-01-05T15:47:08.594167Z","iopub.status.idle":"2023-01-05T15:47:08.600048Z","shell.execute_reply.started":"2023-01-05T15:47:08.594123Z","shell.execute_reply":"2023-01-05T15:47:08.598866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Dense_model.fit(\n    x_train,\n    y_train,\n    epochs = 10,\n    batch_size=20,\n    validation_data = (x_test,y_test),\n    class_weight= class_weights,\n    callbacks = callback2   \n)","metadata":{"execution":{"iopub.status.busy":"2023-01-05T15:47:08.60239Z","iopub.execute_input":"2023-01-05T15:47:08.602934Z","iopub.status.idle":"2023-01-05T15:47:27.015223Z","shell.execute_reply.started":"2023-01-05T15:47:08.602783Z","shell.execute_reply":"2023-01-05T15:47:27.01429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_Dense = load_model('/kaggle/working/Dense_best.h5', custom_objects={\"f1_score\": f1_score })\nbest_Dense.evaluate(x_test,y_test)","metadata":{"execution":{"iopub.status.busy":"2023-01-05T15:47:29.935517Z","iopub.execute_input":"2023-01-05T15:47:29.935883Z","iopub.status.idle":"2023-01-05T15:47:35.638819Z","shell.execute_reply.started":"2023-01-05T15:47:29.935853Z","shell.execute_reply":"2023-01-05T15:47:35.637696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\nimport seaborn as sns\nplt.figure(figsize=(15, 5))\npreds = best_Dense.predict(x_test)\npreds = (preds >= 0.5).astype(np.int32)\ncm = confusion_matrix(y_test, preds)\ndf_cm = pd.DataFrame(cm, index=['LAA', 'CE'], columns=['LAA', 'CE'])\nplt.subplot(121)\nplt.title(\"Confusion matrix for our model\\n\")\nsns.heatmap(df_cm, annot=True, fmt=\"d\", cmap=\"YlGnBu\")\nplt.ylabel(\"Predicted\")\nplt.xlabel(\"Actual\")","metadata":{"execution":{"iopub.status.busy":"2023-01-05T15:47:42.430539Z","iopub.execute_input":"2023-01-05T15:47:42.4309Z","iopub.status.idle":"2023-01-05T15:47:44.944063Z","shell.execute_reply.started":"2023-01-05T15:47:42.43087Z","shell.execute_reply":"2023-01-05T15:47:44.943096Z"},"trusted":true},"execution_count":null,"outputs":[]}]}