{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport matplotlib.pyplot as plt\nimport glob\nimport cv2\nimport tensorflow as tf","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-13T13:34:34.041492Z","iopub.execute_input":"2023-01-13T13:34:34.042298Z","iopub.status.idle":"2023-01-13T13:34:39.459313Z","shell.execute_reply.started":"2023-01-13T13:34:34.042185Z","shell.execute_reply":"2023-01-13T13:34:39.458136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This notebook aims to optimise the training of the RSNA breast cancer dataset. The dataset is converted to a balanced one in terms of cancer/no cancer samples by upsampling the cancer class. So far the following has been explored:\n\n- Using a simple 2D CNN\n- Using pre-trained architectures: DenseNet, VGG16\n- No special image processing \n- Contrast enhancement\n- Median blur\n\nSo far the models have failed to learn anything. I still aim to explore more image transformation techniques, as well as trying different image sizes. Currently I'm using the pre-made (512,512) pixel dataset from 'rsna-mammography-images-as-pngs'. I had been using the (1024,1024) pixel dataset, however the models take a very long time to train. \n\nPerhaps the most important thing to try is to extract a region of interest, as there is a lot of blank space in the images.","metadata":{}},{"cell_type":"code","source":"path='/kaggle/input/rsna-breast-cancer-detection/'\ntrain_data=pd.read_csv(path+'train.csv')\ntest_data=pd.read_csv(path+'test.csv')","metadata":{"execution":{"iopub.status.busy":"2023-01-13T13:34:39.465954Z","iopub.execute_input":"2023-01-13T13:34:39.466991Z","iopub.status.idle":"2023-01-13T13:34:39.581942Z","shell.execute_reply.started":"2023-01-13T13:34:39.466952Z","shell.execute_reply":"2023-01-13T13:34:39.580929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.hist(train_data['cancer'])","metadata":{"execution":{"iopub.status.busy":"2023-01-13T13:34:39.587161Z","iopub.execute_input":"2023-01-13T13:34:39.587627Z","iopub.status.idle":"2023-01-13T13:34:39.916014Z","shell.execute_reply.started":"2023-01-13T13:34:39.587583Z","shell.execute_reply":"2023-01-13T13:34:39.915053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Will need to upsample the cancer images to balance out the classes","metadata":{}},{"cell_type":"markdown","source":"Get paths to train images","metadata":{}},{"cell_type":"code","source":"train_images = sorted(glob.glob(\"../input/rsna-mammography-images-as-pngs/images_as_pngs_512/train_images_processed_512/*/*\"))\n","metadata":{"execution":{"iopub.status.busy":"2023-01-13T13:34:51.263569Z","iopub.execute_input":"2023-01-13T13:34:51.263957Z","iopub.status.idle":"2023-01-13T13:36:00.937495Z","shell.execute_reply.started":"2023-01-13T13:34:51.263923Z","shell.execute_reply":"2023-01-13T13:36:00.936484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_images[0:5]","metadata":{"execution":{"iopub.status.busy":"2023-01-13T13:39:43.687061Z","iopub.execute_input":"2023-01-13T13:39:43.687662Z","iopub.status.idle":"2023-01-13T13:39:43.69487Z","shell.execute_reply.started":"2023-01-13T13:39:43.687627Z","shell.execute_reply":"2023-01-13T13:39:43.693628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data['path']=train_images","metadata":{"execution":{"iopub.status.busy":"2023-01-13T13:39:46.098107Z","iopub.execute_input":"2023-01-13T13:39:46.099111Z","iopub.status.idle":"2023-01-13T13:39:46.114606Z","shell.execute_reply.started":"2023-01-13T13:39:46.09907Z","shell.execute_reply":"2023-01-13T13:39:46.113564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data_cancer_1=train_data[train_data['cancer']==1]\ntrain_data_cancer_0=train_data[train_data['cancer']==0]","metadata":{"execution":{"iopub.status.busy":"2023-01-13T13:39:48.306798Z","iopub.execute_input":"2023-01-13T13:39:48.307276Z","iopub.status.idle":"2023-01-13T13:39:48.331607Z","shell.execute_reply.started":"2023-01-13T13:39:48.307231Z","shell.execute_reply":"2023-01-13T13:39:48.330646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data_cancer_0.shape[0]/train_data_cancer_1.shape[0]","metadata":{"execution":{"iopub.status.busy":"2023-01-13T13:39:50.688527Z","iopub.execute_input":"2023-01-13T13:39:50.688889Z","iopub.status.idle":"2023-01-13T13:39:50.696288Z","shell.execute_reply.started":"2023-01-13T13:39:50.688857Z","shell.execute_reply":"2023-01-13T13:39:50.695042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Will need to create around 46 copies of the cancer=1 dataset in order to achieve class balance","metadata":{}},{"cell_type":"code","source":"from sklearn.utils import resample\ncancer_upsample = resample(train_data_cancer_1, replace=True, n_samples=train_data_cancer_0.shape[0],\nrandom_state=42)\n\nprint(cancer_upsample.shape)","metadata":{"execution":{"iopub.status.busy":"2023-01-13T13:39:53.001965Z","iopub.execute_input":"2023-01-13T13:39:53.003161Z","iopub.status.idle":"2023-01-13T13:39:53.256905Z","shell.execute_reply.started":"2023-01-13T13:39:53.003117Z","shell.execute_reply":"2023-01-13T13:39:53.255689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cancer_upsample.columns","metadata":{"execution":{"iopub.status.busy":"2023-01-13T13:39:56.191671Z","iopub.execute_input":"2023-01-13T13:39:56.192146Z","iopub.status.idle":"2023-01-13T13:39:56.205262Z","shell.execute_reply.started":"2023-01-13T13:39:56.19211Z","shell.execute_reply":"2023-01-13T13:39:56.204289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_all = pd.concat([train_data_cancer_0, cancer_upsample])\n\nprint(data_all.head())\n\nprint(data_all[\"cancer\"].value_counts())\n\ndata_all.groupby('cancer').size().plot(kind='pie', y = \"v1\",label = \"Type\",autopct='%1.1f%%')\n","metadata":{"execution":{"iopub.status.busy":"2023-01-13T13:39:58.282322Z","iopub.execute_input":"2023-01-13T13:39:58.282693Z","iopub.status.idle":"2023-01-13T13:39:58.441268Z","shell.execute_reply.started":"2023-01-13T13:39:58.282663Z","shell.execute_reply":"2023-01-13T13:39:58.439864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Make a function to load and process the images","metadata":{}},{"cell_type":"code","source":"def load_and_preprocess_image(path,contrast_en=False,median_filter=False):\n    # load image\n    img = cv2.imread(path, cv2.IMREAD_COLOR)\n    # normalise\n    # img = img/255\n    # Convert float to 8-bit type\n    if(img.dtype!='uint8'):\n        img = np.array(img,dtype=np.uint8)\n    if (contrast_en=='True'):\n        # Convert the image to grayscale\n        print('Enhancing contrast')\n        gray_img = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n        # For contrast enhancement, equalize the histogram\n        img_enhanced = cv2.equalizeHist(gray_img)\n        return img_enhanced\n    if (median_filter==\"True\"):\n        print('Applying median blur')\n        blur_img=cv2.medianBlur(img)\n        return blur_img\n    ## To do: Try using transformation techniques e.g. Fourier, wavelet and Radon Transform\n    else:\n        return img","metadata":{"execution":{"iopub.status.busy":"2023-01-13T13:40:03.598672Z","iopub.execute_input":"2023-01-13T13:40:03.599413Z","iopub.status.idle":"2023-01-13T13:40:03.606783Z","shell.execute_reply.started":"2023-01-13T13:40:03.599376Z","shell.execute_reply":"2023-01-13T13:40:03.605727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Function to visualise the effect of pre-processing","metadata":{}},{"cell_type":"code","source":"def vis_preprocess_effect(data,contrast_en=False,median_filter=False):\n    plt.ion()\n    for i in range(0,5):\n        print(i)\n        plt.subplot(121)\n        plt.title('No preprocess')\n        plt.imshow(load_and_preprocess_image(data_all['path'][i]))\n        plt.subplot(122)\n        if (contrast_en==True):\n            plt.title('With contrast enhancement')\n            plt.imshow(load_and_preprocess_image(data_all['path'][i],contrast_en=True))\n            plt.pause(2)\n            plt.close()\n        if (median_filter==True):\n            plt.title('With median filter')\n            plt.imshow(load_and_preprocess_image(data_all['path'][i],median_filter=True))\n            plt.pause(2)\n            plt.close()","metadata":{"execution":{"iopub.status.busy":"2023-01-13T13:40:07.072744Z","iopub.execute_input":"2023-01-13T13:40:07.073293Z","iopub.status.idle":"2023-01-13T13:40:07.084065Z","shell.execute_reply.started":"2023-01-13T13:40:07.073249Z","shell.execute_reply":"2023-01-13T13:40:07.082817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vis_preprocess_effect(data_all,median_filter=True)","metadata":{"execution":{"iopub.status.busy":"2023-01-13T13:40:14.799971Z","iopub.execute_input":"2023-01-13T13:40:14.800366Z","iopub.status.idle":"2023-01-13T13:40:26.981319Z","shell.execute_reply.started":"2023-01-13T13:40:14.800336Z","shell.execute_reply":"2023-01-13T13:40:26.980291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Not much visual difference applying the median filter, but can try anyway","metadata":{}},{"cell_type":"code","source":"vis_preprocess_effect(data_all,contrast_en=True)","metadata":{"execution":{"iopub.status.busy":"2023-01-13T13:40:28.018466Z","iopub.execute_input":"2023-01-13T13:40:28.018848Z","iopub.status.idle":"2023-01-13T13:40:40.027741Z","shell.execute_reply.started":"2023-01-13T13:40:28.018802Z","shell.execute_reply":"2023-01-13T13:40:40.02434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Not much visible difference using contrast enhancement, but can try it anyway","metadata":{}},{"cell_type":"markdown","source":"Make a generator function to get the images and labels","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.utils import to_categorical\n\ndef data_generator(df, batch_size=32,contrast_en=False,median_filter=False):\n  while True:\n    # Shuffle the dataframe rows\n    df = df.sample(frac=1).reset_index(drop=True)\n    \n    # Divide the dataframe into batches\n    for i in range(0, df.shape[0], batch_size):\n      batch_df = df.iloc[i:i+batch_size, :]\n      \n      # Load and preprocess the images\n      images = []\n      for image_path in batch_df['path']:\n        image = load_and_preprocess_image(image_path,contrast_en,median_filter)\n        # Normalise the pre-processed image\n        images.append(image/255)\n      images = np.array(images)\n      \n      # Convert the labels to categorical format\n      labels = to_categorical(batch_df['cancer'].values, num_classes=2)\n      \n      yield images, labels\n","metadata":{"execution":{"iopub.status.busy":"2023-01-13T13:40:42.022922Z","iopub.execute_input":"2023-01-13T13:40:42.024063Z","iopub.status.idle":"2023-01-13T13:40:42.850965Z","shell.execute_reply.started":"2023-01-13T13:40:42.023991Z","shell.execute_reply":"2023-01-13T13:40:42.850016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Divide the data into a training and validation generator set","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\n# Split the dataframe into training and validation sets\ndf_train, df_val = train_test_split(data_all, test_size=0.2)\n\n# Create the generator functions\ntrain_generator = data_generator(df_train,median_filter=True)\nval_generator = data_generator(df_val,median_filter=True)\n","metadata":{"execution":{"iopub.status.busy":"2023-01-13T13:40:48.453594Z","iopub.execute_input":"2023-01-13T13:40:48.454017Z","iopub.status.idle":"2023-01-13T13:40:48.572828Z","shell.execute_reply.started":"2023-01-13T13:40:48.453961Z","shell.execute_reply":"2023-01-13T13:40:48.571768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img,label=next(train_generator)","metadata":{"execution":{"iopub.status.busy":"2023-01-13T13:40:53.48523Z","iopub.execute_input":"2023-01-13T13:40:53.485692Z","iopub.status.idle":"2023-01-13T13:40:54.059793Z","shell.execute_reply.started":"2023-01-13T13:40:53.485653Z","shell.execute_reply":"2023-01-13T13:40:54.058487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Look at histogram of flattened image array to see range of values","metadata":{}},{"cell_type":"code","source":"for i in range(0,31):\n    plt.hist(img[i].flatten(),alpha=0.5)","metadata":{"execution":{"iopub.status.busy":"2023-01-13T13:40:55.533035Z","iopub.execute_input":"2023-01-13T13:40:55.533406Z","iopub.status.idle":"2023-01-13T13:40:56.811872Z","shell.execute_reply.started":"2023-01-13T13:40:55.533375Z","shell.execute_reply":"2023-01-13T13:40:56.81092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img.shape","metadata":{"execution":{"iopub.status.busy":"2023-01-12T15:16:40.773319Z","iopub.execute_input":"2023-01-12T15:16:40.773699Z","iopub.status.idle":"2023-01-12T15:16:40.780678Z","shell.execute_reply.started":"2023-01-12T15:16:40.773666Z","shell.execute_reply":"2023-01-12T15:16:40.779495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(img[0])","metadata":{"execution":{"iopub.status.busy":"2023-01-12T15:16:42.952654Z","iopub.execute_input":"2023-01-12T15:16:42.95308Z","iopub.status.idle":"2023-01-12T15:16:43.264052Z","shell.execute_reply.started":"2023-01-12T15:16:42.953046Z","shell.execute_reply":"2023-01-12T15:16:43.263046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label","metadata":{"execution":{"iopub.status.busy":"2023-01-12T11:39:21.062257Z","iopub.execute_input":"2023-01-12T11:39:21.062724Z","iopub.status.idle":"2023-01-12T11:39:21.071824Z","shell.execute_reply.started":"2023-01-12T11:39:21.062689Z","shell.execute_reply":"2023-01-12T11:39:21.070844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The following class pF1 is from the notebook 'RSNA EfficientNetV2 Training Tensorflow TPU'","metadata":{}},{"cell_type":"code","source":"class pF1(tf.keras.metrics.Metric):\n    def __init__(self, name='pF1', **kwargs):\n        super(pF1, self).__init__(name=name, **kwargs)\n        self.tc = self.add_weight(name='tc', initializer='zeros')\n        self.tp = self.add_weight(name='tp', initializer='zeros')\n        self.fp = self.add_weight(name='fp', initializer='zeros')\n\n    def update_state(self, y_true, y_pred, sample_weight=None):\n        self.tc.assign_add(tf.cast(tf.reduce_sum(y_true), tf.float32))\n        self.tp.assign_add(tf.cast(tf.reduce_sum((y_pred[y_true == 1])), tf.float32))\n        self.fp.assign_add(tf.cast(tf.reduce_sum((y_pred[y_true == 0])), tf.float32))\n\n    def result(self):\n        if self.tc == 0 or (self.tp + self.fp) == 0:\n            return 0.0\n        else:\n            precision = self.tp / (self.tp + self.fp)\n            recall = self.tp / (self.tc)\n            return 2 * (precision * recall) / (precision + recall)\n\n        def reset_state(self):\n            self.tc.assign(0)\n            self.tp.assign(0)\n            self.fp.assign(0)","metadata":{"execution":{"iopub.status.busy":"2023-01-13T13:41:01.721574Z","iopub.execute_input":"2023-01-13T13:41:01.721958Z","iopub.status.idle":"2023-01-13T13:41:01.732576Z","shell.execute_reply.started":"2023-01-13T13:41:01.721927Z","shell.execute_reply":"2023-01-13T13:41:01.731279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Define metrics and loss","metadata":{}},{"cell_type":"code","source":"metrics = [pF1(),\n           tf.keras.metrics.Precision(),\n           tf.keras.metrics.Recall(),\n           tf.keras.metrics.AUC(),\n           tf.keras.metrics.BinaryAccuracy()\n        ]\nloss=tf.keras.losses.BinaryCrossentropy(from_logits=False)","metadata":{"execution":{"iopub.status.busy":"2023-01-13T13:41:04.473747Z","iopub.execute_input":"2023-01-13T13:41:04.475295Z","iopub.status.idle":"2023-01-13T13:41:08.749744Z","shell.execute_reply.started":"2023-01-13T13:41:04.475247Z","shell.execute_reply":"2023-01-13T13:41:08.748708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Try using a pre-trained model, eg VGG16","metadata":{}},{"cell_type":"code","source":"from keras.applications.vgg16 import VGG16\nfrom keras.applications.densenet import DenseNet121\nfrom keras.models import Model\nfrom keras.layers import Flatten,Dense,Dropout,Conv2D,MaxPooling2D","metadata":{"execution":{"iopub.status.busy":"2023-01-13T13:41:12.818256Z","iopub.execute_input":"2023-01-13T13:41:12.818909Z","iopub.status.idle":"2023-01-13T13:41:12.827105Z","shell.execute_reply.started":"2023-01-13T13:41:12.818877Z","shell.execute_reply":"2023-01-13T13:41:12.826065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def model_pipeline(model,save_model):\n    model = model\n    model.summary()\n    model.compile(loss=loss, optimizer='adam', metrics=metrics)\n    history = model.fit(train_generator, \n                              steps_per_epoch=len(df_train)//32, \n                              epochs=2,\n                              validation_data=val_generator, \n                              validation_steps=len(df_val)//32)\n    model.save(save_model+'.h5')\n    return model","metadata":{"execution":{"iopub.status.busy":"2023-01-13T13:41:14.76729Z","iopub.execute_input":"2023-01-13T13:41:14.76773Z","iopub.status.idle":"2023-01-13T13:41:14.777167Z","shell.execute_reply.started":"2023-01-13T13:41:14.767693Z","shell.execute_reply":"2023-01-13T13:41:14.776048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def VGG_model():\n# load the model\n    model_vgg16_base = VGG16(include_top=False, input_shape=(512,512,3))\n    model_vgg16_base.trainable=False\n    full_model = model_vgg16_base.output\n    full_model = Flatten(name=\"flatten\")(full_model)\n    full_model = Dense(128, activation=\"relu\")(full_model)\n    full_model = Dropout(0.5)(full_model)\n    full_model = Dense(2, activation=\"softmax\")(full_model)\n    model_VGG = Model(inputs=model_vgg16_base.input, outputs=full_model)\n    return model_VGG","metadata":{"execution":{"iopub.status.busy":"2023-01-11T14:50:19.314127Z","iopub.execute_input":"2023-01-11T14:50:19.314502Z","iopub.status.idle":"2023-01-11T14:50:19.321521Z","shell.execute_reply.started":"2023-01-11T14:50:19.314469Z","shell.execute_reply":"2023-01-11T14:50:19.320241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def DenseNet_model():\n    # Instantiate the DenseNet121 model\n    densenet = DenseNet121(weights='imagenet', include_top=False, input_shape=(512,512,3))\n    densenet.trainable = False\n    full_model = densenet.output\n    full_model = Flatten(name=\"flatten\")(full_model)\n    full_model = Dense(128, activation=\"relu\")(full_model)\n    full_model = Dropout(0.5)(full_model)\n    full_model = Dense(2, activation=\"softmax\")(full_model)\n    model_densenet = Model(inputs=densenet.input, outputs=full_model)\n    return model_densenet\n","metadata":{"execution":{"iopub.status.busy":"2023-01-12T11:39:43.060758Z","iopub.execute_input":"2023-01-12T11:39:43.061162Z","iopub.status.idle":"2023-01-12T11:39:43.068853Z","shell.execute_reply.started":"2023-01-12T11:39:43.06113Z","shell.execute_reply":"2023-01-12T11:39:43.067528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vgg_16_model=model_pipeline(VGG_model(),'vgg_16_model')","metadata":{"execution":{"iopub.status.busy":"2023-01-11T14:50:22.278541Z","iopub.execute_input":"2023-01-11T14:50:22.279234Z","iopub.status.idle":"2023-01-11T15:26:30.522764Z","shell.execute_reply.started":"2023-01-11T14:50:22.279196Z","shell.execute_reply":"2023-01-11T15:26:30.521705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"denseNet_model=model_pipeline(DenseNet_model(),'DenseNet_model')","metadata":{"execution":{"iopub.status.busy":"2023-01-12T11:39:49.276414Z","iopub.execute_input":"2023-01-12T11:39:49.277197Z","iopub.status.idle":"2023-01-12T12:12:22.893932Z","shell.execute_reply.started":"2023-01-12T11:39:49.277157Z","shell.execute_reply":"2023-01-12T12:12:22.892348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Try EfficientNet","metadata":{}},{"cell_type":"code","source":"!pip install --no-deps  /kaggle/input/kerasefficientnetv2/keras_efficientnet_v2-1.2.2-py3-none-any.whl","metadata":{"execution":{"iopub.status.busy":"2023-01-13T13:41:24.451155Z","iopub.execute_input":"2023-01-13T13:41:24.452128Z","iopub.status.idle":"2023-01-13T13:41:27.25527Z","shell.execute_reply.started":"2023-01-13T13:41:24.452081Z","shell.execute_reply":"2023-01-13T13:41:27.254042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#keras_efficientnet_v2.EfficientNetV2T\n#keras-efficientnet-v2\nimport keras_efficientnet_v2","metadata":{"execution":{"iopub.status.busy":"2023-01-13T13:41:27.258096Z","iopub.execute_input":"2023-01-13T13:41:27.258546Z","iopub.status.idle":"2023-01-13T13:41:27.273255Z","shell.execute_reply.started":"2023-01-13T13:41:27.258503Z","shell.execute_reply":"2023-01-13T13:41:27.272219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def EfficientNet_v2():\n    # Instantiate the DenseNet121 model\n    #densenet = DenseNet121(weights='imagenet', include_top=False, input_shape=(512,512,3))\n    efficientnet=keras_efficientnet_v2.EfficientNetV2T(input_shape=(512,512,3),\n            pretrained='imagenet',\n            num_classes=2,\n            classifier_activation='sigmoid',\n            dropout=0.30)\n    efficientnet.trainable = False\n    full_model = efficientnet.output\n    full_model = Flatten(name=\"flatten\")(full_model)\n    full_model = Dense(128, activation=\"relu\")(full_model)\n    full_model = Dropout(0.5)(full_model)\n    full_model = Dense(2, activation=\"softmax\")(full_model)\n    model_efficientnet = Model(inputs=efficientnet.input, outputs=full_model)\n    return model_efficientnet","metadata":{"execution":{"iopub.status.busy":"2023-01-13T13:42:04.244957Z","iopub.execute_input":"2023-01-13T13:42:04.245638Z","iopub.status.idle":"2023-01-13T13:42:04.252933Z","shell.execute_reply.started":"2023-01-13T13:42:04.245601Z","shell.execute_reply":"2023-01-13T13:42:04.251377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"EfficientNet_v2_model=model_pipeline(EfficientNet_v2(),'EfficientNet_model_v2')","metadata":{"execution":{"iopub.status.busy":"2023-01-13T14:53:28.556411Z","iopub.execute_input":"2023-01-13T14:53:28.556877Z","iopub.status.idle":"2023-01-13T15:53:05.403239Z","shell.execute_reply.started":"2023-01-13T14:53:28.556826Z","shell.execute_reply":"2023-01-13T15:53:05.402051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Look at the predictions from the validation set and how they compare to the labels","metadata":{}},{"cell_type":"code","source":"def val_accuracy(model,n_batches):\n    list_batch_acc=[]\n    for i in range(0,n_batches):\n        preds_sample=model.predict(next(val_generator)[0])\n        true_sample=next(val_generator)[1]\n        batch_accuracy=(np.sum(true_sample==np.round(preds_sample,1))/2)/len(preds_sample)\n        print(batch_accuracy*100,'%')\n        list_batch_acc.append(batch_accuracy)\n    return list_batch_acc,preds_sample,true_sample","metadata":{"execution":{"iopub.status.busy":"2023-01-13T15:53:05.524552Z","iopub.execute_input":"2023-01-13T15:53:05.525328Z","iopub.status.idle":"2023-01-13T15:53:05.531872Z","shell.execute_reply.started":"2023-01-13T15:53:05.52529Z","shell.execute_reply":"2023-01-13T15:53:05.53069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#list_batch_accuracy,preds,true=val_accuracy(vgg_16_model,10)","metadata":{"execution":{"iopub.status.busy":"2023-01-12T13:05:47.416938Z","iopub.execute_input":"2023-01-12T13:05:47.417321Z","iopub.status.idle":"2023-01-12T13:05:47.422314Z","shell.execute_reply.started":"2023-01-12T13:05:47.417287Z","shell.execute_reply":"2023-01-12T13:05:47.421159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#list_batch_accuracy,preds,true=val_accuracy(denseNet_model,10)\nlist_batch_accuracy,preds,true=val_accuracy(EfficientNet_v2_model,10)","metadata":{"execution":{"iopub.status.busy":"2023-01-13T14:47:03.014431Z","iopub.execute_input":"2023-01-13T14:47:03.014791Z","iopub.status.idle":"2023-01-13T14:47:20.045255Z","shell.execute_reply.started":"2023-01-13T14:47:03.014761Z","shell.execute_reply":"2023-01-13T14:47:20.044066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Explore why predictions are so poor:","metadata":{}},{"cell_type":"code","source":"preds[0:5]","metadata":{"execution":{"iopub.status.busy":"2023-01-13T14:47:34.047346Z","iopub.execute_input":"2023-01-13T14:47:34.047832Z","iopub.status.idle":"2023-01-13T14:47:34.057433Z","shell.execute_reply.started":"2023-01-13T14:47:34.047786Z","shell.execute_reply":"2023-01-13T14:47:34.056052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"true[0:5]","metadata":{"execution":{"iopub.status.busy":"2023-01-13T14:48:08.836076Z","iopub.execute_input":"2023-01-13T14:48:08.837056Z","iopub.status.idle":"2023-01-13T14:48:08.845218Z","shell.execute_reply.started":"2023-01-13T14:48:08.837006Z","shell.execute_reply":"2023-01-13T14:48:08.843874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Not suprising - since the model is not learning anything, it outputs roughly (0.5,0.5) in the predictions","metadata":{}},{"cell_type":"markdown","source":"Make a simple 2D CNN model","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense\nfrom tensorflow.keras.models import Sequential","metadata":{"execution":{"iopub.status.busy":"2023-01-11T13:55:25.173342Z","iopub.execute_input":"2023-01-11T13:55:25.173733Z","iopub.status.idle":"2023-01-11T13:55:25.183894Z","shell.execute_reply.started":"2023-01-11T13:55:25.173701Z","shell.execute_reply":"2023-01-11T13:55:25.182936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def manual_model():\n    input_shape = (512, 512, 3)\n\n    # Create the model\n    model = Sequential()\n    model.add(Conv2D(16, kernel_size=(3,3), activation='relu', input_shape=input_shape))\n    model.add(MaxPooling2D(pool_size=(2,2)))\n    model.add(Conv2D(32, kernel_size=(3,3), activation='relu'))\n    model.add(MaxPooling2D(pool_size=(2,2)))\n    model.add(Conv2D(64, kernel_size=(3,3), activation='relu'))\n    model.add(MaxPooling2D(pool_size=(2,2)))\n    model.add(Conv2D(64, kernel_size=(3,3), activation='relu'))\n    model.add(MaxPooling2D(pool_size=(2,2)))\n    model.add(Flatten())\n    model.add(Dense(128, activation='relu'))\n    model.add(Dense(2, activation='sigmoid'))\n    return model","metadata":{"execution":{"iopub.status.busy":"2023-01-11T13:53:20.924928Z","iopub.execute_input":"2023-01-11T13:53:20.925289Z","iopub.status.idle":"2023-01-11T13:53:20.935666Z","shell.execute_reply.started":"2023-01-11T13:53:20.92526Z","shell.execute_reply":"2023-01-11T13:53:20.934707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model2=model_pipeline(manual_model(),'my_model2')","metadata":{"execution":{"iopub.status.busy":"2023-01-11T13:56:08.111002Z","iopub.execute_input":"2023-01-11T13:56:08.111464Z","iopub.status.idle":"2023-01-11T14:26:45.052422Z","shell.execute_reply.started":"2023-01-11T13:56:08.111423Z","shell.execute_reply":"2023-01-11T14:26:45.050966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Look at validation accuracy in batches of 32 images","metadata":{}},{"cell_type":"code","source":"list_batch_accuracy,preds,true=val_accuracy(model2,10)","metadata":{"execution":{"iopub.status.busy":"2023-01-11T14:37:53.279491Z","iopub.execute_input":"2023-01-11T14:37:53.2802Z","iopub.status.idle":"2023-01-11T14:38:04.978778Z","shell.execute_reply.started":"2023-01-11T14:37:53.28016Z","shell.execute_reply":"2023-01-11T14:38:04.977504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('predictions=',np.round(preds))\nprint('true=',true)","metadata":{"execution":{"iopub.status.busy":"2023-01-11T14:39:28.596218Z","iopub.execute_input":"2023-01-11T14:39:28.596604Z","iopub.status.idle":"2023-01-11T14:39:28.605076Z","shell.execute_reply.started":"2023-01-11T14:39:28.596572Z","shell.execute_reply":"2023-01-11T14:39:28.603717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Average accuracy for this collection of batches is ',np.mean(np.asarray(list_batch_accuracy))*100,'%') ","metadata":{"execution":{"iopub.status.busy":"2023-01-11T14:36:58.387056Z","iopub.execute_input":"2023-01-11T14:36:58.387453Z","iopub.status.idle":"2023-01-11T14:36:58.393993Z","shell.execute_reply.started":"2023-01-11T14:36:58.387422Z","shell.execute_reply":"2023-01-11T14:36:58.392731Z"},"trusted":true},"execution_count":null,"outputs":[]}]}