{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# TensorFlow libraries\nimport tensorflow as tf\nfrom tensorflow.keras.applications.resnet_v2 import ResNet50V2\nfrom tensorflow.keras.optimizers import RMSprop\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\nfrom tensorflow.keras.layers import Dense, GlobalAveragePooling2D, Dropout\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.optimizers import Adam\n\n# basic libraries\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report, confusion_matrix\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport cv2\nimport os\n\nimport glob\nfrom glob import glob","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:43:05.174699Z","iopub.execute_input":"2023-04-18T07:43:05.17549Z","iopub.status.idle":"2023-04-18T07:43:15.822405Z","shell.execute_reply.started":"2023-04-18T07:43:05.175448Z","shell.execute_reply":"2023-04-18T07:43:15.821296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# the path to the image data\nRSNA_512_path = '/kaggle/input/rsna-breast-cancer-512-pngs'","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:43:15.824932Z","iopub.execute_input":"2023-04-18T07:43:15.825759Z","iopub.status.idle":"2023-04-18T07:43:15.83113Z","shell.execute_reply.started":"2023-04-18T07:43:15.825712Z","shell.execute_reply":"2023-04-18T07:43:15.829775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read the csv data.\ndf_train = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:43:15.832478Z","iopub.execute_input":"2023-04-18T07:43:15.833117Z","iopub.status.idle":"2023-04-18T07:43:15.970863Z","shell.execute_reply.started":"2023-04-18T07:43:15.833078Z","shell.execute_reply":"2023-04-18T07:43:15.969825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# the number of total patients\nlen(df_train)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:43:15.974154Z","iopub.execute_input":"2023-04-18T07:43:15.974565Z","iopub.status.idle":"2023-04-18T07:43:15.981122Z","shell.execute_reply.started":"2023-04-18T07:43:15.974524Z","shell.execute_reply":"2023-04-18T07:43:15.980009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# the number of patients having implant\nlen(df_train[df_train['implant'] == 1])","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:43:15.982976Z","iopub.execute_input":"2023-04-18T07:43:15.983736Z","iopub.status.idle":"2023-04-18T07:43:16.001281Z","shell.execute_reply.started":"2023-04-18T07:43:15.983699Z","shell.execute_reply":"2023-04-18T07:43:16.000022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# the number of patients without malignant cancer\nlen(df_train[df_train['cancer'] == 0])","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:43:16.00409Z","iopub.execute_input":"2023-04-18T07:43:16.004363Z","iopub.status.idle":"2023-04-18T07:43:16.019034Z","shell.execute_reply.started":"2023-04-18T07:43:16.004337Z","shell.execute_reply":"2023-04-18T07:43:16.017717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# the number of patient who took biopsy\nlen(df_train[df_train['biopsy'] == 1])","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:43:16.020567Z","iopub.execute_input":"2023-04-18T07:43:16.021006Z","iopub.status.idle":"2023-04-18T07:43:16.030357Z","shell.execute_reply.started":"2023-04-18T07:43:16.020971Z","shell.execute_reply":"2023-04-18T07:43:16.029165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# the number of patients having malignant cancer\nlen(df_train[df_train['cancer'] == 1])","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:43:16.032313Z","iopub.execute_input":"2023-04-18T07:43:16.032709Z","iopub.status.idle":"2023-04-18T07:43:16.043072Z","shell.execute_reply.started":"2023-04-18T07:43:16.032671Z","shell.execute_reply":"2023-04-18T07:43:16.041747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# the number of patients whose malignant cancer is invasive\nlen(df_train[df_train['invasive'] == 1])","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:43:16.045044Z","iopub.execute_input":"2023-04-18T07:43:16.045703Z","iopub.status.idle":"2023-04-18T07:43:16.055085Z","shell.execute_reply.started":"2023-04-18T07:43:16.045662Z","shell.execute_reply":"2023-04-18T07:43:16.053817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# the same as above\nlen(df_train[(df_train['cancer'] == 1) & (df_train['invasive'] == 1)])","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:43:16.06193Z","iopub.execute_input":"2023-04-18T07:43:16.062226Z","iopub.status.idle":"2023-04-18T07:43:16.071348Z","shell.execute_reply.started":"2023-04-18T07:43:16.062198Z","shell.execute_reply":"2023-04-18T07:43:16.070287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Most of the cases are normal or not-malignant cancer. Thus, physicians sometimes overlook cancer.\ndata = pd.DataFrame(np.concatenate([['Total'] * len(df_train) , ['Maglignant Cancer'] *  len(df_train[df_train['cancer'] == 1]), ['Invasive Cancer'] *  len(df_train[(df_train['cancer'] == 1) & (df_train['invasive'] == 1)])]), columns = [\"class\"])\n\nsns.countplot(x = 'class', data = data)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:43:16.073013Z","iopub.execute_input":"2023-04-18T07:43:16.073857Z","iopub.status.idle":"2023-04-18T07:43:16.351689Z","shell.execute_reply.started":"2023-04-18T07:43:16.073816Z","shell.execute_reply":"2023-04-18T07:43:16.350603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Around 3000 patient took biopsy and malignant cancer was found from some of them.\ndata = pd.DataFrame(np.concatenate([['Biopsy'] * len(df_train[df_train['biopsy'] == 1]) , ['Malignant Cancer'] *  len(df_train[df_train['cancer'] == 1]), ['Invasive Cancer'] *  len(df_train[(df_train['cancer'] == 1) & (df_train['invasive'] == 1)])]), columns = [\"class\"])\n\nsns.countplot(x = 'class', data = data)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:43:16.353282Z","iopub.execute_input":"2023-04-18T07:43:16.354945Z","iopub.status.idle":"2023-04-18T07:43:16.574811Z","shell.execute_reply.started":"2023-04-18T07:43:16.354899Z","shell.execute_reply":"2023-04-18T07:43:16.57383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# the number of not-malignant cancer cases from biopsy\nlen(df_train[(df_train['biopsy'] == 1) & (df_train['cancer'] == 0)])","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:43:16.576576Z","iopub.execute_input":"2023-04-18T07:43:16.577202Z","iopub.status.idle":"2023-04-18T07:43:16.588655Z","shell.execute_reply.started":"2023-04-18T07:43:16.577161Z","shell.execute_reply":"2023-04-18T07:43:16.587398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# the number of malignant cancer cases from biopsy\nlen(df_train[(df_train['biopsy'] == 1) & (df_train['cancer'] == 1)])","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:43:16.590523Z","iopub.execute_input":"2023-04-18T07:43:16.590981Z","iopub.status.idle":"2023-04-18T07:43:16.600374Z","shell.execute_reply.started":"2023-04-18T07:43:16.590915Z","shell.execute_reply":"2023-04-18T07:43:16.59919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 60% of biopsy resulted in not-malignanct cancer.\ndata = pd.DataFrame(np.concatenate([['Biopsy but Not Malignant'] * len(df_train[(df_train['biopsy'] == 1) & (df_train['cancer'] == 0)]) , ['Malignant Cancer'] *  len(df_train[df_train['cancer'] == 1]), ['Invasive Cancer'] *  len(df_train[(df_train['cancer'] == 1) & (df_train['invasive'] == 1)])]), columns = [\"class\"])\n\nsns.countplot(x = 'class', data = data)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:43:16.602044Z","iopub.execute_input":"2023-04-18T07:43:16.602418Z","iopub.status.idle":"2023-04-18T07:43:16.829446Z","shell.execute_reply.started":"2023-04-18T07:43:16.602378Z","shell.execute_reply":"2023-04-18T07:43:16.828341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The not-malignant cancer cases were limited into biopsy cases.\nDF_train = df_train[df_train['biopsy'] == 1].reset_index(drop = True)\nDF_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:43:16.831098Z","iopub.execute_input":"2023-04-18T07:43:16.831951Z","iopub.status.idle":"2023-04-18T07:43:16.854282Z","shell.execute_reply.started":"2023-04-18T07:43:16.831911Z","shell.execute_reply":"2023-04-18T07:43:16.853284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The number of positive (malignant) and negative (not-malignat) cases should be the same\n# to create a balanced dataset.\nDF_train = DF_train.groupby(['cancer']).apply(lambda x: x.sample(1158, replace = True)\n                                                      ).reset_index(drop = True)\nprint('New Data Size:', DF_train.shape[0])","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:43:16.858145Z","iopub.execute_input":"2023-04-18T07:43:16.858498Z","iopub.status.idle":"2023-04-18T07:43:16.878626Z","shell.execute_reply.started":"2023-04-18T07:43:16.858464Z","shell.execute_reply":"2023-04-18T07:43:16.87716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Generally, invasive cancer is confirmed by biopsy, not by mammography.\n# Maybe it is also extremely difficult for AI to detect invasive cancer from mammography.\ndata = pd.DataFrame(np.concatenate([['Biopsy but Not Malignant'] * len(DF_train[(DF_train['biopsy'] == 1) & (DF_train['cancer'] == 0)]) , ['Malignant Cancer'] *  len(DF_train[DF_train['cancer'] == 1]), ['Invasive Cancer'] *  len(DF_train[(DF_train['cancer'] == 1) & (DF_train['invasive'] == 1)])]), columns = [\"class\"])\n\nsns.countplot(x = 'class', data = data)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:43:16.880654Z","iopub.execute_input":"2023-04-18T07:43:16.881054Z","iopub.status.idle":"2023-04-18T07:43:17.107867Z","shell.execute_reply.started":"2023-04-18T07:43:16.881015Z","shell.execute_reply":"2023-04-18T07:43:17.10666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create the Path to Each Image","metadata":{}},{"cell_type":"code","source":"# Create the path to each image.\nfor i in range(len(DF_train)):\n    DF_train.loc[i, 'path'] = os.path.join(RSNA_512_path + '/' + str(DF_train.loc[i, 'patient_id']) + '_' + str(DF_train.loc[i, 'image_id']) + '.png')\nDF_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:43:17.109476Z","iopub.execute_input":"2023-04-18T07:43:17.110149Z","iopub.status.idle":"2023-04-18T07:43:18.056987Z","shell.execute_reply.started":"2023-04-18T07:43:17.110109Z","shell.execute_reply":"2023-04-18T07:43:18.055783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# a sample path\nDF_train.loc[0, 'path']","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:43:18.05856Z","iopub.execute_input":"2023-04-18T07:43:18.059351Z","iopub.status.idle":"2023-04-18T07:43:18.067843Z","shell.execute_reply.started":"2023-04-18T07:43:18.059314Z","shell.execute_reply":"2023-04-18T07:43:18.066244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# a sample image\nimg = cv2.imread(DF_train.loc[0, 'path'])\nplt.imshow(img, cmap = 'gray')","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:43:18.069561Z","iopub.execute_input":"2023-04-18T07:43:18.070997Z","iopub.status.idle":"2023-04-18T07:43:18.425631Z","shell.execute_reply.started":"2023-04-18T07:43:18.07095Z","shell.execute_reply":"2023-04-18T07:43:18.424456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:43:18.426862Z","iopub.execute_input":"2023-04-18T07:43:18.427284Z","iopub.status.idle":"2023-04-18T07:43:18.437241Z","shell.execute_reply.started":"2023-04-18T07:43:18.427242Z","shell.execute_reply":"2023-04-18T07:43:18.436044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img.shape","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:43:18.439473Z","iopub.execute_input":"2023-04-18T07:43:18.440398Z","iopub.status.idle":"2023-04-18T07:43:18.447855Z","shell.execute_reply.started":"2023-04-18T07:43:18.440341Z","shell.execute_reply":"2023-04-18T07:43:18.446637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Normal and cancer images must be equally distrubuted.\ntrain_df, val_df = train_test_split(DF_train, \n                                   test_size = 0.30, \n                                   random_state = 2018,\n                                   stratify = DF_train[['cancer']])\n\nprint('train', train_df.shape[0], 'validation', val_df.shape[0])\nprint('train', train_df['cancer'].value_counts())\nprint('validation', val_df['cancer'].value_counts())\ntrain_df.sample(1)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:43:18.44965Z","iopub.execute_input":"2023-04-18T07:43:18.450436Z","iopub.status.idle":"2023-04-18T07:43:18.488098Z","shell.execute_reply.started":"2023-04-18T07:43:18.450397Z","shell.execute_reply":"2023-04-18T07:43:18.486692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# training data\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:43:18.490346Z","iopub.execute_input":"2023-04-18T07:43:18.490832Z","iopub.status.idle":"2023-04-18T07:43:18.511107Z","shell.execute_reply.started":"2023-04-18T07:43:18.490786Z","shell.execute_reply":"2023-04-18T07:43:18.509642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# validation data\nval_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:43:18.513108Z","iopub.execute_input":"2023-04-18T07:43:18.514158Z","iopub.status.idle":"2023-04-18T07:43:18.537541Z","shell.execute_reply.started":"2023-04-18T07:43:18.514094Z","shell.execute_reply":"2023-04-18T07:43:18.536488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pick up normal images from the training data.\ntrain_df_normal = train_df[train_df['cancer'] == 0].reset_index(drop = True)\nprint(len(train_df_normal))\ntrain_df_normal.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:43:18.539133Z","iopub.execute_input":"2023-04-18T07:43:18.541939Z","iopub.status.idle":"2023-04-18T07:43:18.564929Z","shell.execute_reply.started":"2023-04-18T07:43:18.541895Z","shell.execute_reply":"2023-04-18T07:43:18.563689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pick up cancer images from the training data.\ntrain_df_cancer = train_df[train_df['cancer'] == 1].reset_index(drop = True)\nprint(len(train_df_cancer))\ntrain_df_cancer.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:43:18.574282Z","iopub.execute_input":"2023-04-18T07:43:18.575402Z","iopub.status.idle":"2023-04-18T07:43:18.599077Z","shell.execute_reply.started":"2023-04-18T07:43:18.575359Z","shell.execute_reply":"2023-04-18T07:43:18.597655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pick up normal images from the validation data.\nval_df_normal = val_df[val_df['cancer'] == 0].reset_index(drop = True)\nprint(len(val_df_normal))\nval_df_normal.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:43:18.600783Z","iopub.execute_input":"2023-04-18T07:43:18.601704Z","iopub.status.idle":"2023-04-18T07:43:18.626349Z","shell.execute_reply.started":"2023-04-18T07:43:18.60166Z","shell.execute_reply":"2023-04-18T07:43:18.625152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pick up cancer images from the validation data.\nval_df_cancer = val_df[val_df['cancer'] == 1].reset_index(drop = True)\nprint(len(val_df_cancer))\nval_df_cancer.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:43:18.628095Z","iopub.execute_input":"2023-04-18T07:43:18.629109Z","iopub.status.idle":"2023-04-18T07:43:18.653367Z","shell.execute_reply.started":"2023-04-18T07:43:18.629049Z","shell.execute_reply":"2023-04-18T07:43:18.652091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**We have to store the 4 datasets in each folder to be used for machine learing of TensorFlow AI model.**","metadata":{}},{"cell_type":"code","source":"import shutil\n# Define the destination directory.\ndestination_dir = '/kaggle/working/train'\ndestination_dir_sub = '/kaggle/working/train/normal'\n\n# Create the destination directory if it doesn't exist.\nif not os.path.exists(destination_dir):\n    os.makedirs(destination_dir)\n\nif not os.path.exists(destination_dir_sub):\n    os.makedirs(destination_dir_sub)   \n    \n# Copy the images to the destination directory.\nfor path in train_df_normal['path']:\n    shutil.copy2(path, destination_dir_sub)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:43:18.655218Z","iopub.execute_input":"2023-04-18T07:43:18.656176Z","iopub.status.idle":"2023-04-18T07:43:25.060161Z","shell.execute_reply.started":"2023-04-18T07:43:18.656133Z","shell.execute_reply":"2023-04-18T07:43:25.058844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the destination directory.\ndestination_dir = '/kaggle/working/train'\ndestination_dir_sub = '/kaggle/working/train/cancer'\n\n# Create the destination directory if it doesn't exist.\nif not os.path.exists(destination_dir):\n    os.makedirs(destination_dir)\n\nif not os.path.exists(destination_dir_sub):\n    os.makedirs(destination_dir_sub)   \n    \n# Copy the images to the destination directory.\nfor path in train_df_cancer['path']:\n    shutil.copy2(path, destination_dir_sub)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:43:25.061863Z","iopub.execute_input":"2023-04-18T07:43:25.062802Z","iopub.status.idle":"2023-04-18T07:43:29.911358Z","shell.execute_reply.started":"2023-04-18T07:43:25.062752Z","shell.execute_reply":"2023-04-18T07:43:29.910218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the destination directory.\ndestination_dir = '/kaggle/working/val'\ndestination_dir_sub = '/kaggle/working/val/normal'\n\n# Create the destination directory if it doesn't exist.\nif not os.path.exists(destination_dir):\n    os.makedirs(destination_dir)\n\nif not os.path.exists(destination_dir_sub):\n    os.makedirs(destination_dir_sub)   \n    \n# Copy the images to the destination directory.\nfor path in val_df_normal['path']:\n    shutil.copy2(path, destination_dir_sub)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:43:29.91547Z","iopub.execute_input":"2023-04-18T07:43:29.915805Z","iopub.status.idle":"2023-04-18T07:43:31.821244Z","shell.execute_reply.started":"2023-04-18T07:43:29.915775Z","shell.execute_reply":"2023-04-18T07:43:31.820152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the destination directory.\ndestination_dir = '/kaggle/working/val'\ndestination_dir_sub = '/kaggle/working/val/cancer'\n\n# Create the destination directory if it doesn't exist.\nif not os.path.exists(destination_dir):\n    os.makedirs(destination_dir)\n\nif not os.path.exists(destination_dir_sub):\n    os.makedirs(destination_dir_sub)   \n    \n# Copy the images to the destination directory.\nfor path in val_df_cancer['path']:\n    shutil.copy2(path, destination_dir_sub)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:43:31.822852Z","iopub.execute_input":"2023-04-18T07:43:31.823338Z","iopub.status.idle":"2023-04-18T07:43:33.338871Z","shell.execute_reply.started":"2023-04-18T07:43:31.823293Z","shell.execute_reply":"2023-04-18T07:43:33.337672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Sample Images","metadata":{}},{"cell_type":"code","source":"import glob\nnormal_train_images = glob.glob('/kaggle/working/train/normal/*.png')\ncancer_train_images = glob.glob('/kaggle/working/train/cancer/*.png')","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:43:33.344234Z","iopub.execute_input":"2023-04-18T07:43:33.347245Z","iopub.status.idle":"2023-04-18T07:43:33.361562Z","shell.execute_reply.started":"2023-04-18T07:43:33.347203Z","shell.execute_reply":"2023-04-18T07:43:33.36048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# See normal images from the training dataset.\nfig, axes = plt.subplots(nrows = 2, ncols = 5, figsize = (15, 10), subplot_kw = {'xticks':[], 'yticks':[]})\nfor i, ax in enumerate(axes.flat):\n    img = cv2.imread(normal_train_images[i])\n    ax.imshow(img)\n    ax.set_title('Normal')\nfig.tight_layout()    \n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:43:33.366208Z","iopub.execute_input":"2023-04-18T07:43:33.367168Z","iopub.status.idle":"2023-04-18T07:43:34.549906Z","shell.execute_reply.started":"2023-04-18T07:43:33.367127Z","shell.execute_reply":"2023-04-18T07:43:34.548754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# See cancer images from the training dataset.\nfig, axes = plt.subplots(nrows = 2, ncols = 5, figsize = (15, 10), subplot_kw = {'xticks':[], 'yticks':[]})\nfor i, ax in enumerate(axes.flat):\n    img = cv2.imread(cancer_train_images[i])\n    ax.imshow(img)\n    ax.set_title('Cancer')\n    \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:43:34.552239Z","iopub.execute_input":"2023-04-18T07:43:34.552895Z","iopub.status.idle":"2023-04-18T07:43:35.488519Z","shell.execute_reply.started":"2023-04-18T07:43:34.552857Z","shell.execute_reply":"2023-04-18T07:43:35.486506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create Image data Generators\n\nThe dataset has already been divided into train and validation datasets, and each dataset includes normal and cancer image files. Thus, image data generators were easily created. ","metadata":{}},{"cell_type":"code","source":"train_datagen = ImageDataGenerator(rescale = 1./255.,\n                                   zoom_range = 0.2)\nval_datagen = ImageDataGenerator(rescale = 1./255.,)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:43:35.490059Z","iopub.execute_input":"2023-04-18T07:43:35.491278Z","iopub.status.idle":"2023-04-18T07:43:35.49767Z","shell.execute_reply.started":"2023-04-18T07:43:35.491238Z","shell.execute_reply":"2023-04-18T07:43:35.496139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_path = '/kaggle/working/train'\nval_path = '/kaggle/working/val'\n\ntrain_generator = train_datagen.flow_from_directory(\n    train_path,\n    target_size = (512, 512),\n    batch_size = 32,\n    class_mode = 'binary'\n)\nvalidation_generator = val_datagen.flow_from_directory(\n        val_path,\n        target_size = (512, 512),\n        batch_size = 16,\n        class_mode = 'binary'\n)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:43:35.499962Z","iopub.execute_input":"2023-04-18T07:43:35.500392Z","iopub.status.idle":"2023-04-18T07:43:35.718945Z","shell.execute_reply.started":"2023-04-18T07:43:35.500344Z","shell.execute_reply":"2023-04-18T07:43:35.717696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Define the Model (Transfer Learning)\n\nHere, ResNet50V2 was employed as the base model for transfer learning. Sigmoid, adam, and binary cross entropy were selected for the final activation function, optimizer, and loss function, respectively.","metadata":{}},{"cell_type":"code","source":"base_model = ResNet50V2(weights = 'imagenet', input_shape = (512, 512, 3), include_top = False)\n\nfor layer in base_model.layers:\n    layer.trainable = False\n    \nmodel = Sequential()\nmodel.add(base_model)\nmodel.add(GlobalAveragePooling2D())\nmodel.add(Dense(128, activation = 'relu'))\nmodel.add(Dropout(0.2))\nmodel.add(Dense(1, activation = 'sigmoid'))\n\nmodel.compile(optimizer = \"adam\", loss = 'binary_crossentropy', metrics = [\"accuracy\"])","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:43:35.721754Z","iopub.execute_input":"2023-04-18T07:43:35.722459Z","iopub.status.idle":"2023-04-18T07:43:41.495454Z","shell.execute_reply.started":"2023-04-18T07:43:35.722416Z","shell.execute_reply":"2023-04-18T07:43:41.49418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:43:41.500503Z","iopub.execute_input":"2023-04-18T07:43:41.502981Z","iopub.status.idle":"2023-04-18T07:43:41.568173Z","shell.execute_reply.started":"2023-04-18T07:43:41.502941Z","shell.execute_reply":"2023-04-18T07:43:41.567232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train the Model\n\nThe model was trained with the train and validation data. Early stopping was added to prevent overfitting. ","metadata":{}},{"cell_type":"code","source":"callback = tf.keras.callbacks.EarlyStopping(monitor = \"val_loss\", mode = \"min\", patience = 4)\n\nhistory = model.fit(train_generator, validation_data = validation_generator, steps_per_epoch = 20, epochs = 15, callbacks = callback)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:43:41.569556Z","iopub.execute_input":"2023-04-18T07:43:41.57008Z","iopub.status.idle":"2023-04-18T07:56:02.232098Z","shell.execute_reply.started":"2023-04-18T07:43:41.570039Z","shell.execute_reply":"2023-04-18T07:56:02.231098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Save the Model","metadata":{}},{"cell_type":"code","source":"model.save('mammography_pred_model.h5')","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:56:02.233824Z","iopub.execute_input":"2023-04-18T07:56:02.234168Z","iopub.status.idle":"2023-04-18T07:56:02.707042Z","shell.execute_reply.started":"2023-04-18T07:56:02.234131Z","shell.execute_reply":"2023-04-18T07:56:02.705934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Metrics\n\nThe accuracy and loss were calculated for both train and validation data in the model training process.","metadata":{}},{"cell_type":"code","source":"accuracy = history.history['accuracy']\nval_accuracy = history.history['val_accuracy']\n\nloss = history.history['loss']\nval_loss = history.history['val_loss']","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:56:02.70905Z","iopub.execute_input":"2023-04-18T07:56:02.709475Z","iopub.status.idle":"2023-04-18T07:56:02.716131Z","shell.execute_reply.started":"2023-04-18T07:56:02.709434Z","shell.execute_reply":"2023-04-18T07:56:02.715034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualizing Accuracy and Loss","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize = (15,10))\n\nplt.subplot(2, 2, 1)\nplt.plot(accuracy, label = \"Training Accuracy\")\nplt.plot(val_accuracy, label = \"Validation Accuracy\")\nplt.ylim(0.4, 1)\nplt.legend(['Train', 'Validation'], loc = 'upper left')\nplt.title(\"Training vs Validation Accuracy\")\nplt.xlabel('epoch')\nplt.ylabel('accuracy')\n\n\nplt.subplot(2, 2, 2)\nplt.plot(loss, label = \"Training Loss\")\nplt.plot(val_loss, label = \"Validation Loss\")\nplt.legend(['Train', 'Validation'], loc = 'upper left')\nplt.title(\"Training vs Validation Loss\")\nplt.xlabel('epoch')\nplt.ylabel('loss')","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:56:02.717558Z","iopub.execute_input":"2023-04-18T07:56:02.71872Z","iopub.status.idle":"2023-04-18T07:56:03.156413Z","shell.execute_reply.started":"2023-04-18T07:56:02.71868Z","shell.execute_reply":"2023-04-18T07:56:03.155306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predictions\n\nIn order to evaluate the quality of the trained model, the outcome was predicted from the test (validation) data and compared with the observed value.","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.models import load_model\nmodel = load_model('/kaggle/working/mammography_pred_model.h5')","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:56:03.157872Z","iopub.execute_input":"2023-04-18T07:56:03.15896Z","iopub.status.idle":"2023-04-18T07:56:05.285904Z","shell.execute_reply.started":"2023-04-18T07:56:03.158916Z","shell.execute_reply":"2023-04-18T07:56:05.284831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = model.predict(validation_generator)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:56:05.287393Z","iopub.execute_input":"2023-04-18T07:56:05.287832Z","iopub.status.idle":"2023-04-18T07:56:14.53278Z","shell.execute_reply.started":"2023-04-18T07:56:05.287787Z","shell.execute_reply":"2023-04-18T07:56:14.531642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# prediction by the AI\npred","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:56:14.534406Z","iopub.execute_input":"2023-04-18T07:56:14.535389Z","iopub.status.idle":"2023-04-18T07:56:14.557553Z","shell.execute_reply.started":"2023-04-18T07:56:14.535343Z","shell.execute_reply":"2023-04-18T07:56:14.556524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = []\nfor prob in pred:\n    if prob >= 0.5:\n        y_pred.append(1)\n    else:\n        y_pred.append(0)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:56:14.559096Z","iopub.execute_input":"2023-04-18T07:56:14.55981Z","iopub.status.idle":"2023-04-18T07:56:14.567083Z","shell.execute_reply.started":"2023-04-18T07:56:14.559761Z","shell.execute_reply":"2023-04-18T07:56:14.565831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(y_pred)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:56:14.568988Z","iopub.execute_input":"2023-04-18T07:56:14.569864Z","iopub.status.idle":"2023-04-18T07:56:14.581083Z","shell.execute_reply.started":"2023-04-18T07:56:14.569824Z","shell.execute_reply":"2023-04-18T07:56:14.579783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.Series(y_pred).value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:56:14.582745Z","iopub.execute_input":"2023-04-18T07:56:14.583814Z","iopub.status.idle":"2023-04-18T07:56:14.598251Z","shell.execute_reply.started":"2023-04-18T07:56:14.583777Z","shell.execute_reply":"2023-04-18T07:56:14.596684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_true = validation_generator.classes","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:56:14.600708Z","iopub.execute_input":"2023-04-18T07:56:14.601155Z","iopub.status.idle":"2023-04-18T07:56:14.60705Z","shell.execute_reply.started":"2023-04-18T07:56:14.601112Z","shell.execute_reply":"2023-04-18T07:56:14.605489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(y_true)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:56:14.609201Z","iopub.execute_input":"2023-04-18T07:56:14.609747Z","iopub.status.idle":"2023-04-18T07:56:14.621526Z","shell.execute_reply.started":"2023-04-18T07:56:14.609707Z","shell.execute_reply":"2023-04-18T07:56:14.620071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Confusion Matrix\n\nConfusion matrix was created with the predicted and observed values. The matrix indicated almost correct predictions by the trained model except that there were two cases observed as false positive. False positive means that a case is actually negative but predicted as positive.","metadata":{}},{"cell_type":"code","source":"cm = confusion_matrix(y_true, y_pred)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:56:14.623845Z","iopub.execute_input":"2023-04-18T07:56:14.624391Z","iopub.status.idle":"2023-04-18T07:56:14.632772Z","shell.execute_reply.started":"2023-04-18T07:56:14.624345Z","shell.execute_reply":"2023-04-18T07:56:14.631367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the class names.\nclass_names = ['Normal', 'Cancer']\n\n# Create the heatmap with class names as tick labels.\nax = sns.heatmap(cm, annot = True, fmt = '.0f', cmap = \"Blues\", annot_kws = {\"size\": 16},\\\n           xticklabels = class_names, yticklabels = class_names)\n\n# Set the axis labels.\nax.set_xlabel(\"Prediction\")\nax.set_ylabel(\"Truth\")","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:56:14.63479Z","iopub.execute_input":"2023-04-18T07:56:14.63625Z","iopub.status.idle":"2023-04-18T07:56:14.927876Z","shell.execute_reply.started":"2023-04-18T07:56:14.636164Z","shell.execute_reply":"2023-04-18T07:56:14.926607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Classification Report\n\n","metadata":{}},{"cell_type":"code","source":"print(classification_report(y_true, y_pred))","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:56:14.929477Z","iopub.execute_input":"2023-04-18T07:56:14.930366Z","iopub.status.idle":"2023-04-18T07:56:14.944488Z","shell.execute_reply.started":"2023-04-18T07:56:14.930321Z","shell.execute_reply":"2023-04-18T07:56:14.942939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Analysing the Results\n\nIt is crucial in the medical field to analyze what kinds of cases were misclassified by the AI model, because medical misdiagnosis must be avoided as much as possible. Thus, it is necessary to identify false positive and false negative cases. Making a data frame and confusion table can visualize the results. As discussed above, false negative cases must be particularly avoided.","metadata":{}},{"cell_type":"code","source":"confusion = []\n\nfor i, j in zip(y_true, y_pred):\n  if i == 0 and j == 0:\n    confusion.append('TN')\n  elif i == 1 and j == 1:\n    confusion.append('TP')\n  elif i == 0 and j == 1:\n    confusion.append('FP')\n  else:\n    confusion.append('FN')","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:56:14.946661Z","iopub.execute_input":"2023-04-18T07:56:14.947102Z","iopub.status.idle":"2023-04-18T07:56:14.959447Z","shell.execute_reply.started":"2023-04-18T07:56:14.947058Z","shell.execute_reply":"2023-04-18T07:56:14.95816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(confusion)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:56:14.961351Z","iopub.execute_input":"2023-04-18T07:56:14.962061Z","iopub.status.idle":"2023-04-18T07:56:14.969923Z","shell.execute_reply.started":"2023-04-18T07:56:14.962016Z","shell.execute_reply":"2023-04-18T07:56:14.968365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"confusion_table = pd.DataFrame(data = confusion, columns = [\"Results\"])\nconfusion_table","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:56:14.97225Z","iopub.execute_input":"2023-04-18T07:56:14.973175Z","iopub.status.idle":"2023-04-18T07:56:14.989025Z","shell.execute_reply.started":"2023-04-18T07:56:14.973131Z","shell.execute_reply":"2023-04-18T07:56:14.987689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"confusion_table = pd.DataFrame({'Predicton':y_pred,\n                                'Truth': y_true,\n                                'Results': confusion})\nconfusion_table","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:56:14.990968Z","iopub.execute_input":"2023-04-18T07:56:14.991526Z","iopub.status.idle":"2023-04-18T07:56:15.008152Z","shell.execute_reply.started":"2023-04-18T07:56:14.991487Z","shell.execute_reply":"2023-04-18T07:56:15.006854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"confusion_table.Results == 'FP'","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:56:15.009918Z","iopub.execute_input":"2023-04-18T07:56:15.010426Z","iopub.status.idle":"2023-04-18T07:56:15.025115Z","shell.execute_reply.started":"2023-04-18T07:56:15.01039Z","shell.execute_reply":"2023-04-18T07:56:15.023558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# list of false positive images\nFPs = confusion_table[confusion_table['Results'] == 'FP']\nFPs","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:56:15.026869Z","iopub.execute_input":"2023-04-18T07:56:15.027646Z","iopub.status.idle":"2023-04-18T07:56:15.044194Z","shell.execute_reply.started":"2023-04-18T07:56:15.027591Z","shell.execute_reply":"2023-04-18T07:56:15.042899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FPs.index","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:56:15.046105Z","iopub.execute_input":"2023-04-18T07:56:15.046999Z","iopub.status.idle":"2023-04-18T07:56:15.055673Z","shell.execute_reply.started":"2023-04-18T07:56:15.046955Z","shell.execute_reply":"2023-04-18T07:56:15.054174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# list of false negative images\nFNs = confusion_table[confusion_table['Results'] == 'FN']\nFNs","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:56:15.059071Z","iopub.execute_input":"2023-04-18T07:56:15.060373Z","iopub.status.idle":"2023-04-18T07:56:15.078024Z","shell.execute_reply.started":"2023-04-18T07:56:15.060281Z","shell.execute_reply":"2023-04-18T07:56:15.076438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FNs.index","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:56:15.080432Z","iopub.execute_input":"2023-04-18T07:56:15.081493Z","iopub.status.idle":"2023-04-18T07:56:15.092251Z","shell.execute_reply.started":"2023-04-18T07:56:15.081444Z","shell.execute_reply":"2023-04-18T07:56:15.090856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Misclassficiation Cases\n\nIt is impoertant to pick up wrong cases judged by the AI and to analyze why the AI made wrong judgements for these images.","metadata":{}},{"cell_type":"code","source":"import glob\nval_images = glob.glob('/kaggle/working/val/*/*.png')","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:56:15.094802Z","iopub.execute_input":"2023-04-18T07:56:15.096133Z","iopub.status.idle":"2023-04-18T07:56:15.105607Z","shell.execute_reply.started":"2023-04-18T07:56:15.096085Z","shell.execute_reply":"2023-04-18T07:56:15.104335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# False positive imgages\nfig, axes = plt.subplots(nrows = 2, ncols = 5, figsize = (15, 10), subplot_kw = {'xticks':[], 'yticks':[]})\nfor i, ax in zip(FPs.index, axes.flat):\n    img = cv2.imread(val_images[i])\n    ax.imshow(img)\n    ax.set_title(\"False Positive Case\")\nfig.tight_layout()    \n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:56:15.107371Z","iopub.execute_input":"2023-04-18T07:56:15.107969Z","iopub.status.idle":"2023-04-18T07:56:16.214253Z","shell.execute_reply.started":"2023-04-18T07:56:15.107926Z","shell.execute_reply":"2023-04-18T07:56:16.212984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# False negative imgages\nfig, axes = plt.subplots(nrows = 2, ncols = 5, figsize = (15, 10), subplot_kw = {'xticks':[], 'yticks':[]})\nfor i, ax in zip(FNs.index, axes.flat):\n    img = cv2.imread(val_images[i])\n    ax.imshow(img)\n    ax.set_title(\"False Negative Case\")\nfig.tight_layout()    \n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:56:16.216374Z","iopub.execute_input":"2023-04-18T07:56:16.216752Z","iopub.status.idle":"2023-04-18T07:56:18.006975Z","shell.execute_reply.started":"2023-04-18T07:56:16.216718Z","shell.execute_reply":"2023-04-18T07:56:18.005843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Finetuning the Model (Unfreeying the Layers of the Model)","metadata":{}},{"cell_type":"code","source":"base_model = ResNet50V2(weights = 'imagenet', input_shape = (512, 512, 3), include_top = False)\n\nfor layer in base_model.layers:\n    layer.trainable = True # Change from False to True.\n    \nmodel = Sequential()\nmodel.add(base_model)\nmodel.add(GlobalAveragePooling2D())\nmodel.add(Dense(128, activation = 'relu'))\nmodel.add(Dropout(0.2))\nmodel.add(Dense(1, activation = 'sigmoid'))\n\nmodel.compile(optimizer = \"adam\", loss = 'binary_crossentropy', metrics = [\"accuracy\"])","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:56:18.008101Z","iopub.execute_input":"2023-04-18T07:56:18.00846Z","iopub.status.idle":"2023-04-18T07:56:20.241153Z","shell.execute_reply.started":"2023-04-18T07:56:18.008425Z","shell.execute_reply":"2023-04-18T07:56:20.239959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:56:20.242451Z","iopub.execute_input":"2023-04-18T07:56:20.24283Z","iopub.status.idle":"2023-04-18T07:56:20.290276Z","shell.execute_reply.started":"2023-04-18T07:56:20.242793Z","shell.execute_reply":"2023-04-18T07:56:20.289108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"callback = tf.keras.callbacks.EarlyStopping(monitor = \"val_loss\", mode = \"min\", patience = 4)\n\nhistory = model.fit(train_generator, validation_data = validation_generator, steps_per_epoch = 20, epochs = 15, callbacks = callback)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:56:20.291785Z","iopub.execute_input":"2023-04-18T07:56:20.29292Z","iopub.status.idle":"2023-04-18T08:12:27.009405Z","shell.execute_reply.started":"2023-04-18T07:56:20.292879Z","shell.execute_reply":"2023-04-18T08:12:27.008296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This time we use validation data to calculate the final accuracy.\nfinal_accuracy = model.evaluate_generator(validation_generator)[1]","metadata":{"execution":{"iopub.status.busy":"2023-04-18T08:12:27.011277Z","iopub.execute_input":"2023-04-18T08:12:27.011804Z","iopub.status.idle":"2023-04-18T08:12:37.4796Z","shell.execute_reply.started":"2023-04-18T08:12:27.011762Z","shell.execute_reply":"2023-04-18T08:12:37.478365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_accuracy","metadata":{"execution":{"iopub.status.busy":"2023-04-18T08:12:37.481368Z","iopub.execute_input":"2023-04-18T08:12:37.482951Z","iopub.status.idle":"2023-04-18T08:12:37.49044Z","shell.execute_reply.started":"2023-04-18T08:12:37.482904Z","shell.execute_reply":"2023-04-18T08:12:37.489248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Metrics","metadata":{}},{"cell_type":"code","source":"accuracy = history.history['accuracy']\nval_accuracy  = history.history['val_accuracy']\n\nloss = history.history['loss']\nval_loss = history.history['val_loss']","metadata":{"execution":{"iopub.status.busy":"2023-04-18T08:12:37.492366Z","iopub.execute_input":"2023-04-18T08:12:37.492894Z","iopub.status.idle":"2023-04-18T08:12:37.501125Z","shell.execute_reply.started":"2023-04-18T08:12:37.492853Z","shell.execute_reply":"2023-04-18T08:12:37.499503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualizing Accuracy and Loss","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize = (15,10))\n\nplt.subplot(2, 2, 1)\nplt.plot(accuracy, label = \"Training Accuracy\")\nplt.plot(val_accuracy, label = \"Validation Accuracy\")\nplt.ylim(0.4, 1)\nplt.legend(['Train', 'Validation'], loc = 'upper left')\nplt.title(\"Training vs Validation Accuracy\")\nplt.xlabel('epoch')\nplt.ylabel('accuracy')\n\n\nplt.subplot(2, 2, 2)\nplt.plot(loss, label = \"Training Loss\")\nplt.plot(val_loss, label = \"Validation Loss\")\nplt.legend(['Train', 'Validation'], loc = 'upper left')\nplt.title(\"Training vs Validation Loss\")\nplt.xlabel('epoch')\nplt.ylabel('loss')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-18T08:12:37.503385Z","iopub.execute_input":"2023-04-18T08:12:37.503711Z","iopub.status.idle":"2023-04-18T08:12:37.925755Z","shell.execute_reply.started":"2023-04-18T08:12:37.503682Z","shell.execute_reply":"2023-04-18T08:12:37.924498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Finetuning is worse than Transfer Learning, because the number of data images is small.","metadata":{}},{"cell_type":"markdown","source":"# Save the Model","metadata":{}},{"cell_type":"code","source":"model.save('mammography_pred_model_finetuning.h5')","metadata":{"execution":{"iopub.status.busy":"2023-04-18T08:12:37.927602Z","iopub.execute_input":"2023-04-18T08:12:37.928634Z","iopub.status.idle":"2023-04-18T08:12:38.915388Z","shell.execute_reply.started":"2023-04-18T08:12:37.928566Z","shell.execute_reply":"2023-04-18T08:12:38.913907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Conclusion\n\nThis AI model may be useful for general physicians who occasionally see female patients for the screening purpose. Further improvement is required to prevent misdiagnosis and unnecessary treatment.","metadata":{}}]}