{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":39272,"databundleVersionId":4629629,"sourceType":"competition"},{"sourceId":4619805,"sourceType":"datasetVersion","datasetId":2688675}],"dockerImageVersionId":30587,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-12-11T20:23:57.413505Z","iopub.execute_input":"2023-12-11T20:23:57.414711Z","iopub.status.idle":"2023-12-11T20:23:57.897974Z","shell.execute_reply.started":"2023-12-11T20:23:57.414659Z","shell.execute_reply":"2023-12-11T20:23:57.896516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Importing libraries\n# TensorFlow libraries\nimport tensorflow as tf\nfrom tensorflow.keras.applications.resnet_v2 import ResNet50V2\nfrom tensorflow.keras.optimizers import RMSprop\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\nfrom tensorflow.keras.layers import Dense, GlobalAveragePooling2D, Dropout\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.optimizers import Adam\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report, confusion_matrix\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport cv2\nimport os\n\nimport glob\nfrom glob import glob\nimport pydicom\n","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:23:57.899873Z","iopub.execute_input":"2023-12-11T20:23:57.900405Z","iopub.status.idle":"2023-12-11T20:24:14.373701Z","shell.execute_reply.started":"2023-12-11T20:23:57.900368Z","shell.execute_reply":"2023-12-11T20:24:14.372627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install pydicom\n","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:24:14.375629Z","iopub.execute_input":"2023-12-11T20:24:14.376975Z","iopub.status.idle":"2023-12-11T20:24:27.71056Z","shell.execute_reply.started":"2023-12-11T20:24:14.37692Z","shell.execute_reply":"2023-12-11T20:24:27.708926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dicom_file_path = '/kaggle/input/rsna-breast-cancer-detection/train_images/10006/1459541791.dcm'\ndicom_data = pydicom.dcmread(dicom_file_path)","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:24:27.714224Z","iopub.execute_input":"2023-12-11T20:24:27.714687Z","iopub.status.idle":"2023-12-11T20:24:27.892342Z","shell.execute_reply.started":"2023-12-11T20:24:27.714645Z","shell.execute_reply":"2023-12-11T20:24:27.891164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#afficher le contenu d'un fichier dicom\ndicom_data = pydicom.dcmread(dicom_file_path)\nprint(dicom_data)","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:24:27.893477Z","iopub.execute_input":"2023-12-11T20:24:27.893895Z","iopub.status.idle":"2023-12-11T20:24:27.910631Z","shell.execute_reply.started":"2023-12-11T20:24:27.893854Z","shell.execute_reply":"2023-12-11T20:24:27.909261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Exploring Dataset\ninput_path = \"/kaggle/input/rsna-breast-cancer-detection\"\n\nos.listdir(input_path)","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:24:27.912344Z","iopub.execute_input":"2023-12-11T20:24:27.912731Z","iopub.status.idle":"2023-12-11T20:24:27.922689Z","shell.execute_reply.started":"2023-12-11T20:24:27.912698Z","shell.execute_reply":"2023-12-11T20:24:27.921686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Data\ndf_train = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\ndf_test = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:24:27.924319Z","iopub.execute_input":"2023-12-11T20:24:27.925343Z","iopub.status.idle":"2023-12-11T20:24:28.065904Z","shell.execute_reply.started":"2023-12-11T20:24:27.925303Z","shell.execute_reply":"2023-12-11T20:24:28.064837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head()\n","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:24:28.067382Z","iopub.execute_input":"2023-12-11T20:24:28.0685Z","iopub.status.idle":"2023-12-11T20:24:28.101193Z","shell.execute_reply.started":"2023-12-11T20:24:28.06846Z","shell.execute_reply":"2023-12-11T20:24:28.099866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Visualiser quelques images dicom","metadata":{}},{"cell_type":"code","source":"def load_dicom_images(directory):\n    dicom_images = []\n    for root, dirs, files in os.walk(directory):\n        for file in files:\n            dicom_images.append(pydicom.dcmread(os.path.join(root, file)))\n    return dicom_images\n\ndef visualize_dicom_images(images, num_samples):\n    sample_images = images[:num_samples]\n    for i, image in enumerate(sample_images):\n        plt.subplot(1, num_samples, i + 1)\n        plt.imshow(image.pixel_array)\n        plt.title(f\"Image {i + 1}\")\n        plt.axis('off')\n    plt.show()\n    \ntrain_dcm_folder = \"/kaggle/input/rsna-breast-cancer-detection/train_images/10006\"\ndcm_img=load_dicom_images(train_dcm_folder)\nvisualize_dicom_images(dcm_img,4)","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:24:28.103106Z","iopub.execute_input":"2023-12-11T20:24:28.103632Z","iopub.status.idle":"2023-12-11T20:24:42.811396Z","shell.execute_reply.started":"2023-12-11T20:24:28.103587Z","shell.execute_reply":"2023-12-11T20:24:42.810205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Afficher la taille de chaque ensemble\nprint(f\"Taille de l'ensemble d'entraînement : {len(df_train)}\")","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:24:42.816246Z","iopub.execute_input":"2023-12-11T20:24:42.816923Z","iopub.status.idle":"2023-12-11T20:24:42.822585Z","shell.execute_reply.started":"2023-12-11T20:24:42.81688Z","shell.execute_reply":"2023-12-11T20:24:42.821202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# the number of patients with malignant cancer\nlen(df_train[df_train['cancer'] == 1])","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:24:42.824245Z","iopub.execute_input":"2023-12-11T20:24:42.825044Z","iopub.status.idle":"2023-12-11T20:24:42.850196Z","shell.execute_reply.started":"2023-12-11T20:24:42.824996Z","shell.execute_reply":"2023-12-11T20:24:42.849292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# the number of patients without malignant cancer\nlen(df_train[df_train['cancer'] == 0])","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:24:42.851326Z","iopub.execute_input":"2023-12-11T20:24:42.8521Z","iopub.status.idle":"2023-12-11T20:24:42.865524Z","shell.execute_reply.started":"2023-12-11T20:24:42.852064Z","shell.execute_reply":"2023-12-11T20:24:42.864623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# the number of patient who took biopsy\nlen(df_train[df_train['biopsy'] == 1])","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:24:42.867103Z","iopub.execute_input":"2023-12-11T20:24:42.867765Z","iopub.status.idle":"2023-12-11T20:24:42.87586Z","shell.execute_reply.started":"2023-12-11T20:24:42.867717Z","shell.execute_reply":"2023-12-11T20:24:42.874934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# the number of patients whose malignant cancer is invasive\nlen(df_train[df_train['invasive'] == 1])","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:24:42.877428Z","iopub.execute_input":"2023-12-11T20:24:42.878041Z","iopub.status.idle":"2023-12-11T20:24:42.885947Z","shell.execute_reply.started":"2023-12-11T20:24:42.878006Z","shell.execute_reply":"2023-12-11T20:24:42.885078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.head()\n","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:24:42.887467Z","iopub.execute_input":"2023-12-11T20:24:42.888127Z","iopub.status.idle":"2023-12-11T20:24:42.903078Z","shell.execute_reply.started":"2023-12-11T20:24:42.888085Z","shell.execute_reply":"2023-12-11T20:24:42.901809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Afficher la taille de test\nprint(f\"Taille de l'ensemble de test : {len(df_test)}\")","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:24:42.9046Z","iopub.execute_input":"2023-12-11T20:24:42.90499Z","iopub.status.idle":"2023-12-11T20:24:42.913382Z","shell.execute_reply.started":"2023-12-11T20:24:42.904955Z","shell.execute_reply":"2023-12-11T20:24:42.912209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#afficher le nombre des differents patients \n#l'age minimal et l'age maximal dans la dataset\n\nnum_patients = df_train['patient_id'].nunique()\nmin_patient_age = df_train['age'].min()\nmax_patient_age = df_train['age'].max()\n\n\nprint(f\"There are {num_patients} different patients in the train set.\\n\")\nprint(f\"The youngest patient is {int(min_patient_age)} years old.\")\nprint(f\"The oldest patient is {int(max_patient_age)} years old.\\n\")\n","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:24:42.915096Z","iopub.execute_input":"2023-12-11T20:24:42.915672Z","iopub.status.idle":"2023-12-11T20:24:42.933506Z","shell.execute_reply.started":"2023-12-11T20:24:42.915635Z","shell.execute_reply":"2023-12-11T20:24:42.932261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#\"Nombre de patients atteints de cancer\"\ncancer_per_patient = df_train.groupby(\"patient_id\")[\"cancer\"].max().values\nn_negative = (cancer_per_patient == 0).sum()\nn_positive = (cancer_per_patient == 1).sum()\n\nfig, ax = plt.subplots()\nbars = ax.bar([\"No cancer\", \"Cancer\"], [n_negative, n_positive])\nax.set(xlabel=\"\", ylabel=\"Nombre\", title=\"Nombre de patients avec cancer\")\nfor bar, count in zip(bars, [n_negative, n_positive]):\n    height = bar.get_height()\n    ax.text(bar.get_x() + bar.get_width() / 2, height, count, ha=\"center\", va=\"bottom\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:24:42.935206Z","iopub.execute_input":"2023-12-11T20:24:42.935741Z","iopub.status.idle":"2023-12-11T20:24:43.174118Z","shell.execute_reply.started":"2023-12-11T20:24:42.935707Z","shell.execute_reply":"2023-12-11T20:24:43.172837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Distribution d'ages\nplt.figure(figsize=(10, 6))\nsns.histplot(df_train.age, bins=50, kde=True, color='skyblue')\nplt.title(\"Distribution  d'ages\")\nplt.xlabel('Age')\nplt.ylabel('Nombre de patients')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:24:43.175956Z","iopub.execute_input":"2023-12-11T20:24:43.176339Z","iopub.status.idle":"2023-12-11T20:24:43.913252Z","shell.execute_reply.started":"2023-12-11T20:24:43.176306Z","shell.execute_reply":"2023-12-11T20:24:43.912106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ages = df_train.groupby('patient_id')['age'].apply(lambda x: x.unique()[0])\ncancer_ages = df_train[df_train['cancer'] == 1].groupby('patient_id')['age'].apply(lambda x: x.unique()[0])\nno_cancer_ages = df_train[df_train['cancer'] == 0].groupby('patient_id')['age'].apply(lambda x: x.unique()[0])\n\nplt.figure(figsize=(7, 7))\n\nsns.histplot(cancer_ages, bins=51, color='red', kde=True)\nsns.histplot(no_cancer_ages, bins=63, color='skyblue', kde=True)\nplt.title(\"Patients avec et sans cancer\", fontsize=14)\nplt.xlabel(\"Age\", fontsize=12)\nplt.ylabel(\"nombre de patients\", fontsize=12)\nplt.xticks(fontsize=10)\nplt.yticks(fontsize=10)\nplt.xlim(33, 89)\nplt.legend([\"Cancer\", \"No cancer\"], fontsize=10)\n\nplt.suptitle(\"Distribution d'age selon le cas du patient\", fontsize=16)\nplt.tight_layout()\nplt.show()\n\n","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:24:43.914567Z","iopub.execute_input":"2023-12-11T20:24:43.914919Z","iopub.status.idle":"2023-12-11T20:24:46.250331Z","shell.execute_reply.started":"2023-12-11T20:24:43.914889Z","shell.execute_reply":"2023-12-11T20:24:46.249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Distribution du cancer des seins chez les Patients\nlabels = ['Negatif', 'Positif']\nsizes = [n_negative / num_patients, n_positive / num_patients]\ncolors = ['#ff9999', '#66b3ff']  # rouge pour negatif, bleu pour positif\nexplode = (0.1, 0)  \n\n# afficher le pie chart\nfig1, ax1 = plt.subplots()\nax1.pie(sizes, explode=explode, labels=labels, colors=colors, autopct='%1.1f%%', startangle=90)\n\nax1.axis('equal')\n\n# Titre\nplt.title('Breast Cancer Distribution in Patients')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:24:46.252043Z","iopub.execute_input":"2023-12-11T20:24:46.252571Z","iopub.status.idle":"2023-12-11T20:24:46.389725Z","shell.execute_reply.started":"2023-12-11T20:24:46.25252Z","shell.execute_reply":"2023-12-11T20:24:46.387974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# jusqu'à 3000 patient a pris biopsy et malignant cancer est trouvé chez quelque uns d'eux.\ndata = pd.DataFrame(np.concatenate([['Biopsy'] * len(df_train[df_train['biopsy'] == 1]) , ['Malignant Cancer'] *  len(df_train[df_train['cancer'] == 1]), ['Invasive Cancer'] *  len(df_train[(df_train['cancer'] == 1) & (df_train['invasive'] == 1)])]), columns = [\"class\"])\n\nsns.countplot(x = 'class', data = data)","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:24:46.392162Z","iopub.execute_input":"2023-12-11T20:24:46.392748Z","iopub.status.idle":"2023-12-11T20:24:46.656986Z","shell.execute_reply.started":"2023-12-11T20:24:46.392697Z","shell.execute_reply":"2023-12-11T20:24:46.655794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# le nombre des cas not-malignant cancer après biopsy\nlen(df_train[(df_train['biopsy'] == 1) & (df_train['cancer'] == 0)])","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:24:46.658693Z","iopub.execute_input":"2023-12-11T20:24:46.659071Z","iopub.status.idle":"2023-12-11T20:24:46.669915Z","shell.execute_reply.started":"2023-12-11T20:24:46.659039Z","shell.execute_reply":"2023-12-11T20:24:46.668831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# le nombre des cas malignant cancer après biopsy\nlen(df_train[(df_train['biopsy'] == 1) & (df_train['cancer'] == 1)])","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:24:46.671451Z","iopub.execute_input":"2023-12-11T20:24:46.671943Z","iopub.status.idle":"2023-12-11T20:24:46.682919Z","shell.execute_reply.started":"2023-12-11T20:24:46.671898Z","shell.execute_reply":"2023-12-11T20:24:46.681641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 60% biopsy resulted in not-malignanct cancer.\ndata = pd.DataFrame(np.concatenate([['Biopsy but Not Malignant'] * len(df_train[(df_train['biopsy'] == 1) & (df_train['cancer'] == 0)]) , ['Malignant Cancer'] *  len(df_train[df_train['cancer'] == 1]), ['Invasive Cancer'] *  len(df_train[(df_train['cancer'] == 1) & (df_train['invasive'] == 1)])]), columns = [\"class\"])\n\nsns.countplot(x = 'class', data = data)","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:24:46.684783Z","iopub.execute_input":"2023-12-11T20:24:46.685466Z","iopub.status.idle":"2023-12-11T20:24:46.923715Z","shell.execute_reply.started":"2023-12-11T20:24:46.685419Z","shell.execute_reply":"2023-12-11T20:24:46.922513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Matrice de Correlation","metadata":{}},{"cell_type":"code","source":"# affiche la matrice de correlation\nnumeric_columns = df_train.select_dtypes(include=['int64', 'float64'])\n\ncorrelation_matrix = numeric_columns.corr()\nplt.figure(figsize=(10,7))\n\nsns.heatmap(correlation_matrix, annot=True, cmap='coolwarm', fmt=\".2f\")\nplt.title('Matrice de Correlation ')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:24:46.925442Z","iopub.execute_input":"2023-12-11T20:24:46.925851Z","iopub.status.idle":"2023-12-11T20:24:47.591987Z","shell.execute_reply.started":"2023-12-11T20:24:46.925815Z","shell.execute_reply":"2023-12-11T20:24:47.59063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Pretraintement des donnees","metadata":{}},{"cell_type":"code","source":"# verifié les valeurs nulles et manquantes dans le DataFrame\ndf_train.isnull().sum()\ndf_train.isna().sum()\n","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:24:47.593806Z","iopub.execute_input":"2023-12-11T20:24:47.59557Z","iopub.status.idle":"2023-12-11T20:24:47.626304Z","shell.execute_reply.started":"2023-12-11T20:24:47.595509Z","shell.execute_reply":"2023-12-11T20:24:47.625071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Selectionner uniquement les columns numeriques du DataFrame\nnumeric_columns = df_train.select_dtypes(include=['int64', 'float64'])\n\n# Creer la matrice de correlation \ncorr_matrix = numeric_columns.corr()\n\n# \"Classer les corrélations entre la colonne 'cancer' et les autres colonnes numériques.\"\n\ncancer_correlations = corr_matrix['cancer'].sort_values(ascending=False)\n\nprint(cancer_correlations)","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:24:47.634471Z","iopub.execute_input":"2023-12-11T20:24:47.63499Z","iopub.status.idle":"2023-12-11T20:24:47.666072Z","shell.execute_reply.started":"2023-12-11T20:24:47.634947Z","shell.execute_reply":"2023-12-11T20:24:47.66499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Creer une dataset equilibrée","metadata":{}},{"cell_type":"code","source":"# \"Le nombre de cas positifs (malignes) et négatifs (non malignes) devrait être le même \n#  pour créer un dataset équilibré.\"\ndf_train = df_train.groupby(['cancer']).apply(lambda x: x.sample(1158, replace = True)\n                                                      ).reset_index(drop = True)\nprint('New Data Size:', df_train.shape[0])","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:24:47.667417Z","iopub.execute_input":"2023-12-11T20:24:47.668533Z","iopub.status.idle":"2023-12-11T20:24:47.690635Z","shell.execute_reply.started":"2023-12-11T20:24:47.668494Z","shell.execute_reply":"2023-12-11T20:24:47.689685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"RSNA_512_path = '/kaggle/input/rsna-breast-cancer-512-pngs'\n#\"Créer le chemin d'accès à chaque image.\"\nfor i in range(len(df_train)):\n    df_train.loc[i, 'path'] = os.path.join(RSNA_512_path + '/' + str(df_train.loc[i, 'patient_id']) + '_' + str(df_train.loc[i, 'image_id']) + '.png')\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:24:47.692288Z","iopub.execute_input":"2023-12-11T20:24:47.692986Z","iopub.status.idle":"2023-12-11T20:24:48.464869Z","shell.execute_reply.started":"2023-12-11T20:24:47.69295Z","shell.execute_reply":"2023-12-11T20:24:48.463664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# exemple image\nimg = cv2.imread(df_train.loc[10, 'path'])\nplt.imshow(img, cmap = 'gray')","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:24:48.46629Z","iopub.execute_input":"2023-12-11T20:24:48.466655Z","iopub.status.idle":"2023-12-11T20:24:48.848696Z","shell.execute_reply.started":"2023-12-11T20:24:48.466622Z","shell.execute_reply":"2023-12-11T20:24:48.847491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img.shape\n","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:24:48.850576Z","iopub.execute_input":"2023-12-11T20:24:48.851359Z","iopub.status.idle":"2023-12-11T20:24:48.859279Z","shell.execute_reply.started":"2023-12-11T20:24:48.851314Z","shell.execute_reply":"2023-12-11T20:24:48.858248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:24:48.860722Z","iopub.execute_input":"2023-12-11T20:24:48.861831Z","iopub.status.idle":"2023-12-11T20:24:48.869448Z","shell.execute_reply.started":"2023-12-11T20:24:48.86179Z","shell.execute_reply":"2023-12-11T20:24:48.868315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Diviser les données entre Training Test et Validation","metadata":{}},{"cell_type":"code","source":"# Split into train and temp (combined validation and test) sets\ntrain_df, temp_df = train_test_split(df_train, \n                                      test_size=0.4, \n                                      random_state=2018, \n                                      stratify=df_train['cancer'])\n\n# Further split temp_df into validation and test sets\nvalidation_df, test_df = train_test_split(temp_df, \n                                           test_size=0.5, \n                                           random_state=2018, \n                                           stratify=temp_df['cancer'])\n\n# afficher la taille de chaque ensemble\nprint('train:', train_df.shape[0])\nprint('validation:', validation_df.shape[0])\nprint('test:', test_df.shape[0])\nprint('train', train_df['cancer'].value_counts())\nprint('test', test_df['cancer'].value_counts())\nprint('validation', validation_df['cancer'].value_counts())\n","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:24:48.871113Z","iopub.execute_input":"2023-12-11T20:24:48.871559Z","iopub.status.idle":"2023-12-11T20:24:48.893305Z","shell.execute_reply.started":"2023-12-11T20:24:48.871527Z","shell.execute_reply":"2023-12-11T20:24:48.892104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:24:48.89484Z","iopub.execute_input":"2023-12-11T20:24:48.895212Z","iopub.status.idle":"2023-12-11T20:24:48.914258Z","shell.execute_reply.started":"2023-12-11T20:24:48.895179Z","shell.execute_reply":"2023-12-11T20:24:48.913147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train_df[train_df['cancer'] == 0])","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:24:48.916077Z","iopub.execute_input":"2023-12-11T20:24:48.916809Z","iopub.status.idle":"2023-12-11T20:24:48.928244Z","shell.execute_reply.started":"2023-12-11T20:24:48.916742Z","shell.execute_reply":"2023-12-11T20:24:48.927369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train_df[train_df['cancer'] == 1])","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:24:48.929881Z","iopub.execute_input":"2023-12-11T20:24:48.930515Z","iopub.status.idle":"2023-12-11T20:24:48.943742Z","shell.execute_reply.started":"2023-12-11T20:24:48.930483Z","shell.execute_reply":"2023-12-11T20:24:48.942394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(test_df[test_df['cancer'] == 0])","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:24:48.945424Z","iopub.execute_input":"2023-12-11T20:24:48.946015Z","iopub.status.idle":"2023-12-11T20:24:48.958475Z","shell.execute_reply.started":"2023-12-11T20:24:48.945962Z","shell.execute_reply":"2023-12-11T20:24:48.957236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(test_df[test_df['cancer'] == 1])","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:24:48.960194Z","iopub.execute_input":"2023-12-11T20:24:48.960711Z","iopub.status.idle":"2023-12-11T20:24:48.971917Z","shell.execute_reply.started":"2023-12-11T20:24:48.960666Z","shell.execute_reply":"2023-12-11T20:24:48.970815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(validation_df[validation_df['cancer'] == 0])","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:24:48.973221Z","iopub.execute_input":"2023-12-11T20:24:48.973681Z","iopub.status.idle":"2023-12-11T20:24:48.985219Z","shell.execute_reply.started":"2023-12-11T20:24:48.973642Z","shell.execute_reply":"2023-12-11T20:24:48.984341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(validation_df[validation_df['cancer'] == 1])","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:24:48.987005Z","iopub.execute_input":"2023-12-11T20:24:48.987769Z","iopub.status.idle":"2023-12-11T20:24:48.999574Z","shell.execute_reply.started":"2023-12-11T20:24:48.987705Z","shell.execute_reply":"2023-12-11T20:24:48.998646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# \"Afficher les premières lignes de l'ensemble d'entraînement.\"\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:24:49.001152Z","iopub.execute_input":"2023-12-11T20:24:49.002014Z","iopub.status.idle":"2023-12-11T20:24:49.025093Z","shell.execute_reply.started":"2023-12-11T20:24:49.001977Z","shell.execute_reply":"2023-12-11T20:24:49.023933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# \"Afficher les premières lignes de l'ensemble de test.\"\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:24:49.026661Z","iopub.execute_input":"2023-12-11T20:24:49.027785Z","iopub.status.idle":"2023-12-11T20:24:49.049461Z","shell.execute_reply.started":"2023-12-11T20:24:49.027673Z","shell.execute_reply":"2023-12-11T20:24:49.048531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# \"Afficher les premières lignes de l'ensemble de validation.\"\nvalidation_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:24:49.050875Z","iopub.execute_input":"2023-12-11T20:24:49.051823Z","iopub.status.idle":"2023-12-11T20:24:49.076035Z","shell.execute_reply.started":"2023-12-11T20:24:49.051775Z","shell.execute_reply":"2023-12-11T20:24:49.074514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Creer un dossier pour chaque ensemble","metadata":{}},{"cell_type":"code","source":"import shutil\n\ntrain_df_normal = train_df[train_df['cancer'] == 0].reset_index(drop = True)\n\n# Definir le dossier destination\ndestination_dir = '/kaggle/working/train'\ndestination_dir_sub = '/kaggle/working/train/normal'\n\nif not os.path.exists(destination_dir):\n    os.makedirs(destination_dir)\n\nif not os.path.exists(destination_dir_sub):\n    os.makedirs(destination_dir_sub)   \n    \n# copier les images vers le dossier destination\nfor path in train_df_normal['path']:\n    shutil.copy2(path, destination_dir_sub)","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:24:49.07778Z","iopub.execute_input":"2023-12-11T20:24:49.07816Z","iopub.status.idle":"2023-12-11T20:24:58.042332Z","shell.execute_reply.started":"2023-12-11T20:24:49.07813Z","shell.execute_reply":"2023-12-11T20:24:58.041096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Definir le dossier destination\ntrain_df_cancer = train_df[train_df['cancer'] == 1].reset_index(drop = True)\n\ndestination_dir = '/kaggle/working/train'\ndestination_dir_sub = '/kaggle/working/train/cancer'\n\nif not os.path.exists(destination_dir):\n    os.makedirs(destination_dir)\n\nif not os.path.exists(destination_dir_sub):\n    os.makedirs(destination_dir_sub)   \n    \n# copier les images vers le dossier destination\nfor path in train_df_cancer['path']:\n    shutil.copy2(path, destination_dir_sub)","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:24:58.043874Z","iopub.execute_input":"2023-12-11T20:24:58.044217Z","iopub.status.idle":"2023-12-11T20:25:04.974497Z","shell.execute_reply.started":"2023-12-11T20:24:58.044189Z","shell.execute_reply":"2023-12-11T20:25:04.973201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df_normal = test_df[test_df['cancer'] == 0].reset_index(drop = True)\n\n# copier les images vers le dossier destination\ndestination_dir = '/kaggle/working/test'\ndestination_dir_sub = '/kaggle/working/test/normal'\n\nif not os.path.exists(destination_dir):\n    os.makedirs(destination_dir)\n\nif not os.path.exists(destination_dir_sub):\n    os.makedirs(destination_dir_sub)   \n    \n# copier les images vers le dossier destination\nfor path in test_df_normal['path']:\n    shutil.copy2(path, destination_dir_sub)","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:25:04.976221Z","iopub.execute_input":"2023-12-11T20:25:04.976783Z","iopub.status.idle":"2023-12-11T20:25:07.642483Z","shell.execute_reply.started":"2023-12-11T20:25:04.976716Z","shell.execute_reply":"2023-12-11T20:25:07.641256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df_cancer = test_df[test_df['cancer'] == 1].reset_index(drop = True)\n\n# copier les images vers le dossier destination\ndestination_dir = '/kaggle/working/test'\ndestination_dir_sub = '/kaggle/working/test/cancer'\n\nif not os.path.exists(destination_dir):\n    os.makedirs(destination_dir)\n\nif not os.path.exists(destination_dir_sub):\n    os.makedirs(destination_dir_sub)   \n    \n# copier les images vers le dossier destination\nfor path in test_df_cancer['path']:\n    shutil.copy2(path, destination_dir_sub)","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:25:07.644555Z","iopub.execute_input":"2023-12-11T20:25:07.645159Z","iopub.status.idle":"2023-12-11T20:25:09.213642Z","shell.execute_reply.started":"2023-12-11T20:25:07.645104Z","shell.execute_reply":"2023-12-11T20:25:09.212253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_df_normal = validation_df[validation_df['cancer'] == 0].reset_index(drop = True)\n\n# copier les images vers le dossier destination\ndestination_dir = '/kaggle/working/validation'\ndestination_dir_sub = '/kaggle/working/validation/normal'\n\nif not os.path.exists(destination_dir):\n    os.makedirs(destination_dir)\n\nif not os.path.exists(destination_dir_sub):\n    os.makedirs(destination_dir_sub)   \n    \n# copier les images vers le dossier destination\nfor path in val_df_normal['path']:\n    shutil.copy2(path, destination_dir_sub)","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:25:09.21534Z","iopub.execute_input":"2023-12-11T20:25:09.215865Z","iopub.status.idle":"2023-12-11T20:25:11.929645Z","shell.execute_reply.started":"2023-12-11T20:25:09.215815Z","shell.execute_reply":"2023-12-11T20:25:11.928173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_df_cancer = validation_df[validation_df['cancer'] == 1].reset_index(drop = True)\n\n# copier les images vers le dossier destination\ndestination_dir = '/kaggle/working/validation'\ndestination_dir_sub = '/kaggle/working/validation/cancer'\n\nif not os.path.exists(destination_dir):\n    os.makedirs(destination_dir)\n\nif not os.path.exists(destination_dir_sub):\n    os.makedirs(destination_dir_sub)   \n    \n# copier les images vers le dossier destination\nfor path in val_df_cancer['path']:\n    shutil.copy2(path, destination_dir_sub)","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:25:11.93137Z","iopub.execute_input":"2023-12-11T20:25:11.93228Z","iopub.status.idle":"2023-12-11T20:25:13.086508Z","shell.execute_reply.started":"2023-12-11T20:25:11.932236Z","shell.execute_reply":"2023-12-11T20:25:13.085004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import glob\nnormal_train_images = glob.glob('/kaggle/working/train/normal/*.png')\ncancer_train_images = glob.glob('/kaggle/working/train/cancer/*.png')","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:25:13.088608Z","iopub.execute_input":"2023-12-11T20:25:13.089126Z","iopub.status.idle":"2023-12-11T20:25:13.101193Z","shell.execute_reply.started":"2023-12-11T20:25:13.089059Z","shell.execute_reply":"2023-12-11T20:25:13.100007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# afficher quelques images normal du training set\nfig, axes = plt.subplots(nrows = 1, ncols = 5, figsize = (15, 10), subplot_kw = {'xticks':[], 'yticks':[]})\nfor i, ax in enumerate(axes.flat):\n    img = cv2.imread(normal_train_images[i])\n    ax.imshow(img)\n    ax.set_title('Normal')\n    \nfig.tight_layout()    \n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:25:13.103034Z","iopub.execute_input":"2023-12-11T20:25:13.103538Z","iopub.status.idle":"2023-12-11T20:25:13.791707Z","shell.execute_reply.started":"2023-12-11T20:25:13.103491Z","shell.execute_reply":"2023-12-11T20:25:13.790448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# afficher quelques images cancer du training set\nfig, axes = plt.subplots(nrows = 1, ncols = 5, figsize = (15, 10), subplot_kw = {'xticks':[], 'yticks':[]})\nfor i, ax in enumerate(axes.flat):\n    img = cv2.imread(cancer_train_images[i])\n    ax.imshow(img)\n    ax.set_title('Cancer')\n    \nplt.show()\n    \n","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:25:13.793098Z","iopub.execute_input":"2023-12-11T20:25:13.793468Z","iopub.status.idle":"2023-12-11T20:25:14.355608Z","shell.execute_reply.started":"2023-12-11T20:25:13.793437Z","shell.execute_reply":"2023-12-11T20:25:14.354657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Create Image data Generators\n******\nThe dataset has already been divided into train , test and validation datasets, and each dataset includes normal and cancer image files. Thus, image data generators were easily created.","metadata":{}},{"cell_type":"code","source":"train_datagen = ImageDataGenerator(rescale = 1./255.,\n                                   zoom_range = 0.2)\nval_datagen = ImageDataGenerator(rescale = 1./255.,)\ntest_datagen = ImageDataGenerator(rescale = 1./255.,)\n","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:25:14.357144Z","iopub.execute_input":"2023-12-11T20:25:14.357803Z","iopub.status.idle":"2023-12-11T20:25:14.363537Z","shell.execute_reply.started":"2023-12-11T20:25:14.357765Z","shell.execute_reply":"2023-12-11T20:25:14.362161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_path = '/kaggle/working/train'\nval_path = '/kaggle/working/validation'\ntest_path ='/kaggle/working/test'\n\ntrain_generator = train_datagen.flow_from_directory(\n    train_path,\n    target_size = (512, 512),\n    batch_size = 32,\n    class_mode = 'binary'\n)\nvalidation_generator = val_datagen.flow_from_directory(\n        val_path,\n        target_size = (512, 512),\n        batch_size = 16,\n        class_mode = 'binary'\n)\ntest_generator = test_datagen.flow_from_directory(\n        test_path,\n        target_size = (512, 512),\n        batch_size = 16,\n        class_mode = 'binary'\n)","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:25:14.365573Z","iopub.execute_input":"2023-12-11T20:25:14.36601Z","iopub.status.idle":"2023-12-11T20:25:14.437114Z","shell.execute_reply.started":"2023-12-11T20:25:14.365967Z","shell.execute_reply":"2023-12-11T20:25:14.436042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Define the Model (Transfer Learning)****\n\n**Here, ResNet50V2 was employed as the base model for transfer learning. It would be also possible to conduct fine tuning with the model. The choice of hyperparameters depends on the type of task by machine learning. Sigmoid, adam, and binary cross entropy were selected for the final activation function, optimizer, and loss function, respectively.","metadata":{}},{"cell_type":"code","source":"base_model = ResNet50V2(weights = 'imagenet', input_shape = (512, 512, 3), include_top = False)\n\nfor layer in base_model.layers:\n    layer.trainable = False\n    \nmodel = Sequential()\nmodel.add(base_model)\nmodel.add(GlobalAveragePooling2D())\nmodel.add(Dense(128, activation = 'relu'))\nmodel.add(Dropout(0.2))\nmodel.add(Dense(1, activation = 'sigmoid'))\n\nmodel.compile(optimizer = \"adam\", loss = 'binary_crossentropy', metrics = [\"accuracy\"])\n","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:25:14.4388Z","iopub.execute_input":"2023-12-11T20:25:14.439189Z","iopub.status.idle":"2023-12-11T20:25:18.386808Z","shell.execute_reply.started":"2023-12-11T20:25:14.439154Z","shell.execute_reply":"2023-12-11T20:25:18.385491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:25:18.388879Z","iopub.execute_input":"2023-12-11T20:25:18.389301Z","iopub.status.idle":"2023-12-11T20:25:18.430851Z","shell.execute_reply.started":"2023-12-11T20:25:18.389264Z","shell.execute_reply":"2023-12-11T20:25:18.429811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Train the Model\n**\nThe model was trained with the train and validation data. Early stopping was added to prevent overfitting. In fact, it might be preferable to set the larger number of epochs.","metadata":{}},{"cell_type":"code","source":"callback = tf.keras.callbacks.EarlyStopping(monitor = \"val_loss\", mode = \"min\", patience = 4)\n\nhistory = model.fit(train_generator, validation_data = validation_generator, steps_per_epoch = 20, epochs = 15, callbacks = callback)","metadata":{"execution":{"iopub.status.busy":"2023-12-11T20:25:18.43229Z","iopub.execute_input":"2023-12-11T20:25:18.433291Z","iopub.status.idle":"2023-12-11T22:27:23.464123Z","shell.execute_reply.started":"2023-12-11T20:25:18.433254Z","shell.execute_reply":"2023-12-11T22:27:23.46234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Save the Model\n******","metadata":{}},{"cell_type":"code","source":"model.save('mammography_pred_model.h5')","metadata":{"execution":{"iopub.status.busy":"2023-12-11T22:27:23.466118Z","iopub.execute_input":"2023-12-11T22:27:23.466542Z","iopub.status.idle":"2023-12-11T22:27:23.849895Z","shell.execute_reply.started":"2023-12-11T22:27:23.466504Z","shell.execute_reply":"2023-12-11T22:27:23.848825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Model Metrics\n\n**The accuracy and loss were calculated for both train and validation data in the model training process.","metadata":{}},{"cell_type":"code","source":"accuracy = history.history['accuracy']\nval_accuracy = history.history['val_accuracy']\n\nloss = history.history['loss']\nval_loss = history.history['val_loss']","metadata":{"execution":{"iopub.status.busy":"2023-12-11T22:27:23.85143Z","iopub.execute_input":"2023-12-11T22:27:23.851798Z","iopub.status.idle":"2023-12-11T22:27:23.857966Z","shell.execute_reply.started":"2023-12-11T22:27:23.851744Z","shell.execute_reply":"2023-12-11T22:27:23.856814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"****Visualizing Accuracy and Loss¶\n","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize = (15,10))\n\nplt.subplot(2, 2, 1)\nplt.plot(accuracy, label = \"Training Accuracy\")\nplt.plot(val_accuracy, label = \"Validation Accuracy\")\nplt.ylim(0.4, 1)\nplt.legend(['Train', 'Validation'], loc = 'upper left')\nplt.title(\"Training vs Validation Accuracy\")\nplt.xlabel('epoch')\nplt.ylabel('accuracy')\n\n\nplt.subplot(2, 2, 2)\nplt.plot(loss, label = \"Training Loss\")\nplt.plot(val_loss, label = \"Validation Loss\")\nplt.legend(['Train', 'Validation'], loc = 'upper left')\nplt.title(\"Training vs Validation Loss\")\nplt.xlabel('epoch')\nplt.ylabel('loss')","metadata":{"execution":{"iopub.status.busy":"2023-12-11T22:27:23.859839Z","iopub.execute_input":"2023-12-11T22:27:23.860264Z","iopub.status.idle":"2023-12-11T22:27:24.49206Z","shell.execute_reply.started":"2023-12-11T22:27:23.860222Z","shell.execute_reply":"2023-12-11T22:27:24.491036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Predictions**\n\nIn order to evaluate the quality of the trained model, the outcome was predicted from the test (validation) data and compared with the observed value. The threshold was set as 0.5 and the predicted value of more than 0.5 was treated as 1, which is positive.","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.models import load_model\nmodel = load_model('/kaggle/working/mammography_pred_model.h5')","metadata":{"execution":{"iopub.status.busy":"2023-12-11T22:27:24.493319Z","iopub.execute_input":"2023-12-11T22:27:24.493652Z","iopub.status.idle":"2023-12-11T22:27:27.66658Z","shell.execute_reply.started":"2023-12-11T22:27:24.493623Z","shell.execute_reply":"2023-12-11T22:27:27.665376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = model.predict(validation_generator)","metadata":{"execution":{"iopub.status.busy":"2023-12-11T22:27:27.668296Z","iopub.execute_input":"2023-12-11T22:27:27.668661Z","iopub.status.idle":"2023-12-11T22:30:20.223193Z","shell.execute_reply.started":"2023-12-11T22:27:27.668631Z","shell.execute_reply":"2023-12-11T22:30:20.221823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# prediction by the AI\npred","metadata":{"execution":{"iopub.status.busy":"2023-12-11T22:30:20.224915Z","iopub.execute_input":"2023-12-11T22:30:20.225311Z","iopub.status.idle":"2023-12-11T22:30:20.240672Z","shell.execute_reply.started":"2023-12-11T22:30:20.225277Z","shell.execute_reply":"2023-12-11T22:30:20.23947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = []\nfor prob in pred:\n    if prob >= 0.5:\n        y_pred.append(1)\n    else:\n        y_pred.append(0)\n","metadata":{"execution":{"iopub.status.busy":"2023-12-11T22:30:20.242499Z","iopub.execute_input":"2023-12-11T22:30:20.243707Z","iopub.status.idle":"2023-12-11T22:30:20.256833Z","shell.execute_reply.started":"2023-12-11T22:30:20.243661Z","shell.execute_reply":"2023-12-11T22:30:20.255852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(y_pred)","metadata":{"execution":{"iopub.status.busy":"2023-12-11T22:30:20.258505Z","iopub.execute_input":"2023-12-11T22:30:20.259473Z","iopub.status.idle":"2023-12-11T22:30:20.270706Z","shell.execute_reply.started":"2023-12-11T22:30:20.259422Z","shell.execute_reply":"2023-12-11T22:30:20.269381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.Series(y_pred).value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-12-11T22:30:20.272553Z","iopub.execute_input":"2023-12-11T22:30:20.272998Z","iopub.status.idle":"2023-12-11T22:30:20.286075Z","shell.execute_reply.started":"2023-12-11T22:30:20.272959Z","shell.execute_reply":"2023-12-11T22:30:20.284853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_true = validation_generator.classes","metadata":{"execution":{"iopub.status.busy":"2023-12-11T22:30:20.287932Z","iopub.execute_input":"2023-12-11T22:30:20.288659Z","iopub.status.idle":"2023-12-11T22:30:20.29409Z","shell.execute_reply.started":"2023-12-11T22:30:20.288616Z","shell.execute_reply":"2023-12-11T22:30:20.292848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(y_true)","metadata":{"execution":{"iopub.status.busy":"2023-12-11T22:30:20.295596Z","iopub.execute_input":"2023-12-11T22:30:20.296002Z","iopub.status.idle":"2023-12-11T22:30:20.308015Z","shell.execute_reply.started":"2023-12-11T22:30:20.295963Z","shell.execute_reply":"2023-12-11T22:30:20.306712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"****Confusion Matrix\n\nConfusion matrix was created with the predicted and observed values. The matrix indicated almost correct predictions by the trained model except that there were two cases observed as false positive. False positive means that a case is actually negative but predicted as positive.","metadata":{}},{"cell_type":"code","source":"cm = confusion_matrix(y_true, y_pred)","metadata":{"execution":{"iopub.status.busy":"2023-12-11T22:30:20.311749Z","iopub.execute_input":"2023-12-11T22:30:20.312244Z","iopub.status.idle":"2023-12-11T22:30:20.320645Z","shell.execute_reply.started":"2023-12-11T22:30:20.312207Z","shell.execute_reply":"2023-12-11T22:30:20.319763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the class names.\nclass_names = ['Normal', 'Cancer']\n\n# Create the heatmap with class names as tick labels.\nax = sns.heatmap(cm, annot = True, fmt = '.0f', cmap = \"Blues\", annot_kws = {\"size\": 16},\\\n           xticklabels = class_names, yticklabels = class_names)\n\n# Set the axis labels.\nax.set_xlabel(\"Prediction\")\nax.set_ylabel(\"Truth\")","metadata":{"execution":{"iopub.status.busy":"2023-12-11T23:18:38.683157Z","iopub.execute_input":"2023-12-11T23:18:38.683697Z","iopub.status.idle":"2023-12-11T23:18:38.981869Z","shell.execute_reply.started":"2023-12-11T23:18:38.683655Z","shell.execute_reply":"2023-12-11T23:18:38.980907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"****Classification Report\n\nClassification report describes precision, recall, and f1-score as to each value. Many people regard accuracy and f1-score as the most important indicator to evaluate an AI model. This may be correct, but is not necessarily correct in the clinical field. It must be considered why AI can be useful for and accepted by healthcare professionals. They expect that AI may be able to reduce their workload. What does it mean to reduce their workload by AI? One idea is to exclude lots of negative cases by AI that healthcare professionals would not have to see in order that they would be able to concentrate on the remaining positive cases to be treated. Thus, it is required that the AI should be able to exclude negative cases without false negatives, which are actually positive but predicted as negative. Otherwise, they would have to re-check the negative cases in order not to miss actually positive cases. They must absolutely avoid clinical negligence! Therefore, if the AI model does not give rise to false negative cases, the model will be considerably acceptable in the healthcare field regardless of the accuracy or f1-score.","metadata":{}},{"cell_type":"code","source":"print(classification_report(y_true, y_pred))","metadata":{"execution":{"iopub.status.busy":"2023-12-11T22:30:20.322421Z","iopub.execute_input":"2023-12-11T22:30:20.323154Z","iopub.status.idle":"2023-12-11T22:30:20.343933Z","shell.execute_reply.started":"2023-12-11T22:30:20.323106Z","shell.execute_reply":"2023-12-11T22:30:20.342521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"****Analysing the Results\n\nIt is crucial in the medical field to analyze what kinds of cases were misclassified by the AI model, because medical misdiagnosis must be avoided as much as possible. Thus, it is necessary to identify false positive and false negative cases. Making a data frame and confusion table can visualize the results. As discussed above, false negative cases must be particularly avoided.","metadata":{}},{"cell_type":"code","source":"confusion = []\n\nfor i, j in zip(y_true, y_pred):\n  if i == 0 and j == 0:\n    confusion.append('TN')\n  elif i == 1 and j == 1:\n    confusion.append('TP')\n  elif i == 0 and j == 1:\n    confusion.append('FP')\n  else:\n    confusion.append('FN')","metadata":{"execution":{"iopub.status.busy":"2023-12-11T22:30:20.345404Z","iopub.execute_input":"2023-12-11T22:30:20.345798Z","iopub.status.idle":"2023-12-11T22:30:20.354856Z","shell.execute_reply.started":"2023-12-11T22:30:20.34574Z","shell.execute_reply":"2023-12-11T22:30:20.353662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(confusion)","metadata":{"execution":{"iopub.status.busy":"2023-12-11T22:30:20.356805Z","iopub.execute_input":"2023-12-11T22:30:20.357314Z","iopub.status.idle":"2023-12-11T22:30:20.366242Z","shell.execute_reply.started":"2023-12-11T22:30:20.357274Z","shell.execute_reply":"2023-12-11T22:30:20.365055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"confusion_table = pd.DataFrame(data = confusion, columns = [\"Results\"])\nconfusion_table","metadata":{"execution":{"iopub.status.busy":"2023-12-11T22:30:20.367306Z","iopub.execute_input":"2023-12-11T22:30:20.367674Z","iopub.status.idle":"2023-12-11T22:30:20.385272Z","shell.execute_reply.started":"2023-12-11T22:30:20.367637Z","shell.execute_reply":"2023-12-11T22:30:20.383777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"confusion_table = pd.DataFrame({'Predicton':y_pred,\n                                'Truth': y_true,\n                                'Results': confusion})\nconfusion_table","metadata":{"execution":{"iopub.status.busy":"2023-12-11T22:30:20.387025Z","iopub.execute_input":"2023-12-11T22:30:20.387406Z","iopub.status.idle":"2023-12-11T22:30:20.403145Z","shell.execute_reply.started":"2023-12-11T22:30:20.387372Z","shell.execute_reply":"2023-12-11T22:30:20.401825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"confusion_table.Results == 'FP'","metadata":{"execution":{"iopub.status.busy":"2023-12-11T22:30:20.404774Z","iopub.execute_input":"2023-12-11T22:30:20.405167Z","iopub.status.idle":"2023-12-11T22:30:20.416913Z","shell.execute_reply.started":"2023-12-11T22:30:20.405131Z","shell.execute_reply":"2023-12-11T22:30:20.415562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# list of false positive images\nFPs = confusion_table[confusion_table['Results'] == 'FP']\nFPs","metadata":{"execution":{"iopub.status.busy":"2023-12-11T22:30:20.41866Z","iopub.execute_input":"2023-12-11T22:30:20.419457Z","iopub.status.idle":"2023-12-11T22:30:20.433605Z","shell.execute_reply.started":"2023-12-11T22:30:20.419413Z","shell.execute_reply":"2023-12-11T22:30:20.432329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FPs.index","metadata":{"execution":{"iopub.status.busy":"2023-12-11T22:30:20.435203Z","iopub.execute_input":"2023-12-11T22:30:20.435556Z","iopub.status.idle":"2023-12-11T22:30:20.448472Z","shell.execute_reply.started":"2023-12-11T22:30:20.435526Z","shell.execute_reply":"2023-12-11T22:30:20.447614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# list of false negative images\nFNs = confusion_table[confusion_table['Results'] == 'FN']\nFNs","metadata":{"execution":{"iopub.status.busy":"2023-12-11T22:30:20.449609Z","iopub.execute_input":"2023-12-11T22:30:20.450605Z","iopub.status.idle":"2023-12-11T22:30:20.467358Z","shell.execute_reply.started":"2023-12-11T22:30:20.450559Z","shell.execute_reply":"2023-12-11T22:30:20.466361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FNs.index","metadata":{"execution":{"iopub.status.busy":"2023-12-11T22:30:20.46884Z","iopub.execute_input":"2023-12-11T22:30:20.469307Z","iopub.status.idle":"2023-12-11T22:30:20.482218Z","shell.execute_reply.started":"2023-12-11T22:30:20.469254Z","shell.execute_reply":"2023-12-11T22:30:20.481139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"****Misclassficiation Cases\n\nIt is impoertant to pick up wrong cases judged by the AI and to analyze why the AI made wrong judgements for these images.","metadata":{}},{"cell_type":"code","source":"import glob\nval_images = glob.glob('/kaggle/working/validation/*/*.png')","metadata":{"execution":{"iopub.status.busy":"2023-12-11T23:14:51.155542Z","iopub.execute_input":"2023-12-11T23:14:51.156059Z","iopub.status.idle":"2023-12-11T23:14:51.163737Z","shell.execute_reply.started":"2023-12-11T23:14:51.156017Z","shell.execute_reply":"2023-12-11T23:14:51.162556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# False positive imgages\nfig, axes = plt.subplots(nrows = 2, ncols = 5, figsize = (15, 10), subplot_kw = {'xticks':[], 'yticks':[]})\nfor i, ax in zip(FPs.index[:min(len(FPs), 10)], axes.flat):\n    img = cv2.imread(val_images[i])\n    ax.imshow(img)\n    ax.set_title(\"False Positive Case\")\nfig.tight_layout()    \n\nplt.show()\n\n","metadata":{"execution":{"iopub.status.busy":"2023-12-11T23:14:54.401076Z","iopub.execute_input":"2023-12-11T23:14:54.401527Z","iopub.status.idle":"2023-12-11T23:14:55.688862Z","shell.execute_reply.started":"2023-12-11T23:14:54.401494Z","shell.execute_reply":"2023-12-11T23:14:55.687781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(val_images[:5])  # Print the first 5 paths for inspection\n","metadata":{"execution":{"iopub.status.busy":"2023-12-11T23:15:16.117267Z","iopub.execute_input":"2023-12-11T23:15:16.117727Z","iopub.status.idle":"2023-12-11T23:15:16.12501Z","shell.execute_reply.started":"2023-12-11T23:15:16.117692Z","shell.execute_reply":"2023-12-11T23:15:16.123578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img = cv2.imread(val_images[0])\nprint(img)  # Print the loaded image data","metadata":{"execution":{"iopub.status.busy":"2023-12-11T23:15:19.543204Z","iopub.execute_input":"2023-12-11T23:15:19.543694Z","iopub.status.idle":"2023-12-11T23:15:19.552505Z","shell.execute_reply.started":"2023-12-11T23:15:19.543654Z","shell.execute_reply":"2023-12-11T23:15:19.551251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# False negative imgages\nfig, axes = plt.subplots(nrows = 2, ncols = 5, figsize = (15, 10), subplot_kw = {'xticks':[], 'yticks':[]})\nfor i, ax in zip(FNs.index, axes.flat):\n    img = cv2.imread(val_images[i])\n    ax.imshow(img)\n    ax.set_title(\"False Negative Case\")\nfig.tight_layout()    \n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-11T23:15:32.145446Z","iopub.execute_input":"2023-12-11T23:15:32.14624Z","iopub.status.idle":"2023-12-11T23:15:33.420948Z","shell.execute_reply.started":"2023-12-11T23:15:32.146198Z","shell.execute_reply":"2023-12-11T23:15:33.419938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"****Finetuning the Model (Unfreeying the Layers of the Model)****\n","metadata":{}},{"cell_type":"code","source":"base_model = ResNet50V2(weights = 'imagenet', input_shape = (512, 512, 3), include_top = False)\n\nfor layer in base_model.layers:\n    layer.trainable = True # Change from False to True.\n    \nmodel = Sequential()\nmodel.add(base_model)\nmodel.add(GlobalAveragePooling2D())\nmodel.add(Dense(128, activation = 'relu'))\nmodel.add(Dropout(0.2))\nmodel.add(Dense(1, activation = 'sigmoid'))\n\nmodel.compile(optimizer = \"adam\", loss = 'binary_crossentropy', metrics = [\"accuracy\"])","metadata":{"execution":{"iopub.status.busy":"2023-12-11T22:30:21.715853Z","iopub.status.idle":"2023-12-11T22:30:21.716436Z","shell.execute_reply.started":"2023-12-11T22:30:21.71615Z","shell.execute_reply":"2023-12-11T22:30:21.716178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-12-11T22:30:21.718247Z","iopub.status.idle":"2023-12-11T22:30:21.718802Z","shell.execute_reply.started":"2023-12-11T22:30:21.718508Z","shell.execute_reply":"2023-12-11T22:30:21.718533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"callback = tf.keras.callbacks.EarlyStopping(monitor = \"val_loss\", mode = \"min\", patience = 4)\n\nhistory = model.fit(train_generator, validation_data = validation_generator, steps_per_epoch = 20, epochs = 50, callbacks = callback)","metadata":{"execution":{"iopub.status.busy":"2023-12-11T22:30:21.720359Z","iopub.status.idle":"2023-12-11T22:30:21.721414Z","shell.execute_reply.started":"2023-12-11T22:30:21.721127Z","shell.execute_reply":"2023-12-11T22:30:21.721155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This time we use validation data to calculate the final accuracy.\nfinal_accuracy = model.evaluate_generator(validation_generator)[1]","metadata":{"execution":{"iopub.status.busy":"2023-12-11T22:30:21.723202Z","iopub.status.idle":"2023-12-11T22:30:21.723719Z","shell.execute_reply.started":"2023-12-11T22:30:21.723457Z","shell.execute_reply":"2023-12-11T22:30:21.723482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_accuracy","metadata":{"execution":{"iopub.status.busy":"2023-12-11T22:30:21.725369Z","iopub.status.idle":"2023-12-11T22:30:21.725899Z","shell.execute_reply.started":"2023-12-11T22:30:21.725614Z","shell.execute_reply":"2023-12-11T22:30:21.72564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"****Model Metrics****","metadata":{}},{"cell_type":"code","source":"accuracy = history.history['accuracy']\nval_accuracy  = history.history['val_accuracy']\n\nloss = history.history['loss']\nval_loss = history.history['val_loss']","metadata":{"execution":{"iopub.status.busy":"2023-12-11T23:16:17.960375Z","iopub.execute_input":"2023-12-11T23:16:17.961123Z","iopub.status.idle":"2023-12-11T23:16:17.966533Z","shell.execute_reply.started":"2023-12-11T23:16:17.961084Z","shell.execute_reply":"2023-12-11T23:16:17.965451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"****Visualizing Accuracy and Loss","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize = (15,10))\n\nplt.subplot(2, 2, 1)\nplt.plot(accuracy, label = \"Training Accuracy\")\nplt.plot(val_accuracy, label = \"Validation Accuracy\")\nplt.ylim(0.4, 1)\nplt.legend(['Train', 'Validation'], loc = 'upper left')\nplt.title(\"Training vs Validation Accuracy\")\nplt.xlabel('epoch')\nplt.ylabel('accuracy')\n\n\nplt.subplot(2, 2, 2)\nplt.plot(loss, label = \"Training Loss\")\nplt.plot(val_loss, label = \"Validation Loss\")\nplt.legend(['Train', 'Validation'], loc = 'upper left')\nplt.title(\"Training vs Validation Loss\")\nplt.xlabel('epoch')\nplt.ylabel('loss')\n\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-12-11T23:16:22.851833Z","iopub.execute_input":"2023-12-11T23:16:22.852347Z","iopub.status.idle":"2023-12-11T23:16:23.419813Z","shell.execute_reply.started":"2023-12-11T23:16:22.852306Z","shell.execute_reply":"2023-12-11T23:16:23.418913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"****Actually, Finetuning is worse than Transfer Learning, because the number of data images is small","metadata":{}},{"cell_type":"code","source":"#save model\nmodel.save('mammography_pred_model_finetuning.h5')","metadata":{"execution":{"iopub.status.busy":"2023-12-12T05:36:47.158989Z","iopub.execute_input":"2023-12-12T05:36:47.159361Z","iopub.status.idle":"2023-12-12T05:36:47.412172Z","shell.execute_reply.started":"2023-12-12T05:36:47.159318Z","shell.execute_reply":"2023-12-12T05:36:47.41107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This time we use validation data to calculate the final accuracy.\nfinal_accuracy = model.evaluate_generator(validation_generator)[1]","metadata":{"execution":{"iopub.status.busy":"2023-12-11T22:30:21.74762Z","iopub.status.idle":"2023-12-11T22:30:21.748038Z","shell.execute_reply.started":"2023-12-11T22:30:21.747855Z","shell.execute_reply":"2023-12-11T22:30:21.747873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_accuracy","metadata":{"execution":{"iopub.status.busy":"2023-12-11T22:30:21.749204Z","iopub.status.idle":"2023-12-11T22:30:21.749553Z","shell.execute_reply.started":"2023-12-11T22:30:21.749383Z","shell.execute_reply":"2023-12-11T22:30:21.749399Z"},"trusted":true},"execution_count":null,"outputs":[]}]}