{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<p id=\"part0\"></p>\n​\n<p style=\"font-family: Arials; line-height: 2; font-size: 44px; font-weight: bold; letter-spacing: 0px; text-align: center; color: #FF8C00\">Breast Cancer</p>\n​\n<img src=\"https://img.medscape.com/thumbnail_library/dt_190424_breast_cancer_800x450.jpg\" width=\"90%\" align=\"center\" hspace=\"20%\" vspace=\"5%\"/>\n​\n<p style=\"font-family: Arials; font-size: 40px; font-style: normal; font-weight: bold; letter-spacing: -2px; color: #000000; line-height:2.0\">Table of content:</p>\n​\n<p style=\"font-family: Arials; font-size: 18px; font-style: normal; font-weight: bold; letter-spacing: 0px; color: #000000; line-height:.5\"><a href=\"#part1\" style=\"color:#000000\"> 1- Importing Libraries</a></p>\n\n<p style=\"font-family: Arials; font-size: 18px; font-style: normal; font-weight: bold; letter-spacing: 0px; color: #000000; line-height:.5\"><a href=\"#part2\" style=\"color:#000000\"> 2- Uploading two needed datasets </a></p>\n\n\n<p style=\"font-family: Arials; font-size: 18px; font-style: normal; font-weight: bold; letter-spacing: 0px; color: #000000; line-height:.5\"><a href=\"#part3\" style=\"color:#000000\"> 3-EDA</a></p>\n\n<p style=\"text-indent: 1vw; font-family: Arials; font-size: 16px; font-style: normal; font-weight: bold; letter-spacing: 0px; color: #000000; line-height:0.5\">\n<a href=\"#part3-1\" style=\"color:#000000\">&nbsp; 3-1 Lable encoding</a></p>\n\n<p style=\"text-indent: 1vw; font-family: Arials; font-size: 16px; font-style: normal; font-weight: bold; letter-spacing: 0px; color: #000000; line-height:0.5\">\n<a href=\"#part3-2\" style=\"color:#000000\">&nbsp; 3-2 Obtaining useful information from datasets</a></p>\n\n<p style=\"text-indent: 1vw; font-family: Arials; font-size: 16px; font-style: normal; font-weight: bold; letter-spacing: 0px; color: #000000; line-height:0.5\">\n<a href=\"#part3-3\" style=\"color:#000000\">&nbsp; 3-3 Working on missing values </a></p>\n\n<p style=\"text-indent: 1vw; font-family: Arials; font-size: 16px; font-style: normal; font-weight: bold; letter-spacing: 0px; color: #000000; line-height:0.5\">\n<a href=\"#part3-3-1\" style=\"color:#000000\">&nbsp; 3-3-1 Filling 'age' column </a></p>\n\n<p style=\"text-indent: 1vw; font-family: Arials; font-size: 16px; font-style: normal; font-weight: bold; letter-spacing: 0px; color: #000000; line-height:0.5\">\n<a href=\"#part3-3-2\" style=\"color:#000000\">&nbsp; 3-3-2 Filling 'BIRADS' column </a></p>\n\n<p style=\"text-indent: 1vw; font-family: Arials; font-size: 16px; font-style: normal; font-weight: bold; letter-spacing: 0px; color: #000000; line-height:0.5\">\n<a href=\"#part3-4\" style=\"color:#000000\">&nbsp; 3-4 Feature selection</a></p>\n\n<p style=\"text-indent: 1vw; font-family: Arials; font-size: 16px; font-style: normal; font-weight: bold; letter-spacing: 0px; color: #000000; line-height:0.5\">\n<a href=\"#part3-4\" style=\"color:#000000\">&nbsp; 3-4-1 Correlation matrix</a></p>\n\n<p style=\"text-indent: 1vw; font-family: Arials; font-size: 16px; font-style: normal; font-weight: bold; letter-spacing: 0px; color: #000000; line-height:0.5\">\n<a href=\"#part3-4\" style=\"color:#000000\">&nbsp; 3-4-2 Mutual information/Gain Entropy</a></p>\n\n<p style=\"text-indent: 1vw; font-family: Arials; font-size: 16px; font-style: normal; font-weight: bold; letter-spacing: 0px; color: #000000; line-height:0.5\">\n<a href=\"#part3-5\" style=\"color:#000000\">&nbsp; 3-5 Visualization</a></p>\n\n<p style=\"font-family: Arials; font-size: 18px; font-style: normal; font-weight: bold; letter-spacing: 0px; color: #000000; line-height:.5\"><a href=\"#part4\" style=\"color:#000000\">4- Creating Pipeline</a></p>\n\n<p style=\"font-family: Arials; font-size: 18px; font-style: normal; font-weight: bold; letter-spacing: 0px; color: #000000; line-height:.5\"><a href=\"#part5\" style=\"color:#000000\">5- Creating Model ResNet50</a></p>\n\n<p style=\"font-family: Arials; font-size: 18px; font-style: normal; font-weight: bold; letter-spacing: 0px; color: #000000; line-height:.5\"><a href=\"#part6\" style=\"color:#000000\">6-Using Callbacks</a></p>\n\n<p style=\"text-indent: 1vw; font-family: Arials; font-size: 16px; font-style: normal; font-weight: bold; letter-spacing: 0px; color: #000000; line-height:0.5\">\n<a href=\"#part6-1\" style=\"color:#000000\">&nbsp; 6-1 Checkpoint </a></p>\n\n<p style=\"text-indent: 1vw; font-family: Arials; font-size: 16px; font-style: normal; font-weight: bold; letter-spacing: 0px; color: #000000; line-height:0.5\">\n<a href=\"#part6-2\" style=\"color:#000000\">&nbsp; 6-2 EarlyStopping </a></p>\n\n<p style=\"text-indent: 1vw; font-family: Arials; font-size: 16px; font-style: normal; font-weight: bold; letter-spacing: 0px; color: #000000; line-height:0.5\">\n<a href=\"#part6-3\" style=\"color:#000000\">&nbsp; 6-3 LearningRateScheduler </a></p>\n\n<p style=\"font-family: Arials; font-size: 18px; font-style: normal; font-weight: bold; letter-spacing: 0px; color: #000000; line-height:.5\"><a href=\"#part7\" style=\"color:#000000\">7- Fiting Model and looking at the results</a></p>\n\n","metadata":{}},{"cell_type":"code","source":"# Check the GPU to make sure that our program works with GPU setting\n!nvidia-smi","metadata":{"execution":{"iopub.status.busy":"2023-10-08T13:30:24.754832Z","iopub.execute_input":"2023-10-08T13:30:24.755446Z","iopub.status.idle":"2023-10-08T13:30:25.75317Z","shell.execute_reply.started":"2023-10-08T13:30:24.75541Z","shell.execute_reply":"2023-10-08T13:30:25.751976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p id=\"part1\"></p>\n\n<p style=\"font-family: Times New Roman; font-size: 20px; font-style: bold; font-weight: bold; letter-spacing: 0px; color: #0000FF\">1- Importing libraris<p>\n<hr style=\"height: 1px; border: 1; background-color: #0000FF\">","metadata":{}},{"cell_type":"code","source":"# For better performance in turning dicom to png files\n# !pip install -qU python-gdcm pydicom pylibjpeg ","metadata":{"execution":{"iopub.status.busy":"2023-10-08T13:30:25.754933Z","iopub.execute_input":"2023-10-08T13:30:25.756092Z","iopub.status.idle":"2023-10-08T13:30:25.76107Z","shell.execute_reply.started":"2023-10-08T13:30:25.756044Z","shell.execute_reply":"2023-10-08T13:30:25.760038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport pydicom\nfrom sklearn import preprocessing\nfrom matplotlib import pyplot as plt\nimport seaborn as sns\nfrom tqdm.notebook import tqdm\nimport cv2, glob, random, os, time, shutil, datetime,time, keras\nimport tensorflow as tf\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.layers import Conv2D,MaxPool2D, BatchNormalization, Activation, Input, Add, Dense, ZeroPadding2D,Flatten, AveragePooling2D, Rescaling\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.optimizers import Adam\nfrom keras import layers, callbacks , metrics\nfrom joblib import Parallel, delayed","metadata":{"execution":{"iopub.status.busy":"2023-10-08T13:30:25.762616Z","iopub.execute_input":"2023-10-08T13:30:25.763733Z","iopub.status.idle":"2023-10-08T13:30:33.884573Z","shell.execute_reply.started":"2023-10-08T13:30:25.763698Z","shell.execute_reply":"2023-10-08T13:30:33.883472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p id=\"part2\"></p>\n\n<p style=\"font-family: Times New Roman; font-size: 20px; font-style: bold; font-weight: bold; letter-spacing: 0px; color: #0000FF\">2-Uploading two needed datasets?</p>\n<hr style=\"height: 1px; border: 1; background-color: #0000FF\">","metadata":{}},{"cell_type":"markdown","source":"**Need two datasets are:**\n \n1st; https://www.kaggle.com/competitions/rsna-breast-cancer-detection    To have access to main dataset\n\n2nd; https://www.kaggle.com/code/mohammadamiri1/dicom-to-png/    To have access to cropped png files, instead of dcm files\n\nIn the second dataset, we did some tasks to crop dicom files to remove uselss spaces , then turned them to png format","metadata":{}},{"cell_type":"markdown","source":"<p id=\"part3\"></p>\n\n<p style=\"font-family: Times New Roman; font-size: 20px; font-style: bold; font-weight: bold; letter-spacing: 0px; color: #0000FF\">3- EDA</p>\n<hr style=\"height: 1px; border: 1; background-color: #0000FF\">","metadata":{}},{"cell_type":"code","source":"# Take a look at train.csv to understand dataset better\ntrain_df_dir = \"/kaggle/input/rsna-breast-cancer-detection/train.csv\"\ntrain_df = pd.read_csv(train_df_dir)\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2023-10-08T13:30:33.888317Z","iopub.execute_input":"2023-10-08T13:30:33.888955Z","iopub.status.idle":"2023-10-08T13:30:34.027257Z","shell.execute_reply.started":"2023-10-08T13:30:33.888926Z","shell.execute_reply":"2023-10-08T13:30:34.026201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Description of columns\n* **site_id** - ID code for the source hospital.\n* **patient_id** - ID code for the patient.\n* **image_id** - ID code for the image.\n* **laterality** - Whether the image is of the left or right breast.\n* **view** - The orientation of the image. The default for a screening exam is to capture two views per breast.\n* **age** - The patient's age in years.\n* **implant** - Whether or not the patient had breast implants. Site 1 only provides breast implant information at the patient level, not at the breast level.\n* **density** - A rating for how dense the breast tissue is, with A being the least dense and D being the most dense. Extremely dense tissue can make diagnosis more difficult. Only provided for train.\n* **machine_id** - An ID code for the imaging device.\n* **cancer** - Whether or not the breast was positive for malignant cancer. The target value. Only provided for train.\n* **biopsy** - Whether or not a follow-up biopsy was performed on the breast. Only provided for train.\n* **invasive** - If the breast is positive for cancer, whether or not the cancer proved to be invasive. Only provided for train.\n* **BIRADS** - 0 if the breast required follow-up, 1 if the breast was rated as negative for cancer, and 2 if the breast was rated as normal. Only provided for train.\n* **prediction_id** - The ID for the matching submission row. Multiple images will share the same prediction ID. Test only.\n* **difficult_negative_case** - True if the case was unusually difficult. Only provided for train.","metadata":{}},{"cell_type":"code","source":"# Take a look at types of columns\ntrain_df.info()","metadata":{"execution":{"iopub.status.busy":"2023-10-08T13:30:34.028652Z","iopub.execute_input":"2023-10-08T13:30:34.029582Z","iopub.status.idle":"2023-10-08T13:30:34.059388Z","shell.execute_reply.started":"2023-10-08T13:30:34.029542Z","shell.execute_reply":"2023-10-08T13:30:34.058211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Here we know that patient_id, image_id, site_id and machine_id cannot be features, affecting the result .So , we turn them into strings\ntrain_df[['patient_id','image_id','site_id','machine_id']] = train_df[['patient_id','image_id','site_id','machine_id']].astype(str)\ntrain_df.info()","metadata":{"execution":{"iopub.status.busy":"2023-10-08T13:30:34.060679Z","iopub.execute_input":"2023-10-08T13:30:34.061253Z","iopub.status.idle":"2023-10-08T13:30:34.224692Z","shell.execute_reply.started":"2023-10-08T13:30:34.061218Z","shell.execute_reply":"2023-10-08T13:30:34.22347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p id=\"part3-1\"></p>\n\n<p style=\"font-family: Times New Roman; font-size: 20px; font-style: bold; font-weight: bold; letter-spacing: 0px; color: #0000FF\">3-1 Label encoding</p>\n<hr style=\"height: 1px; border: 1; background-color: #0000FF\">","metadata":{}},{"cell_type":"code","source":"# As you can see we need to use label encoding here.We need to encoding culomns whcih are objects\nlable_encoder = preprocessing.LabelEncoder()\ntrain_df[['laterality','view','density','difficult_negative_case']] = train_df[['laterality','view','density','difficult_negative_case']].apply(lable_encoder.fit_transform)\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2023-10-08T13:30:34.226243Z","iopub.execute_input":"2023-10-08T13:30:34.227229Z","iopub.status.idle":"2023-10-08T13:30:34.299432Z","shell.execute_reply.started":"2023-10-08T13:30:34.227186Z","shell.execute_reply":"2023-10-08T13:30:34.298261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p id=\"part3-2\"></p>\n\n<p style=\"font-family: Times New Roman; font-size: 20px; font-style: bold; font-weight: bold; letter-spacing: 0px; color: #0000FF\">3-2 Obtaining useful information from datasets</p>\n<hr style=\"height: 1px; border: 1; background-color: #0000FF\">","metadata":{}},{"cell_type":"code","source":"Cases_with_cancer = np.where(train_df.cancer==1)\nCases_without_cancer = np.where(train_df.cancer==0)\nCases_particularly_difficult = np.where(train_df.difficult_negative_case==1)\nCases_not_particularly_difficult = np.where(train_df.difficult_negative_case==0)\nNo_of_hospitals = len(pd.unique(train_df.site_id))\nNo_of_patients = len(pd.unique(train_df.patient_id))\nNo_of_images = len(pd.unique(train_df.image_id))\nNo_of_cases_with_cancer = len(Cases_with_cancer[0])\nNo_of_cases_without_cancer = len(Cases_without_cancer[0])\nNo_of_cases_recgoznied_difficult= len(Cases_particularly_difficult[0])\nNo_of_cases_not_recgoznied_difficult = len(Cases_not_particularly_difficult[0])\nMean_age_of_pateients = np.mean(train_df.age)\n\n\n\nuseful_data = {'No_of_hospitals':No_of_hospitals,'No_of_patients ':No_of_patients ,\n     'No_of_images':No_of_images,'No_ofcases_with_cancer':No_of_cases_with_cancer,\n     'No_of_cases_without_cancer':No_of_cases_without_cancer,'No_of_cases_recgoznied_difficult':No_of_cases_recgoznied_difficult,\n     'No_of_cases_not_recgoznied_difficult':No_of_cases_not_recgoznied_difficult,'Mean_age_of_pateients':Mean_age_of_pateients}\n\nusefull_data = pd.DataFrame(useful_data, index=[0])\nusefull_data","metadata":{"execution":{"iopub.status.busy":"2023-10-08T13:30:34.301004Z","iopub.execute_input":"2023-10-08T13:30:34.301642Z","iopub.status.idle":"2023-10-08T13:30:34.333756Z","shell.execute_reply.started":"2023-10-08T13:30:34.301605Z","shell.execute_reply":"2023-10-08T13:30:34.332696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p id=\"part3-3\"></p>\n\n<p style=\"font-family: Times New Roman; font-size: 20px; font-style: bold; font-weight: bold; letter-spacing: 0px; color: #0000FF\">3-3 Working on missing values</p>\n<hr style=\"height: 1px; border: 1; background-color: #0000FF\">","metadata":{}},{"cell_type":"markdown","source":"First, we must take a look Null/NaN values in diffrent columns . Then, try to fill them with appropriate values","metadata":{}},{"cell_type":"code","source":"# Show how many Null/NaN values we have in dataset\npd.isnull(train_df).sum()","metadata":{"execution":{"iopub.status.busy":"2023-10-08T13:30:34.335058Z","iopub.execute_input":"2023-10-08T13:30:34.336054Z","iopub.status.idle":"2023-10-08T13:30:34.359122Z","shell.execute_reply.started":"2023-10-08T13:30:34.336018Z","shell.execute_reply":"2023-10-08T13:30:34.357998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p id=\"part3-3-1\"></p>\n\n<p style=\"font-family: Times New Roman; font-size: 20px; font-style: bold; font-weight: bold; letter-spacing: 0px; color: #0000FF\">3-3-1 Filling 'age' column</p>\n<hr style=\"height: 1px; border: 1; background-color: #0000FF\">","metadata":{}},{"cell_type":"markdown","source":"**In fact, for this amount of Nan/Null in age clomun, there is no need to fill them. Just we can delete the rows including these values. However, here, to learn how to deal with them, we used linear regreesion method to predict NaN/Null values in age column.**","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LinearRegression\nli = LinearRegression()\n# With the help of following correlation matrix, we could understand that fetures like site_id, cancer, invasive, implant and difficult\n# are more important than others on age target. So, we creat a dataframe with these feautre as input and age as target\n\nnew_dataframe_with_null = train_df[['site_id','cancer','invasive','implant','difficult_negative_case','age']]\nnew_dataframe_without_null = train_df[['site_id','cancer','invasive','implant','difficult_negative_case','age']].dropna()\n\n# set x_train and y_train\nx_train = new_dataframe_without_null.iloc[:,:5] # 'site_id','cancer','invasive','implant','difficult_negative_case'\ny_train = new_dataframe_without_null.iloc[:,-1]  # 'age'\n\n# Set x_test, inculdig parameters where age == null\nx_test = new_dataframe_with_null[new_dataframe_with_null['age'].isnull()].drop(columns='age')\n\n# Run model\nli.fit(x_train,y_train)\n\n# Predict on x_test\npredicted = li.predict(x_test)\nprint('Here, you can see predicetd ages for missing values in main dataset')\nprint(f'-'*50)\nprint(predicted)","metadata":{"execution":{"iopub.status.busy":"2023-10-08T13:30:34.360995Z","iopub.execute_input":"2023-10-08T13:30:34.361927Z","iopub.status.idle":"2023-10-08T13:30:34.714514Z","shell.execute_reply.started":"2023-10-08T13:30:34.361891Z","shell.execute_reply":"2023-10-08T13:30:34.713428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fill missing age values with  predicted which has been tunrned to integer\ntrain_df.loc[train_df.age.isnull(), 'age'] = predicted.astype(int)","metadata":{"execution":{"iopub.status.busy":"2023-10-08T13:30:34.715871Z","iopub.execute_input":"2023-10-08T13:30:34.716786Z","iopub.status.idle":"2023-10-08T13:30:34.725541Z","shell.execute_reply.started":"2023-10-08T13:30:34.716749Z","shell.execute_reply":"2023-10-08T13:30:34.72394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2023-10-08T13:30:34.727199Z","iopub.execute_input":"2023-10-08T13:30:34.728023Z","iopub.status.idle":"2023-10-08T13:30:34.77332Z","shell.execute_reply.started":"2023-10-08T13:30:34.727982Z","shell.execute_reply":"2023-10-08T13:30:34.772219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p id=\"part3-3-2\"></p>\n\n<p style=\"font-family: Times New Roman; font-size: 20px; font-style: bold; font-weight: bold; letter-spacing: 0px; color: #0000FF\">3-3-2 Filling 'BIRADS' column</p>\n<hr style=\"height: 1px; border: 1; background-color: #0000FF\">","metadata":{}},{"cell_type":"markdown","source":"Use Knn algorithm to predicted BIRADS","metadata":{}},{"cell_type":"code","source":"from sklearn.neighbors import KNeighborsClassifier\nknn = KNeighborsClassifier(n_neighbors=3)\n\n#With the help of correlation matrix, we could understand that fetures 'cancer','invasive','difficult_negative_case','density','machine_id','biopsy'\n# are more important than others on 'BIRADS' target. So, we creat a dataframe with these feautre and 'BIRADS' target\n\nnew_dataframe_with_null = train_df[['cancer','invasive','difficult_negative_case','density','machine_id','biopsy','BIRADS']]\nnew_dataframe_without_null = train_df[['cancer','invasive','difficult_negative_case','density','machine_id','biopsy','BIRADS']].dropna()\n\n# set x_train and y_train\nx_train = new_dataframe_without_null.iloc[:,:6] # 'cancer','invasive','difficult_negative_case','density','machine_id','biopsy'\ny_train = new_dataframe_without_null.iloc[:,-1]  # 'BIRADS'\n\n# Run model\nknn.fit(x_train,y_train)\n\n# Set x_test, inculdig parameters where BIRADS == null\nx_test = new_dataframe_with_null[new_dataframe_with_null['BIRADS'].isnull()].drop(columns='BIRADS')\n\n# Predicted\npredicted = knn.predict(x_test)\nprint('Number of predicted values:',len(predicted))\npredicted","metadata":{"execution":{"iopub.status.busy":"2023-10-08T13:30:34.782084Z","iopub.execute_input":"2023-10-08T13:30:34.783536Z","iopub.status.idle":"2023-10-08T13:30:36.20549Z","shell.execute_reply.started":"2023-10-08T13:30:34.783501Z","shell.execute_reply":"2023-10-08T13:30:36.204366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.unique(predicted, return_counts=True)","metadata":{"execution":{"iopub.status.busy":"2023-10-08T13:30:36.206983Z","iopub.execute_input":"2023-10-08T13:30:36.207913Z","iopub.status.idle":"2023-10-08T13:30:36.216622Z","shell.execute_reply.started":"2023-10-08T13:30:36.207872Z","shell.execute_reply":"2023-10-08T13:30:36.21543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fill missing BIRADS values with predicted\ntrain_df.loc[train_df.BIRADS.isnull(), 'BIRADS'] = predicted","metadata":{"execution":{"iopub.status.busy":"2023-10-08T13:30:36.218127Z","iopub.execute_input":"2023-10-08T13:30:36.219267Z","iopub.status.idle":"2023-10-08T13:30:36.226525Z","shell.execute_reply.started":"2023-10-08T13:30:36.219233Z","shell.execute_reply":"2023-10-08T13:30:36.225245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2023-10-08T13:30:36.227909Z","iopub.execute_input":"2023-10-08T13:30:36.228865Z","iopub.status.idle":"2023-10-08T13:30:36.251283Z","shell.execute_reply.started":"2023-10-08T13:30:36.22883Z","shell.execute_reply":"2023-10-08T13:30:36.250235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p id=\"part3-4\"></p>\n\n<p style=\"font-family: Times New Roman; font-size: 20px; font-style: bold; font-weight: bold; letter-spacing: 0px; color: #0000FF\">3-4 Feature selection</p>\n<hr style=\"height: 1px; border: 1; background-color: #0000FF\">","metadata":{}},{"cell_type":"markdown","source":"**Two common fueature selecetion are correlation matrix and mutual information/Gian Entropy**","metadata":{}},{"cell_type":"markdown","source":"<p id=\"part3-4-1\"></p>\n\n<p style=\"font-family: Times New Roman; font-size: 20px; font-style: bold; font-weight: bold; letter-spacing: 0px; color: #0000FF\">3-4-1 Correlation matrix</p>\n<hr style=\"height: 1px; border: 1; background-color: #0000FF\">","metadata":{}},{"cell_type":"code","source":"# We need to know the effect of fetures on each other and on traget\nf, ax = plt.subplots(figsize=(15, 12))\ncorr = train_df.corr()\nsns.heatmap(((corr + 1) * 50),annot=True,linewidth=.5,\n            cmap='crest',\n            square=True, ax=ax, vmin=0, vmax=100)","metadata":{"execution":{"iopub.status.busy":"2023-10-08T13:30:36.252688Z","iopub.execute_input":"2023-10-08T13:30:36.253225Z","iopub.status.idle":"2023-10-08T13:30:37.040799Z","shell.execute_reply.started":"2023-10-08T13:30:36.253192Z","shell.execute_reply":"2023-10-08T13:30:37.039769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p id=\"part3-4-2\"></p>\n\n<p style=\"font-family: Times New Roman; font-size: 20px; font-style: bold; font-weight: bold; letter-spacing: 0px; color: #0000FF\">3-4-2 Mutual Info/Gain Entropy</p>\n<hr style=\"height: 1px; border: 1; background-color: #0000FF\">","metadata":{}},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2023-10-08T13:30:37.042262Z","iopub.execute_input":"2023-10-08T13:30:37.043213Z","iopub.status.idle":"2023-10-08T13:30:37.070189Z","shell.execute_reply.started":"2023-10-08T13:30:37.043171Z","shell.execute_reply":"2023-10-08T13:30:37.068924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"For mutual Info we need to specify x_train and y_train. Here, 'cancer' column located in the middle of dataset is the target.So, for convinient we moved it to the last column.","metadata":{}},{"cell_type":"code","source":"cancer_column = train_df.pop('cancer')","metadata":{"execution":{"iopub.status.busy":"2023-10-08T13:30:37.071825Z","iopub.execute_input":"2023-10-08T13:30:37.072967Z","iopub.status.idle":"2023-10-08T13:30:37.079488Z","shell.execute_reply.started":"2023-10-08T13:30:37.072922Z","shell.execute_reply":"2023-10-08T13:30:37.077877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.insert(13,'cancer',cancer_column)\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2023-10-08T13:30:37.081488Z","iopub.execute_input":"2023-10-08T13:30:37.083089Z","iopub.status.idle":"2023-10-08T13:30:37.109964Z","shell.execute_reply.started":"2023-10-08T13:30:37.083042Z","shell.execute_reply":"2023-10-08T13:30:37.108692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.feature_selection import mutual_info_classif\nx_train = train_df.iloc[:,:13]\ny_train =train_df.iloc[:,-1]\nmutual_info = mutual_info_classif(x_train,y_train)\nmutual_info","metadata":{"execution":{"iopub.status.busy":"2023-10-08T13:30:37.111421Z","iopub.execute_input":"2023-10-08T13:30:37.112409Z","iopub.status.idle":"2023-10-08T13:30:40.255763Z","shell.execute_reply.started":"2023-10-08T13:30:37.112368Z","shell.execute_reply":"2023-10-08T13:30:40.254659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mutual_info = pd.Series(mutual_info)\nmutual_info.index = x_train.columns\nmutual_info.sort_values(ascending=False)","metadata":{"execution":{"iopub.status.busy":"2023-10-08T13:30:40.257367Z","iopub.execute_input":"2023-10-08T13:30:40.258066Z","iopub.status.idle":"2023-10-08T13:30:40.267886Z","shell.execute_reply.started":"2023-10-08T13:30:40.258022Z","shell.execute_reply":"2023-10-08T13:30:40.266643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mutual_info.sort_values(ascending=False).plot.bar(figsize=(8,8))","metadata":{"execution":{"iopub.status.busy":"2023-10-08T13:30:40.269406Z","iopub.execute_input":"2023-10-08T13:30:40.270428Z","iopub.status.idle":"2023-10-08T13:30:40.536023Z","shell.execute_reply.started":"2023-10-08T13:30:40.270386Z","shell.execute_reply":"2023-10-08T13:30:40.535042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p id=\"part3-5\"></p>\n\n<p style=\"font-family: Times New Roman; font-size: 20px; font-style: bold; font-weight: bold; letter-spacing: 0px; color: #0000FF\">3-5 Visualization</p>\n<hr style=\"height: 1px; border: 1; background-color: #0000FF\">","metadata":{}},{"cell_type":"code","source":"fig = plt.figure(figsize=(8,8))\nsns.displot(data=train_df, x=train_df.age,kde=True,hue=train_df.cancer)","metadata":{"execution":{"iopub.status.busy":"2023-10-08T13:30:40.537575Z","iopub.execute_input":"2023-10-08T13:30:40.538166Z","iopub.status.idle":"2023-10-08T13:30:41.48732Z","shell.execute_reply.started":"2023-10-08T13:30:40.538128Z","shell.execute_reply":"2023-10-08T13:30:41.486315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig,(ax1,ax2)= plt.subplots(1,2,figsize=(10,6))\nsns.histplot(data= train_df,x=train_df[train_df.cancer==0].age,kde=True,ax=ax1)\nax1.set_title('No cancer detected',weight='bold',size=15)\nax1.axvline(x=58,ls=':',color='red',)\nax1.text(x=58,y=2200,s=f'Mean:{int(train_df[train_df.cancer==0].age.mean())}')\nax2.set_title('Cancer detected',weight='bold',size=15)\nax2.axvline(x=63,ls=':',color='red',)\nax2.text(x=63,y=140,s=f'Mean:{int(train_df[train_df.cancer==1].age.mean())}')\nsns.histplot(data= train_df,x=train_df[train_df.cancer==1].age,kde=True,ax=ax2)\n","metadata":{"execution":{"iopub.status.busy":"2023-10-08T13:30:41.488837Z","iopub.execute_input":"2023-10-08T13:30:41.489476Z","iopub.status.idle":"2023-10-08T13:30:42.274127Z","shell.execute_reply.started":"2023-10-08T13:30:41.489437Z","shell.execute_reply":"2023-10-08T13:30:42.273137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p id=\"part4\"></p>\n\n<p style=\"font-family: Times New Roman; font-size: 20px; font-style: bold; font-weight: bold; letter-spacing: 0px; color: #0000FF\">4- Creating pipeline</p>\n<hr style=\"height: 1px; border: 1; background-color: #0000FF\">","metadata":{}},{"cell_type":"code","source":"# copy files into two directories called Cancer_Yes and Cancer_NO\nfrom os import listdir\nfrom os.path import isfile, join\nmypath = '/kaggle/input/dicom-to-png/Cancer'\n\ncancer_files = [f for f in listdir(mypath) if isfile(join(mypath, f))]\nlenght = len(cancer_files)\nos.makedirs('/kaggle/working/Output/Cancer_Yes',exist_ok=True)\nfor i in cancer_files:\n    src = join(mypath,i)\n    dst = join(\"/kaggle/working/Output/Cancer_Yes/\",i)\n    shutil.copy(src, dst)\n    \nmypath = '/kaggle/input/dicom-to-png/NO_Cancer'\nno_cancer_files =[f for f in listdir(mypath) if isfile(join(mypath, f))]\nlenght+= len(no_cancer_files)\nos.makedirs('/kaggle/working/Output/Cancer_No',exist_ok=True)\nfor i in no_cancer_files:\n    src = join(mypath,i)\n    dst = join(\"/kaggle/working/Output/Cancer_No/\",i)\n    shutil.copy(src, dst)\nprint('*'*80)\nprint(f'There are {lenght} images to train.')    \nprint('*'*80)","metadata":{"execution":{"iopub.status.busy":"2023-10-08T13:30:42.275802Z","iopub.execute_input":"2023-10-08T13:30:42.276515Z","iopub.status.idle":"2023-10-08T13:36:35.88155Z","shell.execute_reply.started":"2023-10-08T13:30:42.276476Z","shell.execute_reply":"2023-10-08T13:36:35.88038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Setting variables for the model\nmain_directory = \"/kaggle/working/Output/\"\nbatch_size = 64\nimg_height = 256\nimg_width = 256","metadata":{"execution":{"iopub.status.busy":"2023-10-08T13:36:35.882908Z","iopub.execute_input":"2023-10-08T13:36:35.883841Z","iopub.status.idle":"2023-10-08T13:36:35.889044Z","shell.execute_reply.started":"2023-10-08T13:36:35.883792Z","shell.execute_reply":"2023-10-08T13:36:35.887914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# creating pipeline using \"tf.keras.utils.image_dataset_from_directory\"\n\nclass UseKerasUtils:\n#      \"\"\"This class has been provided to use tf.keras.utils.image_dataset_from_directory\"\n#         function to create a pipeline\"\"\"\n    \n    def __init__(self,main_directory, batch_size, img_height, img_width):\n        self.data_dir = main_directory\n        self.batch_size = batch_size\n        self.img_height = img_height\n        self.img_width = img_width\n\n    def create_train_ds(self):\n        train_ds = tf.keras.utils.image_dataset_from_directory(\n        self.data_dir, \n        labels='inferred', # Since we have a main directories, including two subdirectories called ; Cacner and No_Cancer\n        validation_split=0.2,  # keep 20 % of them to validate the result\n        subset=\"training\",\n        seed=123,\n        shuffle = True,\n        color_mode=\"grayscale\", # since our images are gray and have jsut one channel\n        image_size=(self.img_height, self.img_width),\n        batch_size=self.batch_size)\n        return train_ds\n    \n    \n    def create_val_ds(self):\n        val_ds = tf.keras.utils.image_dataset_from_directory(\n        self.data_dir, \n        labels='inferred',\n        validation_split=0.2,  # keep 20 % of them to validate the result\n        subset=\"validation\",\n        seed=123,\n        shuffle = True,\n        color_mode=\"grayscale\",\n        image_size=(self.img_height, self.img_width),\n        batch_size=self.batch_size)\n        return val_ds","metadata":{"execution":{"iopub.status.busy":"2023-10-08T13:36:35.890731Z","iopub.execute_input":"2023-10-08T13:36:35.891509Z","iopub.status.idle":"2023-10-08T13:36:35.902794Z","shell.execute_reply.started":"2023-10-08T13:36:35.891472Z","shell.execute_reply":"2023-10-08T13:36:35.901828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Call functions and create two train and validation datasets \nfirst_pipeline = UseKerasUtils(main_directory, batch_size, img_height, img_width)\ntrain_ds1 = first_pipeline.create_train_ds()\nval_ds1 = first_pipeline.create_val_ds()","metadata":{"execution":{"iopub.status.busy":"2023-10-08T13:36:35.904141Z","iopub.execute_input":"2023-10-08T13:36:35.905067Z","iopub.status.idle":"2023-10-08T13:36:40.463346Z","shell.execute_reply.started":"2023-10-08T13:36:35.905034Z","shell.execute_reply":"2023-10-08T13:36:40.462268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_names = train_ds1.class_names\nprint(f\"There are {len(class_names)} classes: \",class_names)","metadata":{"execution":{"iopub.status.busy":"2023-10-08T13:36:40.464872Z","iopub.execute_input":"2023-10-08T13:36:40.465882Z","iopub.status.idle":"2023-10-08T13:36:40.472454Z","shell.execute_reply.started":"2023-10-08T13:36:40.465836Z","shell.execute_reply":"2023-10-08T13:36:40.471116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# show the size of batches\nfor images, labels in val_ds1.take(8):\n    print('Output of pipe line has this dimension:')\n    print(images.shape)\n    print(('Batch_size','img_height','img_width','Channel'))\n    print('-'*40)\n    print('Number of labels:',labels.shape)\n    print(labels.numpy())\n    print('-'*40)\n    for i in images:\n        print(f'maximum pixel is:{np.max(np.squeeze(i))} and minimum pixel is:{np.min(np.squeeze(i))}. So, you need to normalize it')\n        print('One image data:\\n',(np.squeeze(i)))\n        break\n    break","metadata":{"execution":{"iopub.status.busy":"2023-10-08T13:36:40.474042Z","iopub.execute_input":"2023-10-08T13:36:40.475048Z","iopub.status.idle":"2023-10-08T13:36:41.073566Z","shell.execute_reply.started":"2023-10-08T13:36:40.475012Z","shell.execute_reply":"2023-10-08T13:36:41.07238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# These codes are used to boost the performance of your model whike readind data from pipeline.\nAUTOTUNE = tf.data.AUTOTUNE\ntrain_ds1 = train_ds1.cache().prefetch(buffer_size=AUTOTUNE)\nval_ds1 = val_ds1.cache().prefetch(buffer_size=AUTOTUNE)","metadata":{"execution":{"iopub.status.busy":"2023-10-08T13:36:41.082537Z","iopub.execute_input":"2023-10-08T13:36:41.082837Z","iopub.status.idle":"2023-10-08T13:36:41.095306Z","shell.execute_reply.started":"2023-10-08T13:36:41.08281Z","shell.execute_reply":"2023-10-08T13:36:41.094174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p id=\"part5\"></p>\n\n<p style=\"font-family: Times New Roman; font-size: 20px; font-style: bold; font-weight: bold; letter-spacing: 0px; color: #0000FF\">5- Creating Model ResNet50</p>\n<hr style=\"height: 1px; border: 1; background-color: #0000FF\">","metadata":{}},{"cell_type":"code","source":" def identity_block(stage, block, inputs, filters, kernel_size = [(1,1),(3,3),(1,1)], strides=[(1,1),(1,1),(1,1)], padding=['valid','same','valid']):\n        \n        # define variables\n        f1,f2,f3 = filters\n        k1,k2,k3 = kernel_size\n        s1,s2,s3 = strides\n        p1,p2,p3 =padding\n        conv_name_base =  'Conv_Stage' + str(stage) +\"_Block_\"+ block \n        bn_name_base = 'BatchNorm_Stage' + str(stage) +\"_Block_\"+ block \n        ac_name_base = 'Activation_Stage' + str(stage) + \"_Block_\" + block\n    \n        x_shortcut = inputs\n\n        # First component of main path\n        x = Conv2D(filters=f1, kernel_size=k1, strides=s1, padding=p1, name=conv_name_base + '_Component1')(inputs)\n        x = BatchNormalization(name=bn_name_base + '_Component1')(x)\n        x = Activation('relu',name=ac_name_base + '_Component1')(x)\n\n        # Second component of main path\n        x = Conv2D(filters=f2, kernel_size=k2, strides=s2, padding=p2, name=conv_name_base + '_Component2')(x)\n        x = BatchNormalization(name=bn_name_base + '_Component2')(x)\n        x = Activation('relu',name=ac_name_base + '_Component2')(x)\n\n        # Third component of main path\n        x = Conv2D(filters=f3, kernel_size=k3, strides=s3, padding=p3, name=conv_name_base + '_Component3')(x)\n        x = BatchNormalization(name=bn_name_base + '_Component3')(x)\n\n        # Final step: Add shortcut value to main path, and pass it through a RELU activation\n        x = Add()([x, x_shortcut])\n        x = Activation('relu',name=ac_name_base + '_Output')(x)\n        return x\n","metadata":{"execution":{"iopub.status.busy":"2023-10-08T13:36:41.096888Z","iopub.execute_input":"2023-10-08T13:36:41.097934Z","iopub.status.idle":"2023-10-08T13:36:41.108086Z","shell.execute_reply.started":"2023-10-08T13:36:41.097897Z","shell.execute_reply":"2023-10-08T13:36:41.107018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" def convolutional_block(stage, block, inputs, filters,  kernel_size = [(1,1),(3,3),(1,1)], strides=[(2,2),(1,1),(1,1)], padding=['valid','same','valid']):\n        \n        # define variables\n\n        f1,f2,f3 = filters\n        k1,k2,k3 = kernel_size\n        s1,s2,s3 = strides\n        p1,p2,p3 =padding\n        conv_name_base =  'Conv_Stage' + str(stage) +\"_Block_\"+ block \n        bn_name_base = 'BatchNorm_Stage' + str(stage) +\"_Block_\"+ block \n        ac_name_base = 'Activation_Stage' + str(stage) + \"_Block_\" + block\n        x_shortcut = inputs\n       \n        # First component of main path \n        x = Conv2D(filters=f1, kernel_size=k1, strides=s1, padding=p1, name=conv_name_base + '_Component1')(inputs)\n        x = BatchNormalization(name=bn_name_base + '_Component1')(x)\n        x = Activation('relu',name=ac_name_base + '_Component1')(x)\n       \n        # Second component of main path (≈3 lines)\n        x = Conv2D(filters=f2, kernel_size=k2, strides=s2, padding=p2, name=conv_name_base + '_Component2')(x)\n        x = BatchNormalization(name=bn_name_base + '_Component2')(x)\n        x = Activation('relu',name=ac_name_base + '_Component2')(x)\n\n        # Third component of main path (≈2 lines)\n        x = Conv2D(filters=f3, kernel_size=k3, strides=s3, padding=p3, name=conv_name_base + '_Component3')(x)\n        x = BatchNormalization(name=bn_name_base + '_Component3')(x)\n       \n        ##### SHORTCUT PATH #### (≈2 lines)\n        x_shortcut = Conv2D(filters=f3, kernel_size=k3, strides=s1, padding=p3, name=conv_name_base + '_Shortcut')(x_shortcut)\n        x_shortcut = BatchNormalization(name=bn_name_base + '_Shortcout')(x_shortcut)\n        \n          # Final step: Add shortcut value to main path, and pass it through a RELU activation (≈2 lines)\n        x = Add()([x, x_shortcut])\n        x = Activation('relu',name=ac_name_base + '_Output')(x)\n        return x\n      \n    ","metadata":{"execution":{"iopub.status.busy":"2023-10-08T13:36:41.109865Z","iopub.execute_input":"2023-10-08T13:36:41.110206Z","iopub.status.idle":"2023-10-08T13:36:41.124625Z","shell.execute_reply.started":"2023-10-08T13:36:41.110173Z","shell.execute_reply":"2023-10-08T13:36:41.123647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def ResNet50(input_shape, outputClasses):\n    \n    # Define the input as a tensor with shape input_shape\n    x_input = Input(input_shape,name= \"My_Picture\")\n    \n    x = Rescaling(scale=1./255)(x_input) # rescaling data to [0-1]\n    # Zero-Padding\n    x = ZeroPadding2D((3, 3), name = \"ZeroPadding_Stage0\")(x)\n\n    # Stage 1\n    x = Conv2D(64, kernel_size=(7, 7), strides=(2, 2), name=\"Conv_Stage1\" )(x)\n    x = BatchNormalization(name=\"BatchNorm_Stage1\")(x)\n    x = Activation('relu', name=\"Activation_Stage1\")(x)\n    x = ZeroPadding2D((1,1), name = \"ZeroPadding_Stage1\")(x)\n    x = MaxPool2D((3, 3), strides=(2, 2), name= \"MaxPooling_Stage1\")(x)\n  \n\n    # Stage 2\n    x = convolutional_block(stage=2, block='a',inputs=x, filters=[64, 64, 256], strides=[(1,1),(1,1),(1,1)])\n    x = identity_block(stage=2, block='b', inputs=x, filters=[64, 64, 256])\n    x = identity_block(stage=2, block='c',inputs=x, filters=[64, 64, 256])\n\n    # Stage 3 \n    x = convolutional_block(stage=3, block='a', inputs=x, filters=[128, 128, 512])\n    x = identity_block(stage=3, block='b', inputs=x, filters=[128, 128, 512])\n    x = identity_block(stage=3, block='c', inputs=x, filters=[128, 128, 512])\n    x = identity_block(stage=3, block='d', inputs=x, filters=[128, 128, 512])\n\n    # Stage 4\n    x = convolutional_block(stage=4, block='a', inputs=x, filters=[256, 256, 1024])\n    x = identity_block(stage=4, block='b', inputs=x, filters=[256, 256, 1024])\n    x = identity_block(stage=4, block='c', inputs=x, filters=[256, 256, 1024])\n    x = identity_block(stage=4, block='d', inputs=x, filters=[256, 256, 1024])\n    x = identity_block(stage=4, block='e', inputs=x, filters=[256, 256, 1024])\n    x = identity_block(stage=4, block='f', inputs=x, filters=[256, 256, 1024])\n\n    # Stage 5\n    x = convolutional_block(stage=5, block='a', inputs=x, filters=[512, 512, 2048])\n    x = identity_block(stage=5, block='b', inputs=x, filters=[512, 512, 2048])\n    x = identity_block(stage=5, block='c', inputs=x, filters=[512, 512, 2048])\n\n    # AVGPOOL \n    x = AveragePooling2D(pool_size=(7, 7), padding='same', name = \"AveragePooling\")(x)\n\n    # output layer\n    x = Flatten(name= 'Faltten_OutputLayer')(x)\n    x = Dense(outputClasses, activation='sigmoid', name='fc' + str(outputClasses))(x)\n\n    # Create model\n    model = Model(inputs=x_input, outputs=x, name='ResNet50')\n\n     # Compile model\n    my_optimizer = Adam(learning_rate=0.01)\n    model.compile(loss='binary_crossentropy',optimizer=my_optimizer, metrics= 'accuracy')\n    return model","metadata":{"execution":{"iopub.status.busy":"2023-10-08T13:36:41.126204Z","iopub.execute_input":"2023-10-08T13:36:41.126559Z","iopub.status.idle":"2023-10-08T13:36:41.145735Z","shell.execute_reply.started":"2023-10-08T13:36:41.126525Z","shell.execute_reply":"2023-10-08T13:36:41.144577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = ResNet50((256,256,1),1)","metadata":{"execution":{"iopub.status.busy":"2023-10-08T13:36:41.147048Z","iopub.execute_input":"2023-10-08T13:36:41.14746Z","iopub.status.idle":"2023-10-08T13:36:42.283141Z","shell.execute_reply.started":"2023-10-08T13:36:41.147421Z","shell.execute_reply":"2023-10-08T13:36:42.282075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"keras.utils.plot_model(model, \"multi_input_and_output_model.png\", show_shapes=True)","metadata":{"execution":{"iopub.status.busy":"2023-10-08T13:36:42.284809Z","iopub.execute_input":"2023-10-08T13:36:42.285502Z","iopub.status.idle":"2023-10-08T13:36:45.12177Z","shell.execute_reply.started":"2023-10-08T13:36:42.28546Z","shell.execute_reply":"2023-10-08T13:36:45.12013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p id=\"part6\"></p>\n\n<p style=\"font-family: Times New Roman; font-size: 20px; font-style: bold; font-weight: bold; letter-spacing: 0px; color: #0000FF\">6- Using callbacks</p>\n<hr style=\"height: 1px; border: 1; background-color: #0000FF\">","metadata":{}},{"cell_type":"markdown","source":"**Here, We use callbacks from keras to have a better control on training model**","metadata":{}},{"cell_type":"markdown","source":"<p id=\"part6-1\"></p>\n\n<p style=\"font-family: Times New Roman; font-size: 20px; font-style: bold; font-weight: bold; letter-spacing: 0px; color: #0000FF\">6-1 Checkpoint</p>\n<hr style=\"height: 1px; border: 1; background-color: #0000FF\">","metadata":{}},{"cell_type":"code","source":"checkpoint_path = '/kaggle/working/Checkpint_Folder/cp.ckpt' # The path to save wights of the model after training\ncheckpoint_dir =os.path.dirname(checkpoint_path)\nMy_ModelCheckpoint_callback =callbacks.ModelCheckpoint(filepath=checkpoint_path,\n                                          metrics='val_accuracy',\n                                          verbose=1,\n                                          save_weights_only = True,\n                                          save_best_only=True,\n                                          mode='auto',\n                                          save_freq ='epoch')\n\n","metadata":{"execution":{"iopub.status.busy":"2023-10-08T13:36:45.124574Z","iopub.execute_input":"2023-10-08T13:36:45.12551Z","iopub.status.idle":"2023-10-08T13:36:45.133142Z","shell.execute_reply.started":"2023-10-08T13:36:45.12545Z","shell.execute_reply":"2023-10-08T13:36:45.132324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p id=\"part6-2\"></p>\n\n<p style=\"font-family: Times New Roman; font-size: 20px; font-style: bold; font-weight: bold; letter-spacing: 0px; color: #0000FF\">6-2 EarlyStopping</p>\n<hr style=\"height: 1px; border: 1; background-color: #0000FF\">","metadata":{}},{"cell_type":"code","source":"My_EarlyStopping_callback = callbacks.EarlyStopping(monitor=\"val_accuracy\",\n                                                 min_delta=.02,\n                                                 patience=5,\n                                                 verbose=1,\n                                                 mode=\"auto\")","metadata":{"execution":{"iopub.status.busy":"2023-10-08T13:36:45.134469Z","iopub.execute_input":"2023-10-08T13:36:45.135381Z","iopub.status.idle":"2023-10-08T13:36:45.148441Z","shell.execute_reply.started":"2023-10-08T13:36:45.135339Z","shell.execute_reply":"2023-10-08T13:36:45.147513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p id=\"part6-3\"></p>\n\n<p style=\"font-family: Times New Roman; font-size: 20px; font-style: bold; font-weight: bold; letter-spacing: 0px; color: #0000FF\">6-3 LearningRateScheduler</p>\n<hr style=\"height: 1px; border: 1; background-color: #0000FF\">\n","metadata":{}},{"cell_type":"code","source":"## Step 0: Download the txt file\n! wget https://raw.githubusercontent.com/Mohammad-Amirifard/Callbacks/main/learning_rate_file.txt","metadata":{"execution":{"iopub.status.busy":"2023-10-08T13:36:45.157728Z","iopub.execute_input":"2023-10-08T13:36:45.158953Z","iopub.status.idle":"2023-10-08T13:36:46.418762Z","shell.execute_reply.started":"2023-10-08T13:36:45.158907Z","shell.execute_reply":"2023-10-08T13:36:46.416793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n## Step 1: Define a function\ndef get_learning_rate_from_file(filename,epoch):\n    with open(filename,'r') as f:\n        for line in f.readlines():\n            line = line.split('#',1)[0]\n            if line:\n                par =line.strip().split(':')\n                j = int(par[0])\n                lr = float(par[1])\n                if j<= epoch and lr >0 :\n                    learning_rate = lr\n                else:\n                    return learning_rate\n                return learning_rate\n        return learning_rate\n\n\n## Step 2: Create a wrapper function to solve one input of the callback.LearninRateSchedule.\ndef get_schedule(filename):\n    def scheduler(epoch):\n        lr = get_learning_rate_from_file(\"/kaggle/working/learning_rate_file.txt\",epoch)\n        return lr\n    return scheduler\n\n## Step 3: call function\nscheduler = get_schedule(\"/kaggle/working/learning_rate_file.txt\")\nMy_LearningRateScheduler_callback = callbacks.LearningRateScheduler(scheduler)\n","metadata":{"execution":{"iopub.status.busy":"2023-10-08T13:36:46.421691Z","iopub.execute_input":"2023-10-08T13:36:46.422209Z","iopub.status.idle":"2023-10-08T13:36:46.44274Z","shell.execute_reply.started":"2023-10-08T13:36:46.422149Z","shell.execute_reply":"2023-10-08T13:36:46.441366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"print(\"The model is ready to train. Let's get started\")","metadata":{"execution":{"iopub.status.busy":"2023-10-08T13:36:46.444636Z","iopub.execute_input":"2023-10-08T13:36:46.448486Z","iopub.status.idle":"2023-10-08T13:36:46.463368Z","shell.execute_reply.started":"2023-10-08T13:36:46.448444Z","shell.execute_reply":"2023-10-08T13:36:46.462119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p id=\"part7\"></p>\n\n<p style=\"font-family: Times New Roman; font-size: 20px; font-style: bold; font-weight: bold; letter-spacing: 0px; color: #0000FF\">7- Fitting Model and looking at the results</p>\n<hr style=\"height: 1px; border: 1; background-color: #0000FF\">","metadata":{}},{"cell_type":"code","source":"My_all_callbacks = [My_ModelCheckpoint_callback, My_EarlyStopping_callback, My_LearningRateScheduler_callback]\n\nhistory = model.fit(\n  train_ds1,\n  validation_data=val_ds1,\n  epochs=20,verbose=1,callbacks=My_all_callbacks\n)\n","metadata":{"execution":{"iopub.status.busy":"2023-10-08T13:36:46.465151Z","iopub.execute_input":"2023-10-08T13:36:46.465936Z","iopub.status.idle":"2023-10-08T13:53:16.619982Z","shell.execute_reply.started":"2023-10-08T13:36:46.4659Z","shell.execute_reply":"2023-10-08T13:53:16.618927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history.history","metadata":{"execution":{"iopub.status.busy":"2023-10-08T13:53:16.624572Z","iopub.execute_input":"2023-10-08T13:53:16.625605Z","iopub.status.idle":"2023-10-08T13:53:16.636054Z","shell.execute_reply.started":"2023-10-08T13:53:16.625562Z","shell.execute_reply":"2023-10-08T13:53:16.634926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1,2,figsize=(12, 4))\nx = range(1,len(history.history['loss'])+1)\ny = history.history['loss']\nax[0].plot(x,y,label='loss')\nax[0].plot(x,history.history['val_loss'],label='val_loss')\nax[0].set_xticks(range(1, len(y)+1))\nax[0].set_xlabel('epoch',fontsize=10)\nax[0].set_ylabel('loss',fontsize=10)\nax[0].legend()\n\n                 \nx = range(1,len(history.history['accuracy'])+1)\ny = history.history['accuracy']\nax[1].plot(x,y,label='accuracy')\nax[1].plot(x,history.history['val_accuracy'],label='val_accuracy')\nax[1].set_xticks(range(1, len(y)+1))\nax[1].set_xlabel('epoch',fontsize=10)\nax[1].set_ylabel('accuracy',fontsize=10)\nax[1].legend()","metadata":{"execution":{"iopub.status.busy":"2023-10-08T13:53:16.638436Z","iopub.execute_input":"2023-10-08T13:53:16.639428Z","iopub.status.idle":"2023-10-08T13:53:17.009313Z","shell.execute_reply.started":"2023-10-08T13:53:16.639382Z","shell.execute_reply":"2023-10-08T13:53:17.008171Z"},"trusted":true},"execution_count":null,"outputs":[]}]}