{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-06T12:58:31.007494Z","iopub.execute_input":"2023-01-06T12:58:31.007945Z","iopub.status.idle":"2023-01-06T12:58:31.014335Z","shell.execute_reply.started":"2023-01-06T12:58:31.007906Z","shell.execute_reply":"2023-01-06T12:58:31.013264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_PATH = '/kaggle/input/rsnabcd-512-png-v2-dataset'\n# train\ndf = pd.read_csv(f'{BASE_PATH}/train.csv')\ndf['image_path'] = f'{BASE_PATH}/train_images'\\\n                    + '/' + df.patient_id.astype(str)\\\n                    + '/' + df.image_id.astype(str)\\\n                    + '.png'\nprint('Train:')\ndisplay(df.head(2))\n\n# test\ntest_df = pd.read_csv(f'{BASE_PATH}/test.csv')\ntest_df['image_path'] = f'{BASE_PATH}/test_images'\\\n                    + '/' + test_df.patient_id.astype(str)\\\n                    + '/' + test_df.image_id.astype(str)\\\n                    + '.png'\nprint('\\nTest:')\ndisplay(test_df.head(2))","metadata":{"execution":{"iopub.status.busy":"2023-01-06T12:58:33.64295Z","iopub.execute_input":"2023-01-06T12:58:33.644043Z","iopub.status.idle":"2023-01-06T12:58:33.949957Z","shell.execute_reply.started":"2023-01-06T12:58:33.643965Z","shell.execute_reply":"2023-01-06T12:58:33.948895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\ntf.io.gfile.exists(df.image_path.iloc[0]), tf.io.gfile.exists(test_df.image_path.iloc[0])","metadata":{"execution":{"iopub.status.busy":"2023-01-06T12:58:38.173147Z","iopub.execute_input":"2023-01-06T12:58:38.173799Z","iopub.status.idle":"2023-01-06T12:58:43.015928Z","shell.execute_reply.started":"2023-01-06T12:58:38.173763Z","shell.execute_reply":"2023-01-06T12:58:43.014912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('train_files:',df.shape[0])\nprint('test_files:',test_df.shape[0])","metadata":{"execution":{"iopub.status.busy":"2023-01-06T12:58:43.017954Z","iopub.execute_input":"2023-01-06T12:58:43.018602Z","iopub.status.idle":"2023-01-06T12:58:43.024141Z","shell.execute_reply.started":"2023-01-06T12:58:43.018563Z","shell.execute_reply":"2023-01-06T12:58:43.023059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Exploration & Image Pre-Processing","metadata":{}},{"cell_type":"code","source":"# Import necessary packages\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport os\nimport seaborn as sns\nsns.set()","metadata":{"execution":{"iopub.status.busy":"2023-01-06T12:58:43.025725Z","iopub.execute_input":"2023-01-06T12:58:43.027356Z","iopub.status.idle":"2023-01-06T12:58:43.589948Z","shell.execute_reply.started":"2023-01-06T12:58:43.027319Z","shell.execute_reply":"2023-01-06T12:58:43.588974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'There are {df.shape[0]} rows and {df.shape[1]} columns in this data frame')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-06T12:58:44.790724Z","iopub.execute_input":"2023-01-06T12:58:44.791512Z","iopub.status.idle":"2023-01-06T12:58:44.812088Z","shell.execute_reply.started":"2023-01-06T12:58:44.791466Z","shell.execute_reply":"2023-01-06T12:58:44.811035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 1.1 Data Types and Null Values Check","metadata":{}},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2023-01-06T12:58:48.359953Z","iopub.execute_input":"2023-01-06T12:58:48.36066Z","iopub.status.idle":"2023-01-06T12:58:48.393414Z","shell.execute_reply.started":"2023-01-06T12:58:48.360625Z","shell.execute_reply":"2023-01-06T12:58:48.39233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 1.2 Unique IDs Check\n\n\"PatientId\" has an identification number for each patient. One thing you'd like to know about a medical dataset like this is if you're looking at repeated data for certain patients or whether each image represents a different person.","metadata":{}},{"cell_type":"code","source":"print(f\"The total patient ids are {df['patient_id'].count()}, from those the unique ids are {df['patient_id'].value_counts().shape[0]} \")","metadata":{"execution":{"iopub.status.busy":"2023-01-06T11:18:25.155759Z","iopub.execute_input":"2023-01-06T11:18:25.156705Z","iopub.status.idle":"2023-01-06T11:18:25.168765Z","shell.execute_reply.started":"2023-01-06T11:18:25.156657Z","shell.execute_reply":"2023-01-06T11:18:25.167259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As you can see, the number of unique patients in the dataset is less than the total number so there must be some overlap. For patients with multiple records, you'll want to make sure they do not show up in both training and test sets in order to avoid data leakage (covered later in this week's lectures).\n\n### 1.3 Data Labels\n","metadata":{}},{"cell_type":"code","source":"columns = df.keys()\ncolumns = list(columns)\nprint(columns)","metadata":{"execution":{"iopub.status.busy":"2023-01-06T11:18:25.171434Z","iopub.execute_input":"2023-01-06T11:18:25.171888Z","iopub.status.idle":"2023-01-06T11:18:25.182798Z","shell.execute_reply.started":"2023-01-06T11:18:25.171851Z","shell.execute_reply":"2023-01-06T11:18:25.181426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Remove unnecesary elements\ncolumns = ['cancer', 'biopsy', 'invasive', 'implant']\n# Get the total classes\nprint(f\"There are {len(columns)} columns of labels for these conditions: {columns}\")","metadata":{"execution":{"iopub.status.busy":"2023-01-06T11:18:25.184173Z","iopub.execute_input":"2023-01-06T11:18:25.184935Z","iopub.status.idle":"2023-01-06T11:18:25.197574Z","shell.execute_reply.started":"2023-01-06T11:18:25.184889Z","shell.execute_reply":"2023-01-06T11:18:25.195945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Print out the number of positive labels for each class\nfor column in columns:\n    print(f\"The class {column} has {df[column].sum()} samples\")","metadata":{"execution":{"iopub.status.busy":"2023-01-06T11:18:25.205247Z","iopub.execute_input":"2023-01-06T11:18:25.205679Z","iopub.status.idle":"2023-01-06T11:18:25.217506Z","shell.execute_reply.started":"2023-01-06T11:18:25.205642Z","shell.execute_reply":"2023-01-06T11:18:25.215832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 1.4 Data Visualization\n\nUsing the image names listed in the csv file, you can retrieve the image associated with each row of data in your dataframe. \n\nRun the cell below to visualize a random selection of images from the dataset.","metadata":{}},{"cell_type":"code","source":"# Extract numpy values from Image column in data frame\nimages = df['image_path'].values\n\n# Extract 9 random images from it\nrandom_images = [np.random.choice(images) for i in range(9)]\n\n# Location of the image dir\n# img_dir = 'data/nih/images-small/'\n\nprint('Display Random Images')\n\n# Adjust the size of your images\nplt.figure(figsize=(15,15))\n\n# Iterate and plot random images\nfor i in range(9):\n    plt.subplot(3, 3, i + 1)\n    img = plt.imread(random_images[i])\n    plt.imshow(img, cmap='gray')\n    plt.axis('off')\n    \n# Adjust subplot parameters to give specified padding\nplt.tight_layout()    ","metadata":{"execution":{"iopub.status.busy":"2023-01-06T11:18:25.219765Z","iopub.execute_input":"2023-01-06T11:18:25.221835Z","iopub.status.idle":"2023-01-06T11:18:27.950966Z","shell.execute_reply.started":"2023-01-06T11:18:25.221784Z","shell.execute_reply":"2023-01-06T11:18:27.943546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 1.5 Investigating a Single Image","metadata":{}},{"cell_type":"code","source":"# Get the first image that was listed in the train_df dataframe\nsample_img = df['image_path'][100]\nraw_image = plt.imread(sample_img)\nplt.imshow(raw_image, cmap='gray')\nplt.colorbar()\nplt.title('Raw Chest X Ray Image')\nprint(f\"The dimensions of the image are {raw_image.shape[0]} pixels width and {raw_image.shape[1]} pixels height, one single color channel\")\nprint(f\"The maximum pixel value is {raw_image.max():.4f} and the minimum is {raw_image.min():.4f}\")\nprint(f\"The mean value of the pixels is {raw_image.mean():.4f} and the standard deviation is {raw_image.std():.4f}\")","metadata":{"execution":{"iopub.status.busy":"2023-01-06T11:18:27.952419Z","iopub.execute_input":"2023-01-06T11:18:27.953204Z","iopub.status.idle":"2023-01-06T11:18:28.55126Z","shell.execute_reply.started":"2023-01-06T11:18:27.953163Z","shell.execute_reply":"2023-01-06T11:18:28.550321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 1.6 Investigating Pixel Value Distribution\n","metadata":{}},{"cell_type":"code","source":"# Plot a histogram of the distribution of the pixels\nsns.distplot(raw_image.ravel(), \n             label=f'Pixel Mean {np.mean(raw_image):.4f} & Standard Deviation {np.std(raw_image):.4f}', kde=False)\nplt.legend(loc='upper center')\nplt.title('Distribution of Pixel Intensities in the Image')\nplt.xlabel('Pixel Intensity')\nplt.ylabel('# Pixels in Image')","metadata":{"execution":{"iopub.status.busy":"2023-01-06T11:18:28.555373Z","iopub.execute_input":"2023-01-06T11:18:28.557974Z","iopub.status.idle":"2023-01-06T11:18:29.032406Z","shell.execute_reply.started":"2023-01-06T11:18:28.557936Z","shell.execute_reply":"2023-01-06T11:18:29.031411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2. Image Preprocessing in Keras\n\nBefore training, you'll first modify your images to be better suited for training a convolutional neural network. For this task you'll use the Keras [ImageDataGenerator](https://keras.io/preprocessing/image/) function to perform data preprocessing and data augmentation.","metadata":{}},{"cell_type":"code","source":"from keras.preprocessing.image import ImageDataGenerator","metadata":{"execution":{"iopub.status.busy":"2023-01-06T12:59:13.216985Z","iopub.execute_input":"2023-01-06T12:59:13.217402Z","iopub.status.idle":"2023-01-06T12:59:13.883655Z","shell.execute_reply.started":"2023-01-06T12:59:13.21737Z","shell.execute_reply":"2023-01-06T12:59:13.882654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Normalize images\nimage_generator = ImageDataGenerator(\n    samplewise_center=True, #Set each sample mean to 0.\n    samplewise_std_normalization= True # Divide each input by its standard deviation\n)","metadata":{"execution":{"iopub.status.busy":"2023-01-06T12:59:15.177385Z","iopub.execute_input":"2023-01-06T12:59:15.177894Z","iopub.status.idle":"2023-01-06T12:59:15.187778Z","shell.execute_reply.started":"2023-01-06T12:59:15.177854Z","shell.execute_reply":"2023-01-06T12:59:15.186589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 2.1 Standardization\n\nThe `image_generator` you created above will act to adjust your image data such that the new mean of the data will be zero, and the standard deviation of the data will be 1.  \n\nIn other words, the generator will replace each pixel value in the image with a new value calculated by subtracting the mean and dividing by the standard deviation.\n\n$$\\frac{x_i - \\mu}{\\sigma}$$","metadata":{}},{"cell_type":"code","source":"# Flow from directory with specified batch size and target image size\n# generator = image_generator.flow_from_dataframe(\n#         dataframe=df,\n# #         directory=\"data/nih/images-small/\",\n#         x_col=\"image_path\", # features\n#         # Let's say we build a model for mass detection\n#         y_col= ['cancer'], # labels\n#         class_mode=\"raw\", # 'Mass' column should be in train_df\n#         batch_size= 1, # images per batch\n#         shuffle=False, # shuffle the rows or not\n#         target_size=(320,320) # width and height of output image\n# )","metadata":{"execution":{"iopub.status.busy":"2023-01-06T11:22:38.245805Z","iopub.execute_input":"2023-01-06T11:22:38.246205Z","iopub.status.idle":"2023-01-06T11:22:38.253801Z","shell.execute_reply.started":"2023-01-06T11:22:38.246172Z","shell.execute_reply":"2023-01-06T11:22:38.252202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Plot a processed image\n# sns.set_style(\"white\")\n# generated_image, label = generator.__getitem__(100)\n# plt.imshow(generated_image[0], cmap='gray')\n# plt.colorbar()\n# plt.title('Raw Chest X Ray Image')\n# print(f\"The dimensions of the image are {generated_image.shape[1]} pixels width and {generated_image.shape[2]} pixels height\")\n# print(f\"The maximum pixel value is {generated_image.max():.4f} and the minimum is {generated_image.min():.4f}\")\n# print(f\"The mean value of the pixels is {generated_image.mean():.4f} and the standard deviation is {generated_image.std():.4f}\")","metadata":{"execution":{"iopub.status.busy":"2023-01-06T11:22:48.831443Z","iopub.execute_input":"2023-01-06T11:22:48.832161Z","iopub.status.idle":"2023-01-06T11:22:48.836891Z","shell.execute_reply.started":"2023-01-06T11:22:48.832125Z","shell.execute_reply":"2023-01-06T11:22:48.835661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Include a histogram of the distribution of the pixels\n# sns.set()\n# plt.figure(figsize=(10, 7))\n\n# # Plot histogram for original iamge\n# sns.distplot(raw_image.ravel(), \n#              label=f'Original Image: mean {np.mean(raw_image):.4f} - Standard Deviation {np.std(raw_image):.4f} \\n '\n#              f'Min pixel value {np.min(raw_image):.4} - Max pixel value {np.max(raw_image):.4}',\n#              color='blue', \n#              kde=False)\n\n# # Plot histogram for generated image\n# sns.distplot(generated_image[0].ravel(), \n#              label=f'Generated Image: mean {np.mean(generated_image[0]):.4f} - Standard Deviation {np.std(generated_image[0]):.4f} \\n'\n#              f'Min pixel value {np.min(generated_image[0]):.4} - Max pixel value {np.max(generated_image[0]):.4}', \n#              color='red', \n#              kde=False)\n\n# # Place legends\n# plt.legend()\n# plt.title('Distribution of Pixel Intensities in the Image')\n# plt.xlabel('Pixel Intensity')\n# plt.ylabel('# Pixel')","metadata":{"execution":{"iopub.status.busy":"2023-01-06T11:23:02.429934Z","iopub.execute_input":"2023-01-06T11:23:02.431134Z","iopub.status.idle":"2023-01-06T11:23:02.43788Z","shell.execute_reply.started":"2023-01-06T11:23:02.431088Z","shell.execute_reply":"2023-01-06T11:23:02.436681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Patient Overlap and Data Leakage\n\nPatient overlap in medical data is a part of a more general problem in machine learning called **data leakage**.  we'll check to see if a patient's ID appears in both the training set and the test set. we should also verify that we don't have patient overlap in the training and validation sets, which is what we'll do here.\n","metadata":{}},{"cell_type":"code","source":"ids = df['patient_id'].values","metadata":{"execution":{"iopub.status.busy":"2023-01-06T12:59:29.219121Z","iopub.execute_input":"2023-01-06T12:59:29.219834Z","iopub.status.idle":"2023-01-06T12:59:29.226627Z","shell.execute_reply.started":"2023-01-06T12:59:29.219797Z","shell.execute_reply":"2023-01-06T12:59:29.223582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ids_ = set(ids)","metadata":{"execution":{"iopub.status.busy":"2023-01-06T12:59:30.864698Z","iopub.execute_input":"2023-01-06T12:59:30.865067Z","iopub.status.idle":"2023-01-06T12:59:30.87519Z","shell.execute_reply.started":"2023-01-06T12:59:30.865035Z","shell.execute_reply":"2023-01-06T12:59:30.87403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(ids_)","metadata":{"execution":{"iopub.status.busy":"2023-01-06T12:59:31.377574Z","iopub.execute_input":"2023-01-06T12:59:31.378283Z","iopub.status.idle":"2023-01-06T12:59:31.38546Z","shell.execute_reply.started":"2023-01-06T12:59:31.378245Z","shell.execute_reply":"2023-01-06T12:59:31.384327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ids_val = list(ids_)[:2384]\nlen(ids_val)","metadata":{"execution":{"iopub.status.busy":"2023-01-06T12:59:31.710676Z","iopub.execute_input":"2023-01-06T12:59:31.710949Z","iopub.status.idle":"2023-01-06T12:59:31.720036Z","shell.execute_reply.started":"2023-01-06T12:59:31.710924Z","shell.execute_reply":"2023-01-06T12:59:31.718884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ids_test = list(ids_)[2384:2784]\nlen(ids_test)","metadata":{"execution":{"iopub.status.busy":"2023-01-06T12:59:31.985508Z","iopub.execute_input":"2023-01-06T12:59:31.985782Z","iopub.status.idle":"2023-01-06T12:59:31.993158Z","shell.execute_reply.started":"2023-01-06T12:59:31.985757Z","shell.execute_reply":"2023-01-06T12:59:31.992155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ids_train=[]\n\nfor i in ids_ :\n    if i in ids_val :\n        pass\n    elif i in ids_test:\n        pass\n    else:\n        ids_train.append(i)","metadata":{"execution":{"iopub.status.busy":"2023-01-06T12:59:32.341621Z","iopub.execute_input":"2023-01-06T12:59:32.342407Z","iopub.status.idle":"2023-01-06T12:59:32.732393Z","shell.execute_reply.started":"2023-01-06T12:59:32.342371Z","shell.execute_reply":"2023-01-06T12:59:32.731516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(ids_train)","metadata":{"execution":{"iopub.status.busy":"2023-01-06T12:59:32.734148Z","iopub.execute_input":"2023-01-06T12:59:32.734469Z","iopub.status.idle":"2023-01-06T12:59:32.741277Z","shell.execute_reply.started":"2023-01-06T12:59:32.734436Z","shell.execute_reply":"2023-01-06T12:59:32.740284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = df[df['patient_id'].isin(ids_train)]","metadata":{"execution":{"iopub.status.busy":"2023-01-06T12:59:58.695595Z","iopub.execute_input":"2023-01-06T12:59:58.695955Z","iopub.status.idle":"2023-01-06T12:59:58.712307Z","shell.execute_reply.started":"2023-01-06T12:59:58.695924Z","shell.execute_reply":"2023-01-06T12:59:58.711393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-01-06T12:59:59.106716Z","iopub.execute_input":"2023-01-06T12:59:59.107161Z","iopub.status.idle":"2023-01-06T12:59:59.114621Z","shell.execute_reply.started":"2023-01-06T12:59:59.107122Z","shell.execute_reply":"2023-01-06T12:59:59.113604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_df = df[df['patient_id'].isin(ids_val)]","metadata":{"execution":{"iopub.status.busy":"2023-01-06T13:00:01.403849Z","iopub.execute_input":"2023-01-06T13:00:01.404536Z","iopub.status.idle":"2023-01-06T13:00:01.413567Z","shell.execute_reply.started":"2023-01-06T13:00:01.404499Z","shell.execute_reply":"2023-01-06T13:00:01.41255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-01-06T13:00:01.815937Z","iopub.execute_input":"2023-01-06T13:00:01.816541Z","iopub.status.idle":"2023-01-06T13:00:01.823646Z","shell.execute_reply.started":"2023-01-06T13:00:01.816503Z","shell.execute_reply":"2023-01-06T13:00:01.822468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = df[df['patient_id'].isin(ids_test)]\ntest_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-01-06T13:00:02.370391Z","iopub.execute_input":"2023-01-06T13:00:02.370724Z","iopub.status.idle":"2023-01-06T13:00:02.380709Z","shell.execute_reply.started":"2023-01-06T13:00:02.370695Z","shell.execute_reply":"2023-01-06T13:00:02.379661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'There are {train_df.shape[0]} rows and {train_df.shape[1]} columns in the training dataframe')\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-06T13:00:06.671409Z","iopub.execute_input":"2023-01-06T13:00:06.671853Z","iopub.status.idle":"2023-01-06T13:00:06.693319Z","shell.execute_reply.started":"2023-01-06T13:00:06.671813Z","shell.execute_reply":"2023-01-06T13:00:06.692249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'There are {valid_df.shape[0]} rows and {valid_df.shape[1]} columns in the validation dataframe')\nvalid_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-06T13:00:11.534376Z","iopub.execute_input":"2023-01-06T13:00:11.534733Z","iopub.status.idle":"2023-01-06T13:00:11.557021Z","shell.execute_reply.started":"2023-01-06T13:00:11.534702Z","shell.execute_reply":"2023-01-06T13:00:11.556069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'There are {test_df.shape[0]} rows and {test_df.shape[1]} columns in the test dataframe')\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-06T11:24:15.575316Z","iopub.execute_input":"2023-01-06T11:24:15.575726Z","iopub.status.idle":"2023-01-06T11:24:15.598047Z","shell.execute_reply.started":"2023-01-06T11:24:15.575692Z","shell.execute_reply":"2023-01-06T11:24:15.597068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 1.2 Extracting Patient IDs ","metadata":{}},{"cell_type":"code","source":"# Extract patient id's for the training set\nids_train = train_df['patient_id'].values\n# Extract patient id's for the validation set\nids_valid = valid_df['patient_id'].values\n# Extract patient id's for the test set\nids_test = test_df['patient_id'].values","metadata":{"execution":{"iopub.status.busy":"2023-01-06T11:24:38.493891Z","iopub.execute_input":"2023-01-06T11:24:38.494262Z","iopub.status.idle":"2023-01-06T11:24:38.501286Z","shell.execute_reply.started":"2023-01-06T11:24:38.494229Z","shell.execute_reply":"2023-01-06T11:24:38.500139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ids_train.shape","metadata":{"execution":{"iopub.status.busy":"2023-01-06T11:24:40.940596Z","iopub.execute_input":"2023-01-06T11:24:40.941749Z","iopub.status.idle":"2023-01-06T11:24:40.948336Z","shell.execute_reply.started":"2023-01-06T11:24:40.941709Z","shell.execute_reply":"2023-01-06T11:24:40.947157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 1.3 Comparing PatientIDs for Train & Validation Sets\n\n2. Convert these arrays of numbers into `set()` datatypes for easy comparison\n3. Identify patient overlap in the intersection of the two sets","metadata":{}},{"cell_type":"code","source":"# Create a \"set\" datastructure of the training set id's to identify unique id's\nids_train_set = set(ids_train)\nprint(f'There are {len(ids_train_set)} unique Patient IDs in the training set')\n# Create a \"set\" datastructure of the validation set id's to identify unique id's\nids_valid_set = set(ids_valid)\nprint(f'There are {len(ids_valid_set)} unique Patient IDs in the valiadtion set')\nids_test_set = set(ids_test)","metadata":{"execution":{"iopub.status.busy":"2023-01-06T11:25:07.864402Z","iopub.execute_input":"2023-01-06T11:25:07.864797Z","iopub.status.idle":"2023-01-06T11:25:07.878308Z","shell.execute_reply.started":"2023-01-06T11:25:07.864765Z","shell.execute_reply":"2023-01-06T11:25:07.877171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Identify patient overlap by looking at the intersection between the sets\npatient_overlap = list(ids_train_set.intersection(ids_valid_set))\nn_overlap = len(patient_overlap)\nprint(f'There are {n_overlap} Patient IDs in both the training and validation sets')\nprint('')\nprint(f'These patients are in both the training and validation datasets:')\nprint(f'{patient_overlap}')","metadata":{"execution":{"iopub.status.busy":"2023-01-06T11:25:21.072323Z","iopub.execute_input":"2023-01-06T11:25:21.073034Z","iopub.status.idle":"2023-01-06T11:25:21.08177Z","shell.execute_reply.started":"2023-01-06T11:25:21.072997Z","shell.execute_reply":"2023-01-06T11:25:21.080496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Build a separate generator for train, valid, and test sets","metadata":{}},{"cell_type":"code","source":"def get_train_generator(df, x_col, y_cols, shuffle=True, batch_size=8, seed=1, target_w = 320, target_h = 320):\n    \"\"\"\n    Return generator for training set, normalizing using batch\n    statistics.\n\n    Args:\n      train_df (dataframe): dataframe specifying training data.\n      image_dir (str): directory where image files are held.\n      x_col (str): name of column in df that holds filenames.\n      y_cols (list): list of strings that hold y labels for images.\n      batch_size (int): images per batch to be fed into model during training.\n      seed (int): random seed.\n      target_w (int): final width of input images.\n      target_h (int): final height of input images.\n    \n    Returns:\n        train_generator (DataFrameIterator): iterator over training set\n    \"\"\"        \n    print(\"getting train generator...\") \n    # normalize images\n    image_generator = ImageDataGenerator(\n        samplewise_center=True,\n        samplewise_std_normalization= True)\n    \n    # flow from directory with specified batch size\n    # and target image size\n    generator = image_generator.flow_from_dataframe(\n            dataframe=df,\n            x_col=x_col,\n            y_col=y_cols,\n            class_mode=\"raw\",\n            batch_size=batch_size,\n            shuffle=shuffle,\n            seed=seed,\n            target_size=(target_w,target_h))\n    \n    return generator","metadata":{"execution":{"iopub.status.busy":"2023-01-06T13:00:18.816423Z","iopub.execute_input":"2023-01-06T13:00:18.817499Z","iopub.status.idle":"2023-01-06T13:00:18.824885Z","shell.execute_reply.started":"2023-01-06T13:00:18.817457Z","shell.execute_reply":"2023-01-06T13:00:18.823854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_test_and_valid_generator(valid_df, test_df, train_df, x_col, y_cols, sample_size=100, batch_size=8, seed=1, target_w = 320, target_h = 320):\n    \"\"\"\n    Return generator for validation set and test test set using \n    normalization statistics from training set.\n\n    Args:\n      valid_df (dataframe): dataframe specifying validation data.\n      test_df (dataframe): dataframe specifying test data.\n      train_df (dataframe): dataframe specifying training data.\n      image_dir (str): directory where image files are held.\n      x_col (str): name of column in df that holds filenames.\n      y_cols (list): list of strings that hold y labels for images.\n      sample_size (int): size of sample to use for normalization statistics.\n      batch_size (int): images per batch to be fed into model during training.\n      seed (int): random seed.\n      target_w (int): final width of input images.\n      target_h (int): final height of input images.\n    \n    Returns:\n        test_generator (DataFrameIterator) and valid_generator: iterators over test set and validation set respectively\n    \"\"\"\n    print(\"getting train and valid generators...\")\n    # get generator to sample dataset\n    raw_train_generator = ImageDataGenerator().flow_from_dataframe(\n        dataframe=train_df, \n        x_col=x_col,\n        y_col=y_cols,\n        class_mode=\"raw\", \n        batch_size=sample_size, \n        shuffle=True, \n        target_size=(target_w, target_h))\n    \n    # get data sample\n    batch = raw_train_generator.next()\n    data_sample = batch[0]\n\n    # use sample to fit mean and std for test set generator\n    image_generator = ImageDataGenerator(\n        featurewise_center=True,\n        featurewise_std_normalization= True)\n    \n    # fit generator to sample from training data\n    image_generator.fit(data_sample)\n\n    # get test generator\n    valid_generator = image_generator.flow_from_dataframe(\n            dataframe=valid_df,\n            x_col=x_col,\n            y_col=y_cols,\n            class_mode=\"raw\",\n            batch_size=batch_size,\n            shuffle=False,\n            seed=seed,\n            target_size=(target_w,target_h))\n\n    test_generator = image_generator.flow_from_dataframe(\n            dataframe=test_df,\n            x_col=x_col,\n            y_col=y_cols,\n            class_mode=\"raw\",\n            batch_size=batch_size,\n            shuffle=False,\n            seed=seed,\n            target_size=(target_w,target_h))\n    return valid_generator, test_generator","metadata":{"execution":{"iopub.status.busy":"2023-01-06T13:00:19.496044Z","iopub.execute_input":"2023-01-06T13:00:19.4964Z","iopub.status.idle":"2023-01-06T13:00:19.507539Z","shell.execute_reply.started":"2023-01-06T13:00:19.496369Z","shell.execute_reply":"2023-01-06T13:00:19.506434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_generator = get_train_generator(train_df, \"image_path\", 'cancer')\nvalid_generator, test_generator= get_test_and_valid_generator(valid_df, test_df, train_df,\"image_path\", 'cancer')","metadata":{"execution":{"iopub.status.busy":"2023-01-06T13:00:22.6534Z","iopub.execute_input":"2023-01-06T13:00:22.653758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x, y = train_generator.__getitem__(0)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(x[7])","metadata":{"execution":{"iopub.status.busy":"2023-01-06T12:33:11.871318Z","iopub.execute_input":"2023-01-06T12:33:11.87235Z","iopub.status.idle":"2023-01-06T12:33:12.164134Z","shell.execute_reply.started":"2023-01-06T12:33:11.872301Z","shell.execute_reply":"2023-01-06T12:33:12.163131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y","metadata":{"execution":{"iopub.status.busy":"2023-01-06T12:32:28.679021Z","iopub.execute_input":"2023-01-06T12:32:28.681705Z","iopub.status.idle":"2023-01-06T12:32:28.687899Z","shell.execute_reply.started":"2023-01-06T12:32:28.681665Z","shell.execute_reply":"2023-01-06T12:32:28.686977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Densenet","metadata":{}},{"cell_type":"code","source":"''' calibrated categorical focal loss'''\n\ndef categorical_focal_loss_fixed(y_true, y_pred, alpha = 1.0, gamma = 2.0):\n        \"\"\"\n        :param y_true: A tensor of the same shape as `y_pred`\n        :param y_pred: A tensor resulting from a softmax\n        :return: Output tensor.\n        \"\"\"\n\n        # Clip the prediction value to prevent NaN's and Inf's\n        epsilon = K.epsilon()\n        y_pred = K.clip(y_pred, epsilon, 1. - epsilon)\n\n        # Calculate Cross Entropy\n        cross_entropy = -y_true * K.log(y_pred)\n\n        # Calculate Focal Loss\n        loss = alpha * K.pow(1 - y_pred, gamma) * cross_entropy\n\n        # Compute mean loss in mini_batch\n        return K.mean(K.sum(loss, axis=-1))  ","metadata":{"execution":{"iopub.status.busy":"2023-01-06T12:33:29.16925Z","iopub.execute_input":"2023-01-06T12:33:29.170243Z","iopub.status.idle":"2023-01-06T12:33:29.17754Z","shell.execute_reply.started":"2023-01-06T12:33:29.170193Z","shell.execute_reply":"2023-01-06T12:33:29.176285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport os\nimport glob\nimport cv2\nimport matplotlib.pyplot as plt\nimport pandas as pd\nimport pickle\nimport tensorflow as tf\nfrom tensorflow.keras import backend as K\nimport shutil\nimport random\nimport seaborn as sns\nfrom keras import layers\nfrom keras.models import Sequential\nfrom tensorflow.keras.preprocessing import image_dataset_from_directory\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom sklearn.metrics import accuracy_score, confusion_matrix,plot_confusion_matrix\nfrom tensorflow.keras.utils import to_categorical\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.models import load_model\nfrom tensorflow.keras.optimizers import Adam, RMSprop\nfrom tensorflow.keras.metrics import categorical_crossentropy\nfrom tensorflow.keras.callbacks import ModelCheckpoint, EarlyStopping\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.model_selection import KFold\nfrom tensorflow.keras.applications import VGG16\nimport itertools\nfrom tensorflow.keras.layers import Flatten, Conv2D, Concatenate, MaxPooling2D, ZeroPadding2D, concatenate, Input, Reshape, GlobalAveragePooling2D, Dense, Dropout, Activation, BatchNormalization, Dropout, LSTM, ConvLSTM2D","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"    # Define the model architecture\n    model = tf.keras.applications.DenseNet169(\n            include_top=False,\n            # weights='imagenet',\n            input_shape=(320,320,3)\n        )\n    # model.summary()\n    model.trainable = True\n    layer = model.output \n    layer = ZeroPadding2D()(layer)\n    layer = Conv2D(32,(3,3), activation='relu', name = 'extra_conv_resnet') (layer)\n    layer = GlobalAveragePooling2D()(layer) \n    logits = Dense(1, \n                  activation='softmax', \n                  name='predictions')(layer)\n    model_resnet = Model(inputs=model.input, \n                    outputs=logits, \n                    name = 'resnet_pretrained')\n#     filepath = f'/content/drive/MyDrive/densenet/Training for fold {fold_no} '+'Densenet169_categorical_focal_loss_fixed.h5'\n#     checkpoint = ModelCheckpoint(filepath, monitor='val_accuracy', \n#                              verbose=1, \n#                              save_weights_only=False, \n#                              save_best_only=True, \n#                              mode='max', \n#                              save_freq='epoch')\n    \n    sgd = tf.keras.optimizers.SGD(lr=0.001, decay=1e-6, momentum=0.9, nesterov=True)  \n    adam = tf.keras.optimizers.Adam(\n    learning_rate=0.001,\n    amsgrad=True)\n    model_resnet.compile(optimizer=sgd, \n                        loss= 'categorical_crossentropy', run_eagerly=True,  #use other losses in the same way\n                        metrics=['accuracy']) \n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model_resnet.fit(train_generator,\n              batch_size=8,\n              epochs=50,\n              validation_data=(valid_generator),\n              verbose=1)","metadata":{"execution":{"iopub.status.busy":"2023-01-06T12:58:01.746343Z","iopub.execute_input":"2023-01-06T12:58:01.746664Z","iopub.status.idle":"2023-01-06T12:58:01.826696Z","shell.execute_reply.started":"2023-01-06T12:58:01.746589Z","shell.execute_reply":"2023-01-06T12:58:01.825218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}