{"cells":[{"metadata":{},"cell_type":"markdown","source":"# Train simple CNN classifier\nHere we train a CNN to predict cancer grade because on tiled histological images. See the following notebook for the generation of these images:  \nhttps://www.kaggle.com/lvulliard/crop-images"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import os\nimport shutil\n\n# There are two ways to load the data from the PANDA dataset:\n# Option 1: Load images using openslide\nimport openslide\n# Option 2: Load images using skimage (requires that tifffile is installed)\nimport skimage.io\n\n# General packages\nimport pandas as pd\nimport numpy as np\nimport matplotlib\nimport matplotlib.pyplot as plt\nimport PIL\nfrom IPython.display import Image, display\nfrom collections import Counter\n\nimport cv2\nimport skimage.io\nfrom tqdm.notebook import tqdm\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import MultiLabelBinarizer\n\nfrom keras.optimizers import Adam\nfrom keras.preprocessing.image import ImageDataGenerator","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import tensorflow as tf\ntf.test.is_gpu_available()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Location of the training images\ndataDir = '../input/panda-resized-train-data-512x512/train_images/train_images/'\n\n# Location of training labels\ntrainLabels = pd.read_csv('/kaggle/input/prostate-cancer-grade-assessment/train.csv').set_index('image_id')\ntestDF = pd.read_csv('/kaggle/input/prostate-cancer-grade-assessment/test.csv').set_index('image_id')\n\n# Output cropped images\n#cropDir = '/kaggle/working/train_images/'\n\ninputShape = (224, 224, 3)\nepochs = 30","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# How many train objects should be included in one batch (higher = faster but less accurate)\n# Take care that the batch size is smaller than the amount of total images analyzed\nbatchSize = 16\nINIT_LR = 0.0001","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"trainDatagen = ImageDataGenerator(rotation_range=30, width_shift_range=0.1,height_shift_range=0.1, validation_split = 0.20,\n                                  zoom_range=0.2, horizontal_flip=True, fill_mode=\"nearest\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"trainDF = pd.DataFrame(list(zip(trainLabels.index + \".png\", trainLabels.isup_grade.astype(str))), \n               columns =['x_col', 'y_col']) \n\n# Uncomment the following if the assumption needs to be re-checked\n# for x in trainDF.x_col:\n#     assert x in os.listdir(dataDir)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"trainGenerator = trainDatagen.flow_from_dataframe(\n    trainDF, x_col=\"x_col\", y_col=\"y_col\",\n    directory=dataDir,  # this is the target directory\n    batch_size=batchSize,\n    class_mode = \"categorical\",\n    subset=\"training\",\n    target_size=(inputShape[0], inputShape[1]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"valGenerator = trainDatagen.flow_from_dataframe(\n    trainDF, x_col=\"x_col\", y_col=\"y_col\",\n    directory=dataDir,  # this is the target directory\n    batch_size=batchSize,\n    class_mode = \"categorical\",\n    subset=\"validation\",\n    target_size=(inputShape[0], inputShape[1]))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Xception model\nSee https://www.kaggle.com/hassanamin/transfer-learning-vgg16-examples-using-tensorflow#Specify-the-Model"},{"metadata":{"trusted":true},"cell_type":"code","source":"from keras.applications import Xception\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Flatten, GlobalAveragePooling2D","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"numClasses = len(set(trainLabels.isup_grade))\nweightFile = \"/kaggle/input/keras-pretrained-models/xception_weights_tf_dim_ordering_tf_kernels_notop.h5\"\n\nmyModel = Sequential()\nmyModel.add(Xception(include_top=False, pooling='avg', weights=weightFile))\nmyModel.add(Dense(6, activation='softmax'))\n\n#myModel.add(activation('softmax'))\n# Say not to train first layer (Xception) model. It is already trained\nmyModel.layers[0].trainable = False","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"myModel.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Optimaztion function\nopt = Adam(lr=INIT_LR, decay=INIT_LR / epochs)\n\nmyModel.compile(loss=\"binary_crossentropy\",\n              optimizer=opt,\n              metrics=[\"accuracy\"])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"H = myModel.fit_generator(trainGenerator,\n                        steps_per_epoch=128,\n                        epochs=epochs, \n                        validation_data=valGenerator,\n                        validation_steps=128,\n                        verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"H_df = pd.DataFrame(H.history)\nH_df[['loss', 'val_loss']].plot()\nH_df[['accuracy', 'val_accuracy']].plot()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Save model\nmyModel.save('/kaggle/working/Xception_'+str(epochs)+'.model')","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}