{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":18647,"databundleVersionId":1126921,"sourceType":"competition"}],"dockerImageVersionId":30408,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nimport pandas as pd\nimport rasterio\nfrom sklearn.utils import shuffle\nimport openslide\n\nimport os\nimport sys\nfrom shutil import copyfile, move\nfrom tqdm import tqdm\nimport h5py\nimport random\nfrom random import randint\n\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras import layers\nfrom tensorflow.keras import Input\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import InputLayer, Input\nfrom tensorflow.keras.layers import Conv2D, Dense, Flatten, Dropout, Activation\nfrom tensorflow.keras.layers import BatchNormalization, Reshape, MaxPooling2D, GlobalAveragePooling2D\nfrom tensorflow.keras.callbacks import Callback\nfrom tensorflow.keras.callbacks import ModelCheckpoint, ReduceLROnPlateau, EarlyStopping\nfrom tensorflow.keras.applications import ResNet50, VGG16\nfrom keras.losses import mean_squared_error\nimport keras as K\nfrom sklearn.metrics import cohen_kappa_score","metadata":{"execution":{"iopub.status.busy":"2023-03-01T17:53:58.858692Z","iopub.execute_input":"2023-03-01T17:53:58.85957Z","iopub.status.idle":"2023-03-01T17:54:07.319219Z","shell.execute_reply.started":"2023-03-01T17:53:58.859524Z","shell.execute_reply":"2023-03-01T17:54:07.318117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(\"/kaggle/input/prostate-cancer-grade-assessment/train.csv\")\nimage_path = \"/kaggle/input/prostate-cancer-grade-assessment/train_images/\"","metadata":{"execution":{"iopub.status.busy":"2023-03-01T17:54:07.322157Z","iopub.execute_input":"2023-03-01T17:54:07.323452Z","iopub.status.idle":"2023-03-01T17:54:07.357461Z","shell.execute_reply.started":"2023-03-01T17:54:07.323392Z","shell.execute_reply":"2023-03-01T17:54:07.356438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-01T17:54:07.359118Z","iopub.execute_input":"2023-03-01T17:54:07.359884Z","iopub.status.idle":"2023-03-01T17:54:07.37802Z","shell.execute_reply.started":"2023-03-01T17:54:07.359846Z","shell.execute_reply":"2023-03-01T17:54:07.376908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train_df)","metadata":{"execution":{"iopub.status.busy":"2023-03-01T17:54:07.379703Z","iopub.execute_input":"2023-03-01T17:54:07.380124Z","iopub.status.idle":"2023-03-01T17:54:07.387436Z","shell.execute_reply.started":"2023-03-01T17:54:07.380088Z","shell.execute_reply":"2023-03-01T17:54:07.386195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_size = 256\ntraining_sample_percentage = 0.9\ntraining_item_count = int(len(train_df)*training_sample_percentage)\ntrain_df[\"image_path\"] = [image_path+image_id+\".tiff\" for image_id in train_df[\"image_id\"]]","metadata":{"execution":{"iopub.status.busy":"2023-03-01T17:54:07.39124Z","iopub.execute_input":"2023-03-01T17:54:07.392046Z","iopub.status.idle":"2023-03-01T17:54:07.405451Z","shell.execute_reply.started":"2023-03-01T17:54:07.392018Z","shell.execute_reply":"2023-03-01T17:54:07.404153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#remove all image file that don't have a mask file\nindex_to_drop = []\nfor idx, row in train_df.iterrows():\n    mask_path = row.image_path.replace(\"train_images\",\"train_label_masks\").replace(\".tiff\",\"_mask.tiff\")\n\n    if not os.path.isfile(mask_path):\n        index_to_drop.append(idx)\n\ntrain_df.drop(index_to_drop,0,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-03-01T17:54:07.406957Z","iopub.execute_input":"2023-03-01T17:54:07.407725Z","iopub.status.idle":"2023-03-01T17:54:25.93267Z","shell.execute_reply.started":"2023-03-01T17:54:07.407683Z","shell.execute_reply":"2023-03-01T17:54:25.931572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"example = openslide.OpenSlide(train_df.iloc[0].image_path)\nprint(example.dimensions)\nclipped_example = example.read_region((5000, 5000), 0, (256, 256))\nplt.imshow(clipped_example)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-01T17:54:25.933976Z","iopub.execute_input":"2023-03-01T17:54:25.934998Z","iopub.status.idle":"2023-03-01T17:54:26.303509Z","shell.execute_reply.started":"2023-03-01T17:54:25.934957Z","shell.execute_reply":"2023-03-01T17:54:26.302383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-01T17:54:26.304434Z","iopub.execute_input":"2023-03-01T17:54:26.304774Z","iopub.status.idle":"2023-03-01T17:54:26.320002Z","shell.execute_reply.started":"2023-03-01T17:54:26.304734Z","shell.execute_reply":"2023-03-01T17:54:26.319004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = shuffle(train_df)\nvalidation_df = train_df[training_item_count:]\ntrain_df = train_df[:training_item_count]","metadata":{"execution":{"iopub.status.busy":"2023-03-01T17:54:26.321926Z","iopub.execute_input":"2023-03-01T17:54:26.322795Z","iopub.status.idle":"2023-03-01T17:54:26.334356Z","shell.execute_reply.started":"2023-03-01T17:54:26.32275Z","shell.execute_reply":"2023-03-01T17:54:26.333367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_single_sample(image_path,image_size=256,training=False,display=False):\n    '''\n    Return a single 256x256 sample\n    with possibility of returning a gleason score using the masks\n    '''\n    \n    image = openslide.OpenSlide(image_path)\n    \n    mask_path = image_path.replace(\"train_images\",\"train_label_masks\").replace(\".tiff\",\"_mask.tiff\")\n    mask = openslide.OpenSlide(mask_path)\n    \n    stacked_image = []\n    groundtruth_per_image = []\n    \n    maximum_iteration = 0\n    selected_sample = False\n    while not selected_sample:\n        sampling_start_x = randint(image_size,image.dimensions[0]-image_size)\n        sampling_start_y = randint(image_size,image.dimensions[1]-image_size)\n\n        clipped_sample = image.read_region((sampling_start_x, sampling_start_y), 0, (256, 256))\n        clipped_array = np.asarray(clipped_sample)\n        \n        #check that the sample is not empty\n        #and use the standard deviation to make sure\n        #there is something happening in the sample\n        if (not np.all(clipped_array==255) and np.std(clipped_array)>20) or maximum_iteration>200:\n            if display:\n                plt.imshow(clipped_sample)\n                plt.show()\n                \n            sampled_image = clipped_array[:,:,:3]\n            \n            if training:\n                clipped_mask = mask.read_region((sampling_start_x, sampling_start_y), 0, (256, 256))\n                groundtruth_per_image.append(np.mean(np.asarray(clipped_mask)[:,:,0]))\n            \n            selected_sample = True\n        maximum_iteration+=1\n    \n    if training: \n        return np.array(sampled_image), np.array(groundtruth_per_image)\n    else:\n        return np.array(sampled_image)","metadata":{"execution":{"iopub.status.busy":"2023-03-01T17:54:26.336109Z","iopub.execute_input":"2023-03-01T17:54:26.336637Z","iopub.status.idle":"2023-03-01T17:54:26.347623Z","shell.execute_reply.started":"2023-03-01T17:54:26.336602Z","shell.execute_reply":"2023-03-01T17:54:26.346486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_random_samples(image_path,image_size=256,display=False):\n    '''\n    Load an image and select random areas.\n    Return a list of 3 images from areas where there is data.\n    '''\n    \n    image = openslide.OpenSlide(image_path)\n    stacked_image = []\n    \n    selected_samples = 0\n    maximum_iteration = 0\n    while selected_samples<3:\n        sampling_start_x = randint(image_size,image.dimensions[0]-image_size)\n        sampling_start_y = randint(image_size,image.dimensions[1]-image_size)\n\n        clipped_sample = image.read_region((sampling_start_x, sampling_start_y), 0, (256, 256))\n        clipped_array = np.asarray(clipped_sample)\n        \n        #check that the sample is not empty\n        #and use the standard deviation to make sure\n        #there is something happening in the sample\n        if (not np.all(clipped_array==255) and np.std(clipped_array)>20) or maximum_iteration>200:\n            if display:\n                plt.imshow(clipped_sample)\n                plt.show()\n\n            stacked_image.append(clipped_array[:,:,:3])\n            selected_samples+=1\n        maximum_iteration+=1\n    return np.array(stacked_image)","metadata":{"execution":{"iopub.status.busy":"2023-03-01T17:54:26.349332Z","iopub.execute_input":"2023-03-01T17:54:26.350134Z","iopub.status.idle":"2023-03-01T17:54:26.359091Z","shell.execute_reply.started":"2023-03-01T17:54:26.350088Z","shell.execute_reply":"2023-03-01T17:54:26.358354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"get_random_samples(train_df.iloc[0].image_path).shape","metadata":{"execution":{"iopub.status.busy":"2023-03-01T17:54:26.360815Z","iopub.execute_input":"2023-03-01T17:54:26.361333Z","iopub.status.idle":"2023-03-01T17:54:27.049435Z","shell.execute_reply.started":"2023-03-01T17:54:26.361272Z","shell.execute_reply":"2023-03-01T17:54:27.048542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_ = get_random_samples(train_df.iloc[0].image_path, display=True)","metadata":{"execution":{"iopub.status.busy":"2023-03-01T17:54:27.053299Z","iopub.execute_input":"2023-03-01T17:54:27.054262Z","iopub.status.idle":"2023-03-01T17:54:28.303478Z","shell.execute_reply.started":"2023-03-01T17:54:27.054221Z","shell.execute_reply":"2023-03-01T17:54:28.302544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output = get_single_sample(train_df.iloc[0].image_path, display=True, training=True)\nprint(output[1])","metadata":{"execution":{"iopub.status.busy":"2023-03-01T17:54:28.307646Z","iopub.execute_input":"2023-03-01T17:54:28.308268Z","iopub.status.idle":"2023-03-01T17:54:28.97008Z","shell.execute_reply.started":"2023-03-01T17:54:28.308237Z","shell.execute_reply":"2023-03-01T17:54:28.968944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def custom_single_image_generator(image_path_list, batch_size=16):\n    '''\n    return an image and a corresponding gleason score from the mask\n    '''\n    \n    while True:\n        for start in range(0, len(image_path_list), batch_size):\n            X_batch = []\n            Y_batch = []\n            end = min(start + batch_size, training_item_count)\n\n            image_info_list = [get_single_sample(image_path, training=True) for image_path in image_path_list[start:end]]\n            X_batch = np.array([image_info[0]/255. for image_info in image_info_list])\n            Y_batch = np.array([image_info[1] for image_info in image_info_list])\n            \n            yield X_batch, Y_batch","metadata":{"execution":{"iopub.status.busy":"2023-03-01T17:54:28.971817Z","iopub.execute_input":"2023-03-01T17:54:28.972209Z","iopub.status.idle":"2023-03-01T17:54:28.980138Z","shell.execute_reply.started":"2023-03-01T17:54:28.972171Z","shell.execute_reply":"2023-03-01T17:54:28.978974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_channel = 3\nimage_shape = (image_size, image_size, num_channel)\n\ndef branch(input_image):\n    x = Conv2D(128, (3, 3))(input_image)\n    x = BatchNormalization()(x)\n    x = Activation('relu')(x)\n    x = Conv2D(128, (3, 3))(x)\n    x = BatchNormalization()(x)\n    x = Activation('relu')(x)\n    x = Conv2D(128, (3, 3))(x)\n    x = BatchNormalization()(x)\n    x = Activation('relu')(x)\n    x = MaxPooling2D(pool_size=(2, 2))(x)\n    \n    x = Conv2D(64, (3, 3))(x)\n    x = BatchNormalization()(x)\n    x = Activation('relu')(x)\n    x = Conv2D(64, (3, 3))(x)\n    x = BatchNormalization()(x)\n    x = Activation('relu')(x)\n    x = Conv2D(64, (3, 3))(x)\n    x = BatchNormalization()(x)\n    x = Activation('relu')(x)\n    x = MaxPooling2D(pool_size=(2, 2))(x)\n    \n    x = Conv2D(32, (3, 3))(x)\n    x = BatchNormalization()(x)\n    x = Activation('relu')(x)\n    x = Conv2D(32, (3, 3))(x)\n    x = BatchNormalization()(x)\n    x = Activation('relu')(x)\n    x = Conv2D(32, (3, 3))(x)\n    x = BatchNormalization()(x)\n    x = Activation('relu')(x)\n    x = GlobalAveragePooling2D()(x)\n    \n    x = layers.Dense(256)(x)\n    x = Activation('relu')(x)\n    \n    return layers.Dropout(0.3)(x)","metadata":{"execution":{"iopub.status.busy":"2023-03-01T17:54:28.981723Z","iopub.execute_input":"2023-03-01T17:54:28.983881Z","iopub.status.idle":"2023-03-01T17:54:28.996324Z","shell.execute_reply.started":"2023-03-01T17:54:28.983831Z","shell.execute_reply":"2023-03-01T17:54:28.995315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input_image = Input(shape=image_shape)\ncore_branch = branch(input_image)\noutput = Dense(1, activation='linear')(core_branch)\n\nbranch_model = Model(input_image,output)","metadata":{"execution":{"iopub.status.busy":"2023-03-01T17:54:28.999594Z","iopub.execute_input":"2023-03-01T17:54:29.000002Z","iopub.status.idle":"2023-03-01T17:54:31.739816Z","shell.execute_reply.started":"2023-03-01T17:54:28.999974Z","shell.execute_reply":"2023-03-01T17:54:31.738756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"branch_model.compile(loss=\"mse\",\n                      optimizer=tf.keras.optimizers.RMSprop(learning_rate=0.001))\ncallbacks = [ReduceLROnPlateau(monitor='val_loss', patience=1, verbose=1, factor=0.5),\n             EarlyStopping(monitor='val_loss', patience=3),\n             ModelCheckpoint(filepath='best_branch.h5', monitor='val_loss', save_best_only=True)]\n\nbatch_size = 16\n\nhistory = branch_model.fit(custom_single_image_generator(train_df[\"image_path\"], batch_size=batch_size),\n                        steps_per_epoch = int(len(train_df)/batch_size),\n                        validation_data=custom_single_image_generator(validation_df[\"image_path\"], batch_size=batch_size),\n                        validation_steps= int(len(validation_df)/batch_size),\n                        epochs=2,\n                        callbacks=callbacks)","metadata":{"execution":{"iopub.status.busy":"2023-03-01T17:54:31.741532Z","iopub.execute_input":"2023-03-01T17:54:31.741914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def custom_generator(image_path_list, groundtruth_list, batch_size=16):\n    num_classes=6\n    while True:\n        for start in range(0, len(image_path_list), batch_size):\n            X_batch = []\n            Y_batch = []\n            end = min(start + batch_size, training_item_count)\n            \n            X_batch = np.array([get_random_samples(image_path)/255. for image_path in image_path_list[start:end]])\n            input_image1 = X_batch[:,0,:,:]\n            input_image2 = X_batch[:,1,:,:]\n            input_image3 = X_batch[:,2,:,:]\n            \n            Y_batch = tf.keras.utils.to_categorical(np.array(groundtruth_list[start:end]),num_classes) \n            \n            yield [input_image1,input_image2,input_image3], Y_batch","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def input_branch(input_image):\n    '''\n    Generate a new input branch using our previous weights\n    \n    '''\n    input_image = Input(shape=image_shape)\n    core_branch = branch(input_image)\n    output = Dense(1, activation='linear')(core_branch)\n    branch_model = Model(input_image,output)\n    branch_model.load_weights(\"../working/best_branch.h5\")\n        \n    new_branch = Model(inputs=branch_model.input, outputs=branch_model.layers[-2].output)\n    \n    for layer in new_branch.layers[:-3]:\n        layer.trainable = False\n    \n    return new_branch","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}