{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":0.022738,"end_time":"2022-11-25T01:43:50.180306","exception":false,"start_time":"2022-11-25T01:43:50.157568","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-01-30T16:38:05.231388Z","iopub.execute_input":"2023-01-30T16:38:05.231754Z","iopub.status.idle":"2023-01-30T16:38:05.237659Z","shell.execute_reply.started":"2023-01-30T16:38:05.231723Z","shell.execute_reply":"2023-01-30T16:38:05.236574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- images scaled\n- increased the data augumentation techniques\n- used eff_vgg_stacked","metadata":{}},{"cell_type":"code","source":"from PIL import Image\nimport tensorflow as tf\nimport tensorflow.keras.preprocessing as image\nimport tensorflow_io as tfio\nfrom sklearn.model_selection import train_test_split\nimport cv2\nimport gc\nimport os\nfrom openslide import OpenSlide\nimport math\nimport tensorflow as tf\nfrom openslide import open_slide\nimport keras\nfrom openslide.deepzoom import DeepZoomGenerator\n\ninp_size=224","metadata":{"papermill":{"duration":7.712145,"end_time":"2022-11-25T01:43:57.896485","exception":false,"start_time":"2022-11-25T01:43:50.18434","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-01-30T16:38:05.526819Z","iopub.execute_input":"2023-01-30T16:38:05.527692Z","iopub.status.idle":"2023-01-30T16:38:05.533538Z","shell.execute_reply.started":"2023-01-30T16:38:05.52766Z","shell.execute_reply":"2023-01-30T16:38:05.5324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"simple_cnn = '/kaggle/input/simple-cnn/simple_cnn.h5'\neff_vgg_stacked = '/kaggle/input/effvgg/effb0-vgg-stacked.h5'\neff_res_stacked = '/kaggle/input/all-models/effb0_res151_stacked.h5'\nvgg_res_stacked='/kaggle/input/all-models/vgg16_res151_stacked.h5'\nvgg='/kaggle/input/all-models/vgg16.h5'\nres='/kaggle/input/all-models/Resnet152.h5'\neff='/kaggle/input/all-models/EfficientNetB0.h5'\n\n#models to ensemble\nl = [eff_res_stacked,vgg_res_stacked,eff,res,vgg,eff_vgg_stacked,simple_cnn]","metadata":{"execution":{"iopub.status.busy":"2023-01-30T16:38:07.039349Z","iopub.execute_input":"2023-01-30T16:38:07.039725Z","iopub.status.idle":"2023-01-30T16:38:07.047644Z","shell.execute_reply.started":"2023-01-30T16:38:07.039694Z","shell.execute_reply":"2023-01-30T16:38:07.046557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models=[]\nfor i in l:\n    models.append(keras.models.load_model(i))","metadata":{"execution":{"iopub.status.busy":"2023-01-30T16:38:17.079519Z","iopub.execute_input":"2023-01-30T16:38:17.080024Z","iopub.status.idle":"2023-01-30T16:39:06.773038Z","shell.execute_reply.started":"2023-01-30T16:38:17.079992Z","shell.execute_reply":"2023-01-30T16:39:06.771875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models","metadata":{"execution":{"iopub.status.busy":"2023-01-30T16:39:33.732875Z","iopub.execute_input":"2023-01-30T16:39:33.733264Z","iopub.status.idle":"2023-01-30T16:39:33.742251Z","shell.execute_reply.started":"2023-01-30T16:39:33.733231Z","shell.execute_reply":"2023-01-30T16:39:33.741235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in models:\n    i.compile(\n        optimizer='adam',\n        loss='categorical_crossentropy',\n        metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2023-01-30T16:39:37.579937Z","iopub.execute_input":"2023-01-30T16:39:37.580417Z","iopub.status.idle":"2023-01-30T16:39:37.641351Z","shell.execute_reply.started":"2023-01-30T16:39:37.580374Z","shell.execute_reply":"2023-01-30T16:39:37.640484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.image import ImageDataGenerator\ndatagen = ImageDataGenerator(\n    rescale=1./255,\n    rotation_range=45, \n    width_shift_range=0.2,\n    height_shift_range=0.2,\n    brightness_range=[0.5, 1.0],\n    zoom_range=[0.5, 1.0],\n    horizontal_flip=True, \n    vertical_flip=True ,\n    validation_split=0.1\n)\n\nbatch_size=32\ndirectory='/kaggle/input/full-data/working'\n\ntrain_generator = datagen.flow_from_directory(\n    directory,\n    target_size=(224, 224),\n    batch_size=batch_size,\n    class_mode='categorical',\n    subset='training',\n)\n\nvalidation_generator = datagen.flow_from_directory(\n    directory,\n    target_size=(224, 224),\n    batch_size=batch_size,\n    class_mode='categorical',\n    subset='validation',\n)","metadata":{"execution":{"iopub.status.busy":"2023-01-30T16:40:51.539624Z","iopub.execute_input":"2023-01-30T16:40:51.539992Z","iopub.status.idle":"2023-01-30T16:40:52.150114Z","shell.execute_reply.started":"2023-01-30T16:40:51.539952Z","shell.execute_reply":"2023-01-30T16:40:52.14814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"histories=[]\nfor i in models:\n    his=i.fit(\n        train_generator,\n        steps_per_epoch=len(train_generator),\n        epochs=30,\n        validation_data=validation_generator,\n    )\n    histories.append(his)","metadata":{"execution":{"iopub.status.busy":"2023-01-30T16:41:06.975414Z","iopub.execute_input":"2023-01-30T16:41:06.976394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n\nfor h in histories:\n    acc = h.history['accuracy']\n    val_acc = h.history['val_accuracy']\n    loss = h.history['loss']\n    val_loss = h.history['val_loss']\n\n    epochs_range = range(30)\n\n    plt.figure(figsize=(12, 8))\n    plt.subplot(1, 2, 1)\n    plt.plot(epochs_range, acc, label='Training Accuracy')\n    plt.plot(epochs_range, val_acc, label='Validation Accuracy')\n    plt.legend(loc='lower right')\n    plt.title('Training and Validation Accuracy')\n\n    plt.subplot(1, 2, 2)\n    plt.plot(epochs_range, loss, label='Training Loss')\n    plt.plot(epochs_range, val_loss, label='Validation Loss')\n    plt.legend(loc='upper right')\n    plt.title('Training and Validation Loss')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-30T16:31:57.872797Z","iopub.execute_input":"2023-01-30T16:31:57.873156Z","iopub.status.idle":"2023-01-30T16:31:58.209112Z","shell.execute_reply.started":"2023-01-30T16:31:57.873126Z","shell.execute_reply":"2023-01-30T16:31:58.208159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_indices = train_generator.class_indices\nprint(class_indices)","metadata":{"execution":{"iopub.status.busy":"2023-01-30T16:08:39.70705Z","iopub.execute_input":"2023-01-30T16:08:39.707429Z","iopub.status.idle":"2023-01-30T16:08:39.713204Z","shell.execute_reply.started":"2023-01-30T16:08:39.707395Z","shell.execute_reply":"2023-01-30T16:08:39.712173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# model testing","metadata":{"execution":{"iopub.status.busy":"2023-01-29T18:16:04.829109Z","iopub.execute_input":"2023-01-29T18:16:04.829525Z","iopub.status.idle":"2023-01-29T18:16:05.01789Z","shell.execute_reply.started":"2023-01-29T18:16:04.829491Z","shell.execute_reply":"2023-01-29T18:16:05.016405Z"}}},{"cell_type":"code","source":"def make_test_file(x):\n    return \"../input/mayo-clinic-strip-ai/test/\" + x + \".tif\"\ntest = pd.read_csv('../input/mayo-clinic-strip-ai/test.csv')\ntest.head()\ntest_data = pd.DataFrame({'image_id': test.image_id.apply(make_test_file)})\ntest_data.head()\npath = '../input/mayo-clinic-strip-ai/test/'","metadata":{"execution":{"iopub.status.busy":"2023-01-30T16:08:40.442926Z","iopub.execute_input":"2023-01-30T16:08:40.443699Z","iopub.status.idle":"2023-01-30T16:08:40.465653Z","shell.execute_reply.started":"2023-01-30T16:08:40.443657Z","shell.execute_reply":"2023-01-30T16:08:40.464798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data","metadata":{"execution":{"iopub.status.busy":"2023-01-30T16:08:40.828152Z","iopub.execute_input":"2023-01-30T16:08:40.828515Z","iopub.status.idle":"2023-01-30T16:08:40.842228Z","shell.execute_reply.started":"2023-01-30T16:08:40.828478Z","shell.execute_reply":"2023-01-30T16:08:40.841393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def norm_HnE(img, Io=250, alpha=1, beta=0.15):\n\n\n    ######## Step 1: Convert RGB to OD ###################\n    ## reference H&E OD matrix.\n    #Can be updated if you know the best values for your image. \n    #Otherwise use the following default values. \n    #Read the above referenced papers on this topic. \n    HERef = np.array([[0.5626, 0.2159],\n                      [0.7201, 0.8012],\n                      [0.4062, 0.5581]])\n    ### reference maximum stain concentrations for H&E\n    maxCRef = np.array([1.9705, 1.0308])\n    \n    \n    # extract the height, width and num of channels of image\n    h, w, c = img.shape\n    \n    # reshape image to multiple rows and 3 columns.\n    #Num of rows depends on the image size (wxh)\n    img = img.reshape((-1,3))\n    \n    # calculate optical density\n    # OD = −log10(I)  \n    #OD = -np.log10(img+0.004)  #Use this when reading images with skimage\n    #Adding 0.004 just to avoid log of zero. \n    \n    OD = -np.log10((img.astype(np.float)+1)/Io) #Use this for opencv imread\n    #Add 1 in case any pixels in the image have a value of 0 (log 0 is indeterminate)\n    \n    \n    ############ Step 2: Remove data with OD intensity less than β ############\n    # remove transparent pixels (clear region with no tissue)\n    ODhat = OD[~np.any(OD < beta, axis=1)] #Returns an array where OD values are above beta\n    #Check by printing ODhat.min()\n    \n    ############# Step 3: Calculate SVD on the OD tuples ######################\n    #Estimate covariance matrix of ODhat (transposed)\n    # and then compute eigen values & eigenvectors.\n    eigvals, eigvecs = np.linalg.eigh(np.cov(ODhat.T))\n    \n    \n    ######## Step 4: Create plane from the SVD directions with two largest values ######\n    #project on the plane spanned by the eigenvectors corresponding to the two \n    # largest eigenvalues    \n    That = ODhat.dot(eigvecs[:,1:3]) #Dot product\n    \n    ############### Step 5: Project data onto the plane, and normalize to unit length ###########\n    ############## Step 6: Calculate angle of each point wrt the first SVD direction ########\n    #find the min and max vectors and project back to OD space\n    phi = np.arctan2(That[:,1],That[:,0])\n    \n    minPhi = np.percentile(phi, alpha)\n    maxPhi = np.percentile(phi, 100-alpha)\n    \n    vMin = eigvecs[:,1:3].dot(np.array([(np.cos(minPhi), np.sin(minPhi))]).T)\n    vMax = eigvecs[:,1:3].dot(np.array([(np.cos(maxPhi), np.sin(maxPhi))]).T)\n    \n    \n    # a heuristic to make the vector corresponding to hematoxylin first and the \n    # one corresponding to eosin second\n    if vMin[0] > vMax[0]:    \n        HE = np.array((vMin[:,0], vMax[:,0])).T\n        \n    else:\n        HE = np.array((vMax[:,0], vMin[:,0])).T\n    \n    \n    # rows correspond to channels (RGB), columns to OD values\n    Y = np.reshape(OD, (-1, 3)).T\n    \n    # determine concentrations of the individual stains\n    C = np.linalg.lstsq(HE,Y, rcond=None)[0]\n    \n    # normalize stain concentrations\n    maxC = np.array([np.percentile(C[0,:], 99), np.percentile(C[1,:],99)])\n    tmp = np.divide(maxC,maxCRef)\n    C2 = np.divide(C,tmp[:, np.newaxis])\n    \n    ###### Step 8: Convert extreme values back to OD space\n    # recreate the normalized image using reference mixing matrix \n    \n    Inorm = np.multiply(Io, np.exp(-HERef.dot(C2)))\n    Inorm[Inorm>255] = 254\n    Inorm = np.reshape(Inorm.T, (h, w, 3)).astype(np.uint8)  \n    \n    # Separating H and E components\n    \n    H = np.multiply(Io, np.exp(np.expand_dims(-HERef[:,0], axis=1).dot(np.expand_dims(C2[0,:], axis=0))))\n    H[H>255] = 254\n    H = np.reshape(H.T, (h, w, 3)).astype(np.uint8)\n    \n    E = np.multiply(Io, np.exp(np.expand_dims(-HERef[:,1], axis=1).dot(np.expand_dims(C2[1,:], axis=0))))\n    E[E>255] = 254\n    E = np.reshape(E.T, (h, w, 3)).astype(np.uint8)\n    \n    return (Inorm, H, E)","metadata":{"execution":{"iopub.status.busy":"2023-01-30T16:08:41.411939Z","iopub.execute_input":"2023-01-30T16:08:41.412293Z","iopub.status.idle":"2023-01-30T16:08:41.429074Z","shell.execute_reply.started":"2023-01-30T16:08:41.412264Z","shell.execute_reply":"2023-01-30T16:08:41.428095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds=[]\nfor x in range(int(test_data.size)):\n    img_path = test_data.image_id[x]\n    slide = open_slide(img_path)\n    tiles=DeepZoomGenerator(slide,tile_size=inp_size,overlap=0,limit_bounds=False)\n    cols,rows = tiles.level_tiles[tiles.level_count-1]\n    temp_preds=[]\n    for row in range(0,rows,15):\n        for col in range(0,cols,15):\n            tile=tiles.get_tile(tiles.level_count-1,(col,row))\n            tile=tile.convert(\"RGB\")\n            tile=np.array(tile)\n            try:\n                if tile.mean()<180 and tile.std()>60:\n                    norm_img,H,E= norm_HnE(tile)\n                    norm_img = np.reshape(norm_img, [1,inp_size, inp_size, 3])\n                    \n                    p=[i.predict(norm_img/255) for i in models]\n                    print(p)\n                    t_p = sum(p)/len(p)\n                    print(t_p)\n                    \n                    \n                    temp_preds.append(t_p)\n                    \n            except:pass        \n            gc.collect()\n            del tile\n        gc.collect()\n        \n        \n    \n    if len(temp_preds) > 0:\n        preds.append(sum(temp_preds)/len(temp_preds))\n    else:\n        preds.append([[0.5,0.5]])\n    del slide\n    del temp_preds\n    del tiles\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-01-30T16:12:17.507989Z","iopub.execute_input":"2023-01-30T16:12:17.508349Z","iopub.status.idle":"2023-01-30T16:14:27.520841Z","shell.execute_reply.started":"2023-01-30T16:12:17.508319Z","shell.execute_reply":"2023-01-30T16:14:27.519805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds","metadata":{"execution":{"iopub.status.busy":"2023-01-30T16:11:00.043408Z","iopub.execute_input":"2023-01-30T16:11:00.043752Z","iopub.status.idle":"2023-01-30T16:11:00.053924Z","shell.execute_reply.started":"2023-01-30T16:11:00.043719Z","shell.execute_reply":"2023-01-30T16:11:00.052725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = pd.DataFrame(np.concatenate(preds))\nsubmission = pd.read_csv('../input/mayo-clinic-strip-ai/sample_submission.csv')\nsubmission.CE = preds.iloc[ : , : 1]\nsubmission.LAA = preds.iloc[ : , 1: 2]\nsubmission = submission.groupby(\"patient_id\").mean()\nsubmission = submission[[\"CE\", \"LAA\"]].round(6).reset_index()\nsubmission.fillna(0.5)\nsubmission","metadata":{"papermill":{"duration":0.231203,"end_time":"2022-11-25T02:39:25.804291","exception":false,"start_time":"2022-11-25T02:39:25.573088","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-01-30T16:11:00.055652Z","iopub.execute_input":"2023-01-30T16:11:00.056311Z","iopub.status.idle":"2023-01-30T16:11:00.086186Z","shell.execute_reply.started":"2023-01-30T16:11:00.056275Z","shell.execute_reply":"2023-01-30T16:11:00.085083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission[[\"patient_id\", \"CE\", \"LAA\"]].to_csv(\"submission.csv\", index=False)\n!head submission.csv","metadata":{"papermill":{"duration":1.394655,"end_time":"2022-11-25T02:39:27.379861","exception":false,"start_time":"2022-11-25T02:39:25.985206","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-01-30T16:11:00.088317Z","iopub.execute_input":"2023-01-30T16:11:00.088745Z","iopub.status.idle":"2023-01-30T16:11:01.162254Z","shell.execute_reply.started":"2023-01-30T16:11:00.08871Z","shell.execute_reply":"2023-01-30T16:11:01.161109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}