{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import openslide\nfrom openslide.deepzoom import DeepZoomGenerator\nfrom matplotlib import pyplot as plt\nfrom IPython.display import FileLink\n\n\nimport os\nimport numpy as np\nimport pandas as pd\n\nfrom tensorflow.keras.models import load_model\nimport cv2\nimport gc\nfrom tqdm import tqdm\nfrom PIL import Image","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train_directory = os.path.join(\"../input/mayo-clinic-strip-ai/\", \"train/\")\ntest_directory = os.path.join(\"../input/mayo-clinic-strip-ai/\", \"test/\")\n\n\n#train_df = pd.read_csv(\"../input/mayo-clinic-strip-ai/train.csv\")\n#train_df[\"image_path\"] = train_df[\"image_id\"].apply(lambda x: os.path.join(train_directory, x+\".tif\"))\n#train_df.head(5)\ntest_df = pd.read_csv(\"../input/mayo-clinic-strip-ai/test.csv\")\ntest_df[\"image_path\"] = test_df[\"image_id\"].apply(lambda x: os.path.join(test_directory, x+\".tif\"))\ntest_df.head(5)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir resized","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def new_function(slide):\n    #slide = openslide.open_slide(train_df['image_path'].iloc[0])\n    \n    tiles = DeepZoomGenerator(slide, tile_size = 256, overlap = 0, limit_bounds = False)\n    \n    max_len = len(tiles.level_tiles)-1\n    cols, rows = tiles.level_tiles[max_len]\n    \n    for row in range(rows):\n        for col in range(cols):\n            name = test_df['image_id'].iloc[i] + '_' + '%d_%d' % (col, row)\n            tile_name = os.path.join('./resized/', name)\n            temp_tile = tiles.get_tile(max_len, (col, row))\n            temp_tile_RGB = temp_tile.convert('RGB')\n            temp_tile_np = np.array(temp_tile_RGB)\n            \n            if temp_tile_np.mean() <230 and temp_tile_np.std() >15:\n                #print('Processing tile number: ', tile_name)\n                plt.imsave(tile_name + '.jpg', temp_tile_np)\n            else:\n                #print('Not processing tile: ', tile_name)\n                print(i)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"length = test_df.shape[0]\n\nfor i in range(length):\n    slide = openslide.open_slide(test_df['image_path'].iloc[i])\n    new_function(slide)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(os.listdir('./resized'))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#from IPython.display import Image\n#Image(filename = './resized/006388_0_46_186.jpg')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_paths = []\nfor i in os.listdir('./resized'):\n    #print(i)\n    tmp = os.path.join('./resized/', i)\n    new_paths.append(tmp)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(new_paths)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_paths[0]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = pd.DataFrame(new_paths, columns = ['file_name'])\nx[['A', 'B', 'C']] = x['file_name'].str.split('/', expand = True)\nx[['patient_id', 'E', 'F', 'G']] = x['C'].str.split('_', expand = True)\nx.drop(['A', 'B', 'C', 'E', 'F', 'G'], axis=1, inplace=True)\nx","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### https://stackoverflow.com/questions/22472213/python-random-selection-per-group\n### HERE I'M SELECTING A RANDOM NUMBER OF IMAGES CONDITIONAL ON THE PATIENT_ID.\n### SO THAT THERE'LL BE IMAGES (THE SAME NUMBER) FOR ALL PATIENTS\n### I DO THIS TO SAVE TIME/RESOURCES. THERE ARE A LOT OF TILES, SO RUNNING THE MODEL FOR EACH TILE WILL BE VERY CONSUMING.\n\n\nsize = 100      # sample size\nreplace = True  # with replacement\nfn = lambda obj: obj.loc[np.random.choice(obj.index, size, replace),:]\nx = x.groupby('patient_id', as_index = False).apply(fn)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_paths = x['file_name'].values","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = load_model('../input/modelaug15/weights-vgg16.h5')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#img_tmp = cv2.imread('./resized/006388_0_46_186.jpg')\n#img_tmp = cv2.resize(img_tmp, (image_size_height, image_size_width))\n#img_tmp = img_tmp/255\n#img_tmp = np.reshape(img_tmp, [1, image_size_height, image_size_width, 3])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# REMEMBER TO UPDATE THE SIZE PARAMETERS ACCORDING TO THE MODEL REQUIREMENTS.\n# 1. CREATE AN EMPTY LIST IN WHICH WE'LL STORE THE PREDICTIONS.\n# 2. READ THE PATHS FROM THE 'NEW_PATHS' (RESIZED) LIST WE CREATED WITH THE DIRECTORIES OF THE RESIZED IMAGES.\n# 3. OPEN THE IMAGES, RESIZE AGAIN ACCORDING TO THE SPECIFICATIONS OF THE MODEL.\n# 4. PRE-PROCESS ACCORDING TO THE SPECIFICATIONS OF THE MODEL (IN THIS CASE JUST STANDARDIZE, /255).\n# 5. MAKE PREDICTIONS FOR EACH IMAGE AND STORE THE PREDICTIONS IN THE LIST.\n# 6. DELETE THE TEMPORARY OBJECTS AND FREE MEMORY.\n# 7. IF THE OUPUT OF print(gc.collect()) IS A NUMBER, THEN IT'S FREEING MEMORY.\n#.   NOW THAT WE HAVE REDUCED THE SIZE OF THE IMAGES (SEE ABOVE) PROBABLY NOT MUCH NEEDED TO FREE MEMORY HERE, \n#.   BUT JUST IN CASE!\n# THIS WILL BE AN ITERATIVE PROCESS (LOOP) FOR EACH IMAGE IN THE TEST_DF (HERE 4; THERE'LL BE MORE IN THE EVALUATION).\n\n\nimage_size_height = 224\nimage_size_width  = 224\n\n\nlength = len(new_paths)\n\n\npreds = []\nfor img_path in tqdm(new_paths, total = length):\n    #img_tmp = tifi.imread(img_path)\n    img_tmp = cv2.imread(img_path)\n    img_tmp = cv2.resize(img_tmp, (image_size_height, image_size_width))\n    img_tmp = img_tmp/255\n    img_tmp = np.reshape(img_tmp, [1, image_size_height, image_size_width, 3])\n    \n    pred_tmp = model.predict(img_tmp)\n    \n    preds.append(pred_tmp)\n    \n    del img_tmp  # to free memory\n    del pred_tmp # to free memory\n    gc.collect() # to free memory\n    #print(gc.collect())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#x = pd.DataFrame(new_paths, columns = ['file_name'])\n#x[['A', 'B', 'C']] = x['file_name'].str.split('/', expand = True)\n#x[['patient_id', 'E', 'F', 'G']] = x['C'].str.split('_', expand = True)\n#x.drop(['A', 'B', 'C', 'E', 'F', 'G'], axis=1, inplace=True)\n#x","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = test_df.merge(x, how = 'inner', on = 'patient_id')\ny","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds2 = pd.DataFrame(np.concatenate(preds), columns = ['CE', 'LAA'])\npreds2.head(10)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.concat([y.reset_index(drop = True), preds2.reset_index(drop = True)], axis = 1)\nsubmission.drop(['image_id', 'center_id', 'image_num', 'image_path', 'file_name'], axis=1, inplace=True)\nsubmission = submission.groupby(\"patient_id\").mean()\nsubmission = submission[[\"CE\", \"LAA\"]].round(6).reset_index()\nsubmission","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv(\"submission.csv\", index = False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}