{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport sys\n\nimport time\n\nimport numpy as np\nimport pandas as pd\n\nimport cv2\nimport tifffile\nimport openslide\nfrom PIL import Image\n\nfrom pathlib import Path\n\nfrom pprint import pprint\n\nimport matplotlib.pyplot as plt\nplt.style.use('seaborn')\n\nimport seaborn as sns\n\nfrom tqdm.notebook import tqdm\n\nfrom pandarallel import pandarallel\nimport multiprocessing\nimport concurrent.futures\n\n\n# LOGGER\nimport logging\n\n# Create a custom logger\nlogger = logging.getLogger(__name__)\n\n# Create handlers\nc_handler = logging.StreamHandler()\nc_handler.setLevel(logging.DEBUG)\nc_handler.setFormatter(logging.Formatter('%(name)s - %(levelname)s - %(message)s'))\n\nlogger.handlers = [c_handler]\nlogger.setLevel(logging.INFO)\nlogger.info('This is an Info')\nlogger.debug('This is an Debug')","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2022-08-14T21:14:23.626888Z","iopub.execute_input":"2022-08-14T21:14:23.627335Z","iopub.status.idle":"2022-08-14T21:14:23.647679Z","shell.execute_reply.started":"2022-08-14T21:14:23.627296Z","shell.execute_reply":"2022-08-14T21:14:23.646641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"multiprocessing.cpu_count()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T21:02:38.348283Z","iopub.execute_input":"2022-08-14T21:02:38.348935Z","iopub.status.idle":"2022-08-14T21:02:38.357431Z","shell.execute_reply.started":"2022-08-14T21:02:38.348895Z","shell.execute_reply":"2022-08-14T21:02:38.35631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not os.path.exists(\"./tiles\"):\n    os.mkdir(\"./tiles\")","metadata":{"execution":{"iopub.status.busy":"2022-08-14T21:12:58.299668Z","iopub.execute_input":"2022-08-14T21:12:58.300431Z","iopub.status.idle":"2022-08-14T21:12:58.304638Z","shell.execute_reply.started":"2022-08-14T21:12:58.300392Z","shell.execute_reply":"2022-08-14T21:12:58.303695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Config:\n    \n    do_generate_tiles = True\n    sanity_check = False\n    \n    ","metadata":{"execution":{"iopub.status.busy":"2022-08-14T21:18:31.275911Z","iopub.execute_input":"2022-08-14T21:18:31.276571Z","iopub.status.idle":"2022-08-14T21:18:31.281957Z","shell.execute_reply.started":"2022-08-14T21:18:31.276532Z","shell.execute_reply":"2022-08-14T21:18:31.28079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def resize(img, new_size):\n    \n    h, w = img.shape[:2]\n    \n    ratio = w / h\n\n    if ratio < 1: # width small\n        r1 = new_size / w\n        new_w = new_size\n        new_h = h * r1\n    else: # height small\n        r1 = new_size / h\n        new_w = w * r1\n        new_h = new_size\n    \n    img = cv2.resize(img, (int(new_w), int(new_h)))\n\n    return img\n\ndef read_tiff_thumb(path, size=512.0):\n    img = tifffile.imread(path)\n    img = resize(img, size)\n    return Image.fromarray(img)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T21:02:38.371332Z","iopub.execute_input":"2022-08-14T21:02:38.372412Z","iopub.status.idle":"2022-08-14T21:02:38.380443Z","shell.execute_reply.started":"2022-08-14T21:02:38.372376Z","shell.execute_reply":"2022-08-14T21:02:38.37946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(\"../input/mayo-clinic-strip-ai/train.csv\")\ntest_df = pd.read_csv(\"../input/mayo-clinic-strip-ai/test.csv\")\nother_df = pd.read_csv(\"../input/mayo-clinic-strip-ai/other.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-08-14T21:02:38.381976Z","iopub.execute_input":"2022-08-14T21:02:38.382718Z","iopub.status.idle":"2022-08-14T21:02:38.412155Z","shell.execute_reply.started":"2022-08-14T21:02:38.382662Z","shell.execute_reply":"2022-08-14T21:02:38.41114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T21:02:38.415051Z","iopub.execute_input":"2022-08-14T21:02:38.415849Z","iopub.status.idle":"2022-08-14T21:02:38.433663Z","shell.execute_reply.started":"2022-08-14T21:02:38.415794Z","shell.execute_reply":"2022-08-14T21:02:38.432745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T21:02:38.43664Z","iopub.execute_input":"2022-08-14T21:02:38.438684Z","iopub.status.idle":"2022-08-14T21:02:38.449532Z","shell.execute_reply.started":"2022-08-14T21:02:38.438652Z","shell.execute_reply":"2022-08-14T21:02:38.448411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"other_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T21:02:38.451322Z","iopub.execute_input":"2022-08-14T21:02:38.45206Z","iopub.status.idle":"2022-08-14T21:02:38.464033Z","shell.execute_reply.started":"2022-08-14T21:02:38.452025Z","shell.execute_reply":"2022-08-14T21:02:38.462946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(nrows=1, ncols=4, figsize=(20,4))\n\ntrain_df.label.value_counts().plot(kind='bar', ax=axes[1], title='Train Target')\ntrain_df.center_id.value_counts().plot(kind='bar', ax=axes[2], title='Centers')\n\nnumber_img, num_patients = np.unique(train_df.patient_id.value_counts().values, return_counts=True)\n\nsns.barplot(x=number_img, y=num_patients, ax=axes[3]).set_title('Images')\n\npd.DataFrame.from_dict([\n    {\n        \"Type\" : \"Train\",\n        \"count\": len(train_df)\n    },\n    {\n        \"Type\" : \"Test\",\n        \"count\": len(test_df)\n    },\n    {\n        \"Type\" : \"Other\",\n        \"count\": len(other_df)\n    }\n]).set_index('Type').plot(kind='bar', ax=axes[0], title='Df Size')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T21:02:38.465623Z","iopub.execute_input":"2022-08-14T21:02:38.466361Z","iopub.status.idle":"2022-08-14T21:02:39.113792Z","shell.execute_reply.started":"2022-08-14T21:02:38.466322Z","shell.execute_reply":"2022-08-14T21:02:39.112645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['image_path'] = train_df.image_id.apply(lambda x : f\"../input/mayo-clinic-strip-ai/train/{x}.tif\")","metadata":{"execution":{"iopub.status.busy":"2022-08-14T21:02:39.120529Z","iopub.execute_input":"2022-08-14T21:02:39.123594Z","iopub.status.idle":"2022-08-14T21:02:39.133126Z","shell.execute_reply.started":"2022-08-14T21:02:39.123542Z","shell.execute_reply":"2022-08-14T21:02:39.132093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2022-08-14T21:02:39.134608Z","iopub.execute_input":"2022-08-14T21:02:39.135342Z","iopub.status.idle":"2022-08-14T21:02:39.161717Z","shell.execute_reply.started":"2022-08-14T21:02:39.135291Z","shell.execute_reply":"2022-08-14T21:02:39.160789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def calculate_tiles(image_path, divider=5):\n    \n    logger.debug(f\"Image Path:{image_path}\")\n        \n    slide = openslide.OpenSlide(image_path)\n    logger.debug(f\"{dict(slide.properties)}\")\n    \n    width, height = slide.dimensions\n    min_size = min(width, height)\n#     tile_size = min_size // divider\n    tile_size = 1024\n    logger.debug(f\"Width:{width}, Height={height}, Min Size={min_size}, Tile Size={tile_size}\")\n    \n    xs, ys = width // tile_size, height // tile_size\n    logger.debug(f\"xs={xs}, ys={ys}\")\n    \n    tiles = []\n    \n    image_id = 0\n    \n    for y in range(ys):\n                        \n        for x in range(xs):\n                                    \n            y_start = y * tile_size\n            x_start = x * tile_size\n                \n            tiles.append({\n                \"_id\" : image_id,\n                \"start\" : (x_start,y_start),\n                \"size\" : (tile_size, tile_size),\n            })\n            \n            image_id += 1\n                    \n    slide.close()            \n    \n    return tiles\n            \ndef collect_tile(info):\n    \n    image_path = info['path']\n    \n    slide = openslide.OpenSlide(image_path)\n    \n    tile = slide.read_region(info['start'], 0, info['size']).convert('RGB').resize((224, 224),Image.BICUBIC)\n    \n    slide.close()\n    \n    return {\n        \"_id\" : info['_id'],\n        \"tile\" : tile\n    }\n    \n\ndef collect_all_tiles(path, run_parallel=False):\n    \n    tile_infos = calculate_tiles(path)\n    \n    for t in tile_infos:\n        t['path'] = path\n    \n    if run_parallel:\n        \n        with concurrent.futures.ThreadPoolExecutor(max_workers=4) as executor:\n            all_result = []\n            for result in executor.map(collect_tile, tile_infos):\n                all_result.append(result)\n                \n    else:\n        all_result = [collect_tile(info) for info in tile_infos]\n        \n    for r in all_result:\n        img = np.array(r['tile'])\n        r['mean'] = img.mean()\n        \n        \n    all_result.sort(key=lambda x : x['mean'])\n    \n    all_result = all_result[:20]\n    \n    all_result.sort(key=lambda x : x['_id'])\n    \n    return all_result","metadata":{"execution":{"iopub.status.busy":"2022-08-14T21:02:39.165966Z","iopub.execute_input":"2022-08-14T21:02:39.167388Z","iopub.status.idle":"2022-08-14T21:02:39.18851Z","shell.execute_reply.started":"2022-08-14T21:02:39.167352Z","shell.execute_reply":"2022-08-14T21:02:39.187488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if Config.sanity_check:\n    train_df = train_df[:10]\n    train_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-14T21:18:20.478829Z","iopub.execute_input":"2022-08-14T21:18:20.479227Z","iopub.status.idle":"2022-08-14T21:18:20.486386Z","shell.execute_reply.started":"2022-08-14T21:18:20.479191Z","shell.execute_reply":"2022-08-14T21:18:20.485407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if Config.do_generate_tiles:\n    \n    result_id_ls = []\n\n    for i in range(len(train_df)):\n        _start = time.time()\n\n        path = train_df.iloc[i].image_path\n\n        print(i, path)\n        result = collect_all_tiles(path, run_parallel=True)\n        result_ids = [f\"{r['_id']}-{r['mean']:.2f}\" for r in result]\n\n        # saving images\n        for r in result:\n            r['tile'].save(f\"./tiles/{Path(path).stem}.{r['_id']}.png\")\n        \n        \n        print(\"Time:\", time.time() - _start, \"Ids=\", result_ids)\n\n        result_id_ls.append([r['_id'] for r in result])\n        \n    print(result_id_ls[:5])\n    \n    train_df['tile_ids'] = result_id_ls\n    \n    display(train_df.head())\n    \n    train_df.to_csv(\"train_df_tiles.csv\")\n\nelse:\n    \n    train_tiles_df = pd.read_csv(\"../input/strip-ai-tiles/train_df_tiles.csv\")\n    train_tiles_df = train_tiles_df[['image_id','tile_ids']]\n    display(train_tiles_df.head())\n    \n    train_df = pd.merge(train_df, train_tiles_df, on=['image_id'])\n    display(train_df.head())","metadata":{"execution":{"iopub.status.busy":"2022-08-14T21:20:43.994101Z","iopub.execute_input":"2022-08-14T21:20:43.994549Z","iopub.status.idle":"2022-08-14T21:25:05.478352Z","shell.execute_reply.started":"2022-08-14T21:20:43.994509Z","shell.execute_reply":"2022-08-14T21:25:05.477474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}