{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<br>\n\n<br><center><img src=\"https://storage.googleapis.com/kaggle-competitions/kaggle/37333/logos/header.png\" width=100%></center>\n\n<h2 style=\"text-align: center; font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: underline; text-transform: none; letter-spacing: 2px; color: #E55CA0; background-color: #ffffff;\">Mayo Clinic - STRIP AI<br><br>No Tile Model</h2><br>\n<h5 style=\"text-align: center; font-family: Verdana; font-size: 12px; font-style: normal; font-weight: bold; text-decoration: None; text-transform: none; letter-spacing: 1px; color: black; background-color: #ffffff;\">CREATED BY: DARIEN SCHETTLER</h5>\n\n<br>\n\n---\n\n<br>\n\n<center><div class=\"alert alert-block alert-danger\" style=\"margin: 2em; line-height: 1.7em; font-family: Verdana;\">\n    <b style=\"font-size: 18px;\">🛑 &nbsp; WARNING:</b><br><br><b>THIS IS A WORK IN PROGRESS</b><br>\n</div></center>\n\n\n<center><div class=\"alert alert-block alert-warning\" style=\"margin: 2em; line-height: 1.7em; font-family: Verdana;\">\n    <b style=\"font-size: 18px;\">👏 &nbsp; IF YOU FORK THIS OR FIND THIS HELPFUL &nbsp; 👏</b><br><br><b style=\"font-size: 22px; color: darkorange\">PLEASE UPVOTE!</b><br><br>This was a lot of work for me and while it may seem silly, it makes me feel appreciated when others like my work. 😅\n</div></center>","metadata":{}},{"cell_type":"markdown","source":"<br>\n\n<a id=\"imports\"></a>\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; background-color: #ffffff; color: #E55CA0;\" id=\"imports\">0&nbsp;&nbsp;IMPORTS & INSTALLS&nbsp;&nbsp;&nbsp;&nbsp;<a href=\"#toc\">&#10514;</a></h1>\n\n<br>\n\n**PYVIPS INSTALL CODE ONLY WORKS PINNED TO ORIGINAL ENVIRONMENT (2020)**\n- Original: https://www.kaggle.com/hirune924/fast-image-region-loading-using-pyvips\n  - Thanks for @hirune924 !!\n- https://www.kaggle.com/code/yukkyo/using-pyvips-without-internet/comments\n- Install needs about 5 minutes","metadata":{}},{"cell_type":"code","source":"print(\"\\n... INSTALLING LOCAL VERSION OF PYVIPS! ...\")\n!dpkg -i --force-depends /kaggle/input/pyvips-local/libvips-apt/libvips-apt/*.deb >/dev/null 2>&1\n!pip install /kaggle/input/pyvips-local/cffi-1.14.4-cp37-cp37m-manylinux1_x86_64.whl\n!pip install /kaggle/input/pyvips-local/pycparser-2.20-py2.py3-none-any.whl\n!pip install /kaggle/input/pyvips-local/pyvips-2.1.13-py2.py3-none-any.whl\nimport pyvips\nprint(\"... INSTALL COMPLETE! ...\\n\")","metadata":{"execution":{"iopub.status.busy":"2022-07-13T22:18:43.237916Z","iopub.execute_input":"2022-07-13T22:18:43.238342Z","iopub.status.idle":"2022-07-13T22:21:03.931601Z","shell.execute_reply.started":"2022-07-13T22:18:43.238293Z","shell.execute_reply":"2022-07-13T22:21:03.930877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"\\n... IMPORTS STARTING ...\\n\")\n\nprint(\"\\n\\tVERSION INFORMATION\")\n\n# Machine Learning and Data Science Imports\nimport tensorflow as tf; print(f\"\\t\\t– TENSORFLOW VERSION: {tf.__version__}\");\nimport pandas as pd; pd.options.mode.chained_assignment = None;\nimport numpy as np; print(f\"\\t\\t– NUMPY VERSION: {np.__version__}\");\n\n# Built In Imports\nfrom glob import glob\nimport openslide\nimport warnings\nimport requests\nimport hashlib\nimport imageio\nimport IPython\nimport sklearn\nimport urllib\nimport zipfile\nimport pickle\nimport random\nimport shutil\nimport string\nimport json\nimport math\nimport time\nimport gzip\nimport ast\nimport sys\nimport io\nimport os\nimport gc\nimport re\n\n# Visualization Imports\nfrom matplotlib.colors import ListedColormap\nfrom matplotlib.patches import Rectangle\nimport matplotlib.patches as patches\nimport plotly.graph_objects as go\nimport matplotlib.pyplot as plt\nfrom tqdm.notebook import tqdm; tqdm.pandas();\nimport plotly.express as px\nimport tifffile as tif\nimport seaborn as sns\nfrom PIL import Image, ImageEnhance; Image.MAX_IMAGE_PIXELS = 5_000_000_000;\nimport matplotlib; print(f\"\\t\\t– MATPLOTLIB VERSION: {matplotlib.__version__}\");\nfrom matplotlib import animation, rc; rc('animation', html='jshtml')\nimport plotly\nimport PIL\nimport cv2\n\nimport plotly.io as pio\nprint(pio.renderers)\n\ndef seed_it_all(seed=7):\n    \"\"\" Attempt to be Reproducible \"\"\"\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    random.seed(seed)\n    np.random.seed(seed)\n    tf.random.set_seed(seed)\nseed_it_all()\n    \nprint(\"\\n\\n... IMPORTS COMPLETE ...\\n\")","metadata":{"execution":{"iopub.status.busy":"2022-07-13T22:21:03.936916Z","iopub.execute_input":"2022-07-13T22:21:03.937507Z","iopub.status.idle":"2022-07-13T22:21:12.037579Z","shell.execute_reply.started":"2022-07-13T22:21:03.937467Z","shell.execute_reply":"2022-07-13T22:21:12.036514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n<a id=\"background_information\"></a>\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; color: #E55CA0; background-color: #ffffff;\" id=\"setup\">1&nbsp;&nbsp;SETUP&nbsp;&nbsp;&nbsp;&nbsp;<a href=\"#toc\">&#10514;</a></h1>\n\n---\n","metadata":{}},{"cell_type":"code","source":"print(\"\\n... BASIC SETUP STARTING ...\\n\\n\")\n\n# Path to competition data\nDATA_DIR = \"/kaggle/input/mayo-clinic-strip-ai\"\n\n# enable XLA optmizations (10% speedup when using @tf.function calls)\ntf.config.optimizer.set_jit(True)\n\n# From dataset\nINPUT_SHAPE = (512,512,3)\nS2I_LBL_MAP = {\"CE\":0, \"LAA\":1}\nI2S_LBL_MAP = {v:k for k,v in S2I_LBL_MAP.items()}\nN_CLASSES=len(S2I_LBL_MAP)\n\n# Open the training dataframe and display the initial dataframe\nTRAIN_DIR = os.path.join(DATA_DIR, \"train\")\nTRAIN_CSV = os.path.join(DATA_DIR, \"train.csv\")\ntrain_df = pd.read_csv(TRAIN_CSV)\ntrain_df[\"image_path\"] = train_df[\"image_id\"].apply(lambda x: os.path.join(TRAIN_DIR, x+\".tif\"))\n\nprint(\"\\n... TRAINING DATAFRAME... \\n\")\ndisplay(train_df)\n\n# Open the testing dataframe and display the initial dataframe\nTEST_DIR = os.path.join(DATA_DIR, \"test\")\nTEST_CSV = os.path.join(DATA_DIR, \"test.csv\")\ntest_df = pd.read_csv(TEST_CSV)\ntest_df[\"image_path\"] = test_df[\"image_id\"].apply(lambda x: os.path.join(TEST_DIR, x+\".tif\"))\n\nprint(\"\\n... TESTING DATAFRAME... \\n\")\ndisplay(test_df)\n\n# Open the sample submission dataframe\nSS_CSV   = os.path.join(DATA_DIR, \"sample_submission.csv\")\nss_df = pd.read_csv(SS_CSV)\n\nprint(\"\\n... SAMPLE SUBMISSION DATAFRAME... \\n\")\ndisplay(ss_df)\n\n# For debugging purposes when the test set hasn't been substituted we will know\nDEBUG=len(ss_df)==4\nprint(f\"\\n\\n\\n... ARE WE DEBUGGING: {DEBUG}... \\n\")\n\nAPPROX_PRIVATE_TEST_LEN = 280\nif DEBUG:\n    print(\"\\n\\n... MODIFYING TEST DATAFRAME TO BE MORE SIMILAR TO PRIVATE DATASET ...\")\n    test_df = train_df.sample(APPROX_PRIVATE_TEST_LEN).reset_index(drop=True).drop(columns=[\"label\"])\n    print(\"\\n... UPDATED TESTING DATAFRAME... \\n\\n\")\n    display(test_df)\n\n# Capture examples that span smallest, median, and large image size examples\nSMALLEST_IMAGE_ID = \"b43ebe_0\"\nSMALLEST_IMAGE_ROW = train_df[train_df.image_id==SMALLEST_IMAGE_ID]\nMEDIAN_IMAGE_ID = \"719165_0\"\nMEDIAN_IMAGE_ROW = train_df[train_df.image_id==MEDIAN_IMAGE_ID]\nLARGEST_IMAGE_ID = \"6baf51_0\"\nLARGEST_IMAGE_ROW = train_df[train_df.image_id==LARGEST_IMAGE_ID]\n\nprint(\"\\n... BASIC SETUP FINISHED ...\\n\\n\")","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-13T22:21:12.038969Z","iopub.execute_input":"2022-07-13T22:21:12.039328Z","iopub.status.idle":"2022-07-13T22:21:12.129187Z","shell.execute_reply.started":"2022-07-13T22:21:12.039288Z","shell.execute_reply":"2022-07-13T22:21:12.128331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n<a id=\"helper_functions\"></a>\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; background-color: #ffffff; color: #E55CA0;\" id=\"helper_functions\">2&nbsp;&nbsp;HELPER FUNCTIONS&nbsp;&nbsp;&nbsp;&nbsp;<a href=\"#toc\">&#10514;</a></h1>","metadata":{}},{"cell_type":"code","source":"# vips image to numpy array\ndef vips2numpy(vi):\n    # map vips formats to np dtypes\n    format_to_dtype = {\n        'uchar': np.uint8,\n        'char': np.int8,\n        'ushort': np.uint16,\n        'short': np.int16,\n        'uint': np.uint32,\n        'int': np.int32,\n        'float': np.float32,\n        'double': np.float64,\n        'complex': np.complex64,\n        'dpcomplex': np.complex128,\n    }\n    return np.ndarray(buffer=vi.write_to_memory(),\n                      dtype=format_to_dtype[vi.format],\n                      shape=[vi.height, vi.width, vi.bands])\n\ndef pyvips_open_downsampled_slide(img_path, downsample_by=8, as_numpy=True, resize_to=(512,512)):\n    \"\"\"\n    \n    Helper function to convert WSI into smaller downscaled version using pyvips.\n    \n    Timing details for MAYO CLINIC STRIP AI dataset:\n        SMALLEST IMAGE BY AREA (4417, 5314)\n            * Function takes ~1 seconds to run\n        MEDIAN IMAGE BY AREA   (17573, 38743)(~30X LARGER THAN SMALLEST IMAGE)\n            * Function takes ~30 seconds to run\n        LARGEST IMAGE BY AREA  (48282, 101406)(~208X LARGER THAN SMALLEST IMAGE)(~7X LARGER THAN MEDIAN IMAGE)\n            * Function takes ~405 seconds to run\n    \n    Args:\n        img_path (str): Path to .tif file to be downsampled\n        downsample_by (int): How many times smaller should resultant \n            image be. i.e. image_size*(1/downsample_by) = new_size\n        as_numpy (bool, optional): Whether to return image as numpy array (default)\n           or leave as PIL.Image object for further manipulation\n    \n    Returns:\n        Downsampled image as a numpy array of type uint8 with only 3 channels\n    \"\"\"\n    \n    # Open the image with PIL\n    tmp_img = pyvips.Image.new_from_file(img_path)    \n    \n    print(\"\\n... APPROXIMATE TIME TO LOAD IMAGE IS AT MOST APPROXIMATELY: \" \\\n          f\"{int((405/(48282*101406))*(tmp_img.width*tmp_img.height))} SECONDS ...\\n\")\n    \n    # if -1 than we downsample by whatever results in the image having dimensions as\n    # close to 512x512 as possible so the image can be resized after\n    \n    if downsample_by==-1:\n        _epsilon = 1e-3\n        downsample_by=min(tmp_img.width, tmp_img.height)/resize_to[0]-_epsilon\n    \n    # Resize the image\n    tmp_img = tmp_img.resize(1/downsample_by)\n    tmp_img = vips2numpy(tmp_img) if as_numpy else tmp_img\n    tmp_img = cv2.resize(tmp_img, resize_to) if resize_to is not None else tmp_img\n    \n    gc.collect(); gc.collect();\n    return tmp_img\n\n# Prove that we can infer on all image sizes available in training dataset \n#   --> (smallest, median and largest)\nif DEBUG:\n    plt.figure(figsize=(20,15))\n    \n    plt.subplot(1,3,1)\n    plt.imshow(pyvips_open_downsampled_slide(SMALLEST_IMAGE_ROW.image_path.values[0], downsample_by=-1, as_numpy=True, resize_to=(512,512)))\n    plt.title(\"SMALLEST IMAGE\", fontweight=\"bold\")\n    \n    plt.subplot(1,3,2)\n    plt.imshow(pyvips_open_downsampled_slide(MEDIAN_IMAGE_ROW.image_path.values[0], downsample_by=-1, as_numpy=True, resize_to=(512,512)))\n    plt.title(\"MEDIAN IMAGE\", fontweight=\"bold\")\n    \n    plt.subplot(1,3,3)\n    plt.imshow(pyvips_open_downsampled_slide(LARGEST_IMAGE_ROW.image_path.values[0], downsample_by=-1, as_numpy=True, resize_to=(512,512)))\n    plt.title(\"LARGEST IMAGE\", fontweight=\"bold\")\n    \n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T22:21:12.130625Z","iopub.execute_input":"2022-07-13T22:21:12.131003Z","iopub.status.idle":"2022-07-13T22:21:12.143614Z","shell.execute_reply.started":"2022-07-13T22:21:12.130967Z","shell.execute_reply":"2022-07-13T22:21:12.142638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n\n<a id=\"inference\"></a>\n\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; color: #E55CA0; background-color: #ffffff;\" id=\"inference\">\n    3&nbsp;&nbsp;INFER ON TEST IMAGES&nbsp;&nbsp;&nbsp;&nbsp;<a href=\"#toc\">&#10514;</a>\n</h1>\n\n---\n","metadata":{}},{"cell_type":"code","source":"# def tta_augment(img_batch):\n#     img_batch = tf.image.random_brightness(img_batch, 0.2)\n#     img_batch = tf.image.random_contrast(img_batch, 0.5, 2.0)\n#     img_batch = tf.image.random_saturation(img_batch, 0.75, 1.25)\n#     img_batch = tf.image.random_hue(img_batch, 0.1)\n#     return img_batch\n\nN_TEST = len(test_df)\ntest_image_paths = test_df.image_path.values\n\n# test time augmentation (not implemented yet)\nDO_TTA = False\nN_TTA = 4\n\n# Model loading\nprint(\"\\n... LOADING MODEL (~60-120 SECONDS)...\\n\")\nMODEL_PATH = \"/kaggle/input/mcsai-no-tiling-model-tpu-tf/efficientnetb6_512_notile\"\nmodel = tf.keras.models.load_model(MODEL_PATH, compile=False)\ndisplay(model.summary())\n\n# Get predictions\npreds = []\nfor img_path in tqdm(test_image_paths, total=N_TEST):\n    preds.append(model(tf.expand_dims(pyvips_open_downsampled_slide(img_path, downsample_by=-1, as_numpy=True, resize_to=(512,512)), axis=0)).numpy())\n    tf.keras.backend.clear_session(); gc.collect(); gc.collect();\npreds = np.concatenate(preds, axis=0)\nprint(preds.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T22:33:03.652719Z","iopub.execute_input":"2022-07-13T22:33:03.653084Z","iopub.status.idle":"2022-07-13T22:33:50.718634Z","shell.execute_reply.started":"2022-07-13T22:33:03.65305Z","shell.execute_reply":"2022-07-13T22:33:50.717734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n\n<a id=\"submission\"></a>\n\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; color: #E55CA0; background-color: #ffffff;\" id=\"submission\">\n    4&nbsp;&nbsp;CREATE SUBMISSION.CSV&nbsp;&nbsp;&nbsp;&nbsp;<a href=\"#toc\">&#10514;</a>\n</h1>\n\n---\n\nRemember we have to identify predictions for the same patient and take the average probability for each class before mapping those probabilities to the patient_id in the sample submission dataframe","metadata":{}},{"cell_type":"code","source":"for lbl, idx in S2I_LBL_MAP.items():\n    test_df[lbl]=preds[:, idx]\n\nCE_test_prob_map = test_df.groupby(\"patient_id\")[\"CE\"].mean().to_dict()\nLAA_test_prob_map = test_df.groupby(\"patient_id\")[\"LAA\"].mean().to_dict()\n\nss_df[\"CE\"] = ss_df[\"patient_id\"].map(lambda x: CE_test_prob_map.get(x, 0.5))\nss_df[\"LAA\"] = ss_df[\"patient_id\"].map(lambda x: LAA_test_prob_map.get(x, 0.5))\nss_df.to_csv(\"submission.csv\", index=False)\n\ndisplay(ss_df)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T22:38:10.116923Z","iopub.execute_input":"2022-07-13T22:38:10.117341Z","iopub.status.idle":"2022-07-13T22:38:10.423824Z","shell.execute_reply.started":"2022-07-13T22:38:10.117304Z","shell.execute_reply":"2022-07-13T22:38:10.422874Z"},"trusted":true},"execution_count":null,"outputs":[]}]}