{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# libvips\n\n[libvips](https://www.libvips.org/) is a demand-driven, horizontally threaded image processing library. Compared to similar libraries, libvips runs quickly and uses little memory. This notebook shows how you can install libvips and use it in an offline environment. [@analokamus](https://www.kaggle.com/analokamus) shared a similar [notebook](https://www.kaggle.com/code/analokamus/how-to-use-pyvips-offline) a year ago for [Mayo Clinic - STRIP AI](https://www.kaggle.com/competitions/mayo-clinic-strip-ai) competition but it is outdated and not working at the moment.","metadata":{}},{"cell_type":"markdown","source":"# Installation","metadata":{}},{"cell_type":"markdown","source":"Update package lists and download libvips and its dependencies.","metadata":{}},{"cell_type":"code","source":"!sudo apt-get update\n!sudo apt-get install libvips-dev -y --no-install-recommends --download-only -o dir::cache='./'","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-10-21T07:15:54.191511Z","iopub.execute_input":"2023-10-21T07:15:54.19196Z","iopub.status.idle":"2023-10-21T07:16:06.447159Z","shell.execute_reply.started":"2023-10-21T07:15:54.191925Z","shell.execute_reply":"2023-10-21T07:16:06.445516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir ./libvips\n!mv ./archives/* ./libvips\n!rm -rf ./archives\n!ls ./libvips","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-10-21T07:16:06.45034Z","iopub.execute_input":"2023-10-21T07:16:06.450797Z","iopub.status.idle":"2023-10-21T07:16:10.944466Z","shell.execute_reply.started":"2023-10-21T07:16:06.450759Z","shell.execute_reply":"2023-10-21T07:16:10.943398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Install libvips and its dependencies.","metadata":{}},{"cell_type":"code","source":"!yes | sudo dpkg -i ./libvips/*.deb","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-10-21T07:16:10.94589Z","iopub.execute_input":"2023-10-21T07:16:10.946281Z","iopub.status.idle":"2023-10-21T07:16:51.929479Z","shell.execute_reply.started":"2023-10-21T07:16:10.94624Z","shell.execute_reply":"2023-10-21T07:16:51.927993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Install python bindings of libvips (pyvips) and wheel archive it.","metadata":{}},{"cell_type":"code","source":"!pip install pyvips\n!pip wheel pyvips\n!mkdir pyvips\n!mv *.whl ./pyvips","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-10-21T07:16:51.935067Z","iopub.execute_input":"2023-10-21T07:16:51.935463Z","iopub.status.idle":"2023-10-21T07:17:19.17626Z","shell.execute_reply.started":"2023-10-21T07:16:51.935431Z","shell.execute_reply":"2023-10-21T07:17:19.174104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Importing Libraries","metadata":{}},{"cell_type":"code","source":"import os\nos.environ['OPENCV_IO_MAX_IMAGE_PIXELS'] = str(pow(2, 40)) # = 10,99,51,16,27,776\n\nimport numpy as np\nimport pandas as pd\nimport cv2\nimport pyvips\nimport matplotlib.pyplot as plt\nimport random\nimport seaborn as sns\n\nfrom glob import glob\nfrom pathlib import Path\nfrom tqdm.notebook import tqdm\n","metadata":{"execution":{"iopub.status.busy":"2023-10-21T07:17:19.178698Z","iopub.execute_input":"2023-10-21T07:17:19.179348Z","iopub.status.idle":"2023-10-21T07:17:21.438255Z","shell.execute_reply.started":"2023-10-21T07:17:19.179292Z","shell.execute_reply":"2023-10-21T07:17:21.436663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Defining functions","metadata":{}},{"cell_type":"code","source":"def visualize_categorical_column_distribution(df, column, title, path=None):\n\n    \"\"\"\n    Visualize distribution of the given categorical column on the given dataframe\n\n    Parameters\n    ----------\n    df: pandas.DataFrame\n        Dataframe with given categorical column\n\n    column: str\n        Name of the categorical column\n\n    title: str\n        Title of the plot\n\n    path: path-like str or None\n        Path of the output file or None (if path is None, plot is displayed with selected backend)\n    \"\"\"\n\n    value_counts = df[column].value_counts()\n    n = value_counts.sum()\n\n    fig, ax = plt.subplots(figsize=(24, df[column].value_counts().shape[0] + 4), dpi=100)\n    ax.bar(\n        x=np.arange(len(value_counts)),\n        height=value_counts.values,\n    )\n    ax.set_xlabel('')\n    ax.set_ylabel('')\n    ax.set_xticks(\n        np.arange(len(value_counts)),\n        [\n            f'{value}\\n{count:,} ({(count / n * 100):.2f}%)' for value, count in value_counts.to_dict().items()\n        ]\n    )\n    ax.tick_params(axis='x', labelsize=15, pad=10)\n    ax.tick_params(axis='y', labelsize=15, pad=10)\n    ax.set_title(title, size=20, pad=15)\n\n    if path is None:\n        plt.show()\n    else:\n        plt.savefig(path)\n        plt.close(fig)\n\n\ndef visualize_continuous_column_distribution(df, column, title, path=None):\n\n    \"\"\"\n    Visualize distribution of the given continuous column on the given dataframe\n\n    Parameters\n    ----------\n    df: pandas.DataFrame\n        Dataframe with given continuous column\n\n    column: str\n        Name of the continuous column,\n\n    title: str\n        Title of the plot\n\n    path: path-like str or None\n        Path of the output file or None (if path is None, plot is displayed with selected backend)\n    \"\"\"\n\n    fig, ax = plt.subplots(figsize=(24, 6), dpi=100)\n    ax.hist(df[column], bins=16)\n    ax.tick_params(axis='x', labelsize=15)\n    ax.tick_params(axis='y', labelsize=15)\n    ax.set_xlabel('')\n    ax.set_ylabel('')\n    ax.set_title(\n        title + f'''\n        {column}\n        Mean: {np.mean(df[column]):.2f} Std: {np.std(df[column]):.2f}\n        Min: {np.min(df[column]):.2f} Max: {np.max(df[column]):.2f}\n        ''',\n        size=20,\n        pad=12.5,\n        loc='center',\n        wrap=True\n    )\n\n    if path is None:\n        plt.show()\n    else:\n        plt.savefig(path, bbox_inches='tight')\n        plt.close(fig)\n        \n        \ndef visualize_continuous_columns_relationship(df, column1, column2, group, title, path=None):\n\n    \"\"\"\n    Visualize relationship of two given continuous columns on the given dataframe\n\n    Parameters\n    ----------\n    df: pandas.DataFrame\n        Dataframe with given continuous columns\n\n    column1: str\n        Name of the first continuous column\n        \n    column2: str\n        Name of the second continuous column\n        \n    group: str or None\n        Name of the group column\n\n    title: str\n        Title of the plot\n\n    path: path-like str or None\n        Path of the output file or None (if path is None, plot is displayed with selected backend)\n    \"\"\"\n\n    fig, ax = plt.subplots(figsize=(16, 8), dpi=100)\n    sns.scatterplot(x=column1, y=column2, hue=group, data=df, ax=ax)\n    ax.tick_params(axis='x', labelsize=15)\n    ax.tick_params(axis='y', labelsize=15)\n    ax.set_xlabel(column1, size=20, labelpad=15)\n    ax.set_ylabel(column2, size=20, labelpad=15)\n    if group is not None:\n        legend = ax.legend(title=group, loc='lower right', prop={'size': 14})\n        legend.get_title().set_fontsize(15)\n    ax.set_title(title, size=20, pad=15)\n\n    if path is None:\n        plt.show()\n    else:\n        plt.savefig(path, bbox_inches='tight')\n        plt.close(fig)","metadata":{"execution":{"iopub.status.busy":"2023-10-21T07:17:21.440422Z","iopub.execute_input":"2023-10-21T07:17:21.441083Z","iopub.status.idle":"2023-10-21T07:17:21.464121Z","shell.execute_reply.started":"2023-10-21T07:17:21.441038Z","shell.execute_reply":"2023-10-21T07:17:21.462658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def resize_with_aspect_ratio(image, longest_edge):\n\n    \"\"\"\n    Resize image while preserving its aspect ratio\n\n    Parameters\n    ----------\n    image: numpy.ndarray of shape (height, width, 3)\n        Image array\n\n    longest_edge: int\n        Desired number of pixels on the longest edge\n\n    Returns\n    -------\n    image: numpy.ndarray of shape (resized_height, resized_width, 3)\n        Resized image array\n    \"\"\"\n\n    height, width = image.shape[:2]\n    scale = longest_edge / max(height, width)\n    image = cv2.resize(image, dsize=(int(np.ceil(width * scale)), int(np.ceil(height * scale))), interpolation=cv2.INTER_LANCZOS4)\n\n    return image\n\n\ndef vips_read_image(image_path, longest_edge):\n    \n    \"\"\"\n    Read image using libvips\n\n    Parameters\n    ----------\n    image_path: str\n        Path of the image\n\n    Returns\n    -------\n    image: numpy.ndarray of shape (height, width, 3)\n        Image array\n    \"\"\"\n    \n    image_thumbnail = pyvips.Image.thumbnail(image_path, longest_edge)\n\n    return np.ndarray(\n        buffer=image_thumbnail.write_to_memory(),\n        dtype=np.uint8,\n        shape=[image_thumbnail.height, image_thumbnail.width, image_thumbnail.bands]\n    )\n\n\ndef visualize_image(image, title, path=None):\n\n    \"\"\"\n    Visualize the given image\n\n    Parameters\n    ----------\n    image: numpy.ndarray of shape (height, width, channel)\n        Image array\n        \n    title: str\n        Title of the plot\n\n    path: str or None\n        Path of the output file or None (if path is None, plot is displayed with selected backend)\n    \"\"\"\n\n    fig, ax = plt.subplots(figsize=(8, 8))\n    ax.imshow(image)\n    ax.set_xlabel('')\n    ax.set_ylabel('')\n    ax.tick_params(axis='x', labelsize=15, pad=10)\n    ax.tick_params(axis='y', labelsize=15, pad=10)\n    ax.set_title(title, size=15, pad=12.5, loc='center', wrap=True)\n\n    if path is None:\n        plt.show()\n    else:\n        plt.savefig(path)\n        plt.close(fig)\n","metadata":{"execution":{"iopub.status.busy":"2023-10-21T07:17:21.466348Z","iopub.execute_input":"2023-10-21T07:17:21.466824Z","iopub.status.idle":"2023-10-21T07:17:21.485006Z","shell.execute_reply.started":"2023-10-21T07:17:21.466765Z","shell.execute_reply":"2023-10-21T07:17:21.483663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualizing a Random image","metadata":{}},{"cell_type":"code","source":"competition_dataset_directory = Path('/kaggle/input/UBC-OCEAN')\ndf_train = pd.read_csv(competition_dataset_directory / 'train.csv')","metadata":{"execution":{"iopub.status.busy":"2023-10-21T07:17:21.486773Z","iopub.execute_input":"2023-10-21T07:17:21.487598Z","iopub.status.idle":"2023-10-21T07:17:21.52448Z","shell.execute_reply.started":"2023-10-21T07:17:21.487562Z","shell.execute_reply":"2023-10-21T07:17:21.523095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"random_image_id = df_train['image_id'].sample(n=1, random_state=random.seed()).values[0]\nrandom_image_id","metadata":{"execution":{"iopub.status.busy":"2023-10-21T07:17:21.526584Z","iopub.execute_input":"2023-10-21T07:17:21.527614Z","iopub.status.idle":"2023-10-21T07:17:21.545615Z","shell.execute_reply.started":"2023-10-21T07:17:21.527557Z","shell.execute_reply":"2023-10-21T07:17:21.544053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_row = df_train[df_train['image_id'] == random_image_id]\nimage_row","metadata":{"execution":{"iopub.status.busy":"2023-10-21T07:17:21.549988Z","iopub.execute_input":"2023-10-21T07:17:21.551469Z","iopub.status.idle":"2023-10-21T07:17:21.573911Z","shell.execute_reply.started":"2023-10-21T07:17:21.551419Z","shell.execute_reply":"2023-10-21T07:17:21.572649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_id = image_row['image_id'].values[0]\nlabel = image_row['label'].values[0]\nimage_width = image_row['image_width'].values[0]\nimage_height = image_row['image_height'].values[0]\nis_tma = image_row['is_tma'].values[0]","metadata":{"execution":{"iopub.status.busy":"2023-10-21T07:17:21.575607Z","iopub.execute_input":"2023-10-21T07:17:21.577075Z","iopub.status.idle":"2023-10-21T07:17:21.584834Z","shell.execute_reply.started":"2023-10-21T07:17:21.577023Z","shell.execute_reply":"2023-10-21T07:17:21.583221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_path = str(competition_dataset_directory / 'train_images' / f'{image_id}.png')\nimage_path","metadata":{"execution":{"iopub.status.busy":"2023-10-21T07:17:21.58801Z","iopub.execute_input":"2023-10-21T07:17:21.588452Z","iopub.status.idle":"2023-10-21T07:17:21.60319Z","shell.execute_reply.started":"2023-10-21T07:17:21.588413Z","shell.execute_reply":"2023-10-21T07:17:21.601501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Reading and resizing an image using libvips takes approximately **2.5** minutes and consumes significantly less RAM. libvips achieves this improvement by only keeping the pixels currently being processed in RAM and by having an efficient, threaded image IO system.","metadata":{}},{"cell_type":"code","source":"%%time\n\nimage = vips_read_image(image_path=image_path, longest_edge=5000)\n\nvisualize_image(\n    image=image,\n    title=f'Image: {image_id} Label: {label}\\n Resized Height: {image.shape[0]} Resized Width: {image.shape[1]}\\nOriginal Height: {image_height} Original Width: {image_width}\\nIs it a TMA: {is_tma} '\n)","metadata":{"execution":{"iopub.status.busy":"2023-10-21T07:17:21.604483Z","iopub.execute_input":"2023-10-21T07:17:21.604918Z","iopub.status.idle":"2023-10-21T07:18:11.213017Z","shell.execute_reply.started":"2023-10-21T07:17:21.604877Z","shell.execute_reply":"2023-10-21T07:18:11.211901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Steps on how to use libvips and pyvips in future training","metadata":{}},{"cell_type":"markdown","source":"For further details refer to the [libvips](https://www.libvips.org/API/current/) and [pyvips](https://libvips.github.io/pyvips/) documentations","metadata":{}},{"cell_type":"markdown","source":"Output of this notebook can be used to install libvips and pyvips for offline submission. This version works with GPU P100 notebooks and the previous works with CPU notebooks.\n\nRefer to this data card: [Libvips/Pyvips for offline use](https://www.kaggle.com/datasets/aditimutha10/libvips-pyvips/data) to know more. ","metadata":{}},{"cell_type":"markdown","source":"# train.csv analysis","metadata":{}},{"cell_type":"code","source":"df_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-10-21T07:18:11.214446Z","iopub.execute_input":"2023-10-21T07:18:11.215254Z","iopub.status.idle":"2023-10-21T07:18:11.228482Z","shell.execute_reply.started":"2023-10-21T07:18:11.215214Z","shell.execute_reply":"2023-10-21T07:18:11.227218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.info()","metadata":{"execution":{"iopub.status.busy":"2023-10-21T07:18:11.230198Z","iopub.execute_input":"2023-10-21T07:18:11.231428Z","iopub.status.idle":"2023-10-21T07:18:11.258644Z","shell.execute_reply.started":"2023-10-21T07:18:11.231367Z","shell.execute_reply":"2023-10-21T07:18:11.257637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df_train.image_id.is_unique)","metadata":{"execution":{"iopub.status.busy":"2023-10-21T07:18:11.260441Z","iopub.execute_input":"2023-10-21T07:18:11.260889Z","iopub.status.idle":"2023-10-21T07:18:11.272658Z","shell.execute_reply.started":"2023-10-21T07:18:11.260853Z","shell.execute_reply":"2023-10-21T07:18:11.271543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"visualize_categorical_column_distribution(\n    df=df_train,\n    column='label',\n    title='Training Set Category Frequencies'\n)","metadata":{"execution":{"iopub.status.busy":"2023-10-21T07:18:11.274532Z","iopub.execute_input":"2023-10-21T07:18:11.275278Z","iopub.status.idle":"2023-10-21T07:18:11.626651Z","shell.execute_reply.started":"2023-10-21T07:18:11.275235Z","shell.execute_reply":"2023-10-21T07:18:11.625463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"visualize_categorical_column_distribution(\n    df=df_train,\n    column='is_tma',\n    title='Training Set Image Type Frequencies (is_tma)'\n)","metadata":{"execution":{"iopub.status.busy":"2023-10-21T07:18:11.628233Z","iopub.execute_input":"2023-10-21T07:18:11.628611Z","iopub.status.idle":"2023-10-21T07:18:11.94076Z","shell.execute_reply.started":"2023-10-21T07:18:11.62858Z","shell.execute_reply":"2023-10-21T07:18:11.939555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"visualize_continuous_columns_relationship(\n    df=df_train,\n    column1='image_width',\n    column2='image_height',\n    group='is_tma',\n    title='Image Height vs Width (WSIs and TMAs)'\n)","metadata":{"execution":{"iopub.status.busy":"2023-10-21T07:18:11.942294Z","iopub.execute_input":"2023-10-21T07:18:11.943111Z","iopub.status.idle":"2023-10-21T07:18:12.409107Z","shell.execute_reply.started":"2023-10-21T07:18:11.943078Z","shell.execute_reply":"2023-10-21T07:18:12.407955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are 25 TMAs but there are only 2 unique height and width values in training set. All of the TMAs in training set are either 3388x3388 or 2964x2964 squares. Those width and height values may not be their raw sizes so they may be resized.","metadata":{}},{"cell_type":"markdown","source":"#### Moving on to data preparation.....","metadata":{}}]}