{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"}],"dockerImageVersionId":30558,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## UBC Ovarian Cancer Subtype Classification and Outlier Detection (UBC-OCEAN) - JPEG Dataset Pipeline","metadata":{}},{"cell_type":"markdown","source":"## 1. Setup","metadata":{}},{"cell_type":"code","source":"import os\nos.environ['OPENCV_IO_MAX_IMAGE_PIXELS'] = str(pow(2, 40))\n\nfrom pathlib import Path\nfrom tqdm.notebook import tqdm\nimport numpy as np\nimport pandas as pd\nimport cv2\nimport matplotlib.pyplot as plt","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-11-19T04:45:08.177515Z","iopub.execute_input":"2024-11-19T04:45:08.17873Z","iopub.status.idle":"2024-11-19T04:45:08.184792Z","shell.execute_reply.started":"2024-11-19T04:45:08.178656Z","shell.execute_reply":"2024-11-19T04:45:08.183491Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"competition_dataset_directory = Path('/kaggle/input/UBC-OCEAN')","metadata":{"execution":{"iopub.status.busy":"2024-11-19T04:45:18.601189Z","iopub.execute_input":"2024-11-19T04:45:18.601862Z","iopub.status.idle":"2024-11-19T04:45:18.606615Z","shell.execute_reply.started":"2024-11-19T04:45:18.601827Z","shell.execute_reply":"2024-11-19T04:45:18.60549Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2. Introduction\n\nThere are 538 images in training set. 513 of them are whole slide images (WSIs) and 25 of them are tissue microarrays (TMAs). All of them take 775 GBs of disk space which could be intimidating and not easily approachable for some audience. It's not even possible to download such large data from some countries due to slow internet connection. That problem is solved to some extend in this notebook.","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv(competition_dataset_directory / 'train.csv')\ndf_test = pd.read_csv(competition_dataset_directory / 'test.csv')\n\ndf_train","metadata":{"execution":{"iopub.status.busy":"2024-11-19T04:45:24.848601Z","iopub.execute_input":"2024-11-19T04:45:24.849386Z","iopub.status.idle":"2024-11-19T04:45:24.90039Z","shell.execute_reply.started":"2024-11-19T04:45:24.84935Z","shell.execute_reply":"2024-11-19T04:45:24.899206Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3. JPEG Compression\n\nJPEG compression is a process that aims to reduce file size by selectively eliminating less noticeable image details. The algorithm is created to preserve the most important details, particularly when using higher quality settings.\n\nIf you choose higher quality settings, you'll only see a drop in quality when you look at the image really closely. But if you use lower quality settings, you'll notice problems with the image more easily. That's the important trade-off between memory and image quality. For large images like the ones here, you can opt towards memory a bit more safely.","metadata":{}},{"cell_type":"code","source":"def visualize_image(image, title, path=None):\n\n    \"\"\"\n    Visualize the given image\n\n    Parameters\n    ----------\n    image: numpy.ndarray of shape (height, width, channel)\n        Image array\n        \n    title: str\n        Title of the plot\n\n    path: str or None\n        Path of the output file or None (if path is None, plot is displayed with selected backend)\n    \"\"\"\n\n    fig, ax = plt.subplots(figsize=(8, 8))\n    ax.imshow(image)\n    ax.set_xlabel('')\n    ax.set_ylabel('')\n    ax.tick_params(axis='x', labelsize=15, pad=10)\n    ax.tick_params(axis='y', labelsize=15, pad=10)\n    ax.set_title(title, size=15, pad=12.5, loc='center', wrap=True)\n\n    if path is None:\n        plt.show()\n    else:\n        plt.savefig(path)\n        plt.close(fig)\n","metadata":{"execution":{"iopub.status.busy":"2024-11-19T04:45:42.417636Z","iopub.execute_input":"2024-11-19T04:45:42.41809Z","iopub.status.idle":"2024-11-19T04:45:42.425402Z","shell.execute_reply.started":"2024-11-19T04:45:42.418055Z","shell.execute_reply":"2024-11-19T04:45:42.423865Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"First row of the training set is selected to show effects of JPEG compression. ","metadata":{}},{"cell_type":"code","source":"first_row = df_train.loc[0]\nraw_image_path = str(competition_dataset_directory / 'train_images' / f'{first_row[\"image_id\"]}.png')\nimage = cv2.imread(raw_image_path)\nraw_image_size = os.path.getsize(raw_image_path) / (1 << 20)","metadata":{"execution":{"iopub.status.busy":"2024-11-19T04:45:49.547675Z","iopub.execute_input":"2024-11-19T04:45:49.54809Z","iopub.status.idle":"2024-11-19T04:46:10.178463Z","shell.execute_reply.started":"2024-11-19T04:45:49.548057Z","shell.execute_reply":"2024-11-19T04:46:10.177566Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Raw image size is **576.34** MBs in lossless PNG format.","metadata":{}},{"cell_type":"code","source":"visualize_image(\n    image=image,\n    title=f'Raw Image Size: {raw_image_size:.2f} MBs\\n Image ID: {first_row[\"image_id\"]} Label: {first_row[\"label\"]}\\nHeight: {image.shape[0]} Width: {image.shape[1]}\\nMean: {np.mean(image):.2f} Std: {np.std(image):.2f}\\nMin: {np.min(image):.2f} Max: {np.max(image):.2f}'\n)","metadata":{"execution":{"iopub.status.busy":"2024-11-19T04:46:10.18012Z","iopub.execute_input":"2024-11-19T04:46:10.180437Z","iopub.status.idle":"2024-11-19T04:46:53.645162Z","shell.execute_reply.started":"2024-11-19T04:46:10.180408Z","shell.execute_reply":"2024-11-19T04:46:53.644068Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"100% JPEG compression reduces the image size from **576.34** to **280.37** MBs. Even though the compression quality is 100%, it is still lossy.","metadata":{}},{"cell_type":"code","source":"jpeg_quality = 100\ncv2.imwrite('./4.jpg', image, [int(cv2.IMWRITE_JPEG_QUALITY), jpeg_quality])\ncompressed_image = cv2.imread('./4.jpg')\ncompressed_image_size = os.path.getsize('./4.jpg') / (1 << 20)\n\nvisualize_image(\n    image=compressed_image,\n    title=f'{jpeg_quality}% JPEG Compression Image Size: {compressed_image_size:.2f} MBs\\n Image ID: {first_row[\"image_id\"]} Label: {first_row[\"label\"]}\\nHeight: {compressed_image.shape[0]} Width: {compressed_image.shape[1]}\\nMean: {np.mean(compressed_image):.2f} Std: {np.std(compressed_image):.2f}\\nMin: {np.min(compressed_image):.2f} Max: {np.max(compressed_image):.2f}'\n)","metadata":{"execution":{"iopub.status.busy":"2024-11-19T04:47:30.124675Z","iopub.execute_input":"2024-11-19T04:47:30.125044Z","iopub.status.idle":"2024-11-19T04:48:43.662163Z","shell.execute_reply.started":"2024-11-19T04:47:30.125014Z","shell.execute_reply":"2024-11-19T04:48:43.661131Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"90% JPEG compression reduces the image size from **576.34** to **114.36** MBs. Decreasing the compression quality by 10% reduces the image size a lot.","metadata":{}},{"cell_type":"code","source":"jpeg_quality = 90\ncv2.imwrite('./4.jpg', image, [int(cv2.IMWRITE_JPEG_QUALITY), jpeg_quality])\ncompressed_image = cv2.imread('./4.jpg')\ncompressed_image_size = os.path.getsize('./4.jpg') / (1 << 20)\n\nvisualize_image(\n    image=compressed_image,\n    title=f'{jpeg_quality}% JPEG Compression Image Size: {compressed_image_size:.2f} MBs\\n Image ID: {first_row[\"image_id\"]} Label: {first_row[\"label\"]}\\nHeight: {compressed_image.shape[0]} Width: {compressed_image.shape[1]}\\nMean: {np.mean(compressed_image):.2f} Std: {np.std(compressed_image):.2f}\\nMin: {np.min(compressed_image):.2f} Max: {np.max(compressed_image):.2f}'\n)","metadata":{"execution":{"iopub.status.busy":"2024-11-19T04:49:47.818924Z","iopub.execute_input":"2024-11-19T04:49:47.819299Z","iopub.status.idle":"2024-11-19T04:50:56.049119Z","shell.execute_reply.started":"2024-11-19T04:49:47.819271Z","shell.execute_reply":"2024-11-19T04:50:56.047955Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"80% JPEG compression reduces the image size from **576.34** to **87.35** MBs. Benefit gain from decreasing the compression quality is starting to diminish at this point. 80% compression quality could be optimal.","metadata":{}},{"cell_type":"code","source":"jpeg_quality = 80\ncv2.imwrite('./4.jpg', image, [int(cv2.IMWRITE_JPEG_QUALITY), jpeg_quality])\ncompressed_image = cv2.imread('./4.jpg')\ncompressed_image_size = os.path.getsize('./4.jpg') / (1 << 20)\n\nvisualize_image(\n    image=compressed_image,\n    title=f'{jpeg_quality}% JPEG Compression Image Size: {compressed_image_size:.2f} MBs\\n Image ID: {first_row[\"image_id\"]} Label: {first_row[\"label\"]}\\nHeight: {compressed_image.shape[0]} Width: {compressed_image.shape[1]}\\nMean: {np.mean(compressed_image):.2f} Std: {np.std(compressed_image):.2f}\\nMin: {np.min(compressed_image):.2f} Max: {np.max(compressed_image):.2f}'\n)","metadata":{"execution":{"iopub.status.busy":"2023-10-18T09:25:18.974389Z","iopub.execute_input":"2023-10-18T09:25:18.975176Z","iopub.status.idle":"2023-10-18T09:26:21.028945Z","shell.execute_reply.started":"2023-10-18T09:25:18.975132Z","shell.execute_reply":"2023-10-18T09:26:21.027646Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"del image, compressed_image","metadata":{"execution":{"iopub.status.busy":"2024-11-19T04:53:48.687718Z","iopub.execute_input":"2024-11-19T04:53:48.688115Z","iopub.status.idle":"2024-11-19T04:53:49.006543Z","shell.execute_reply.started":"2024-11-19T04:53:48.688082Z","shell.execute_reply":"2024-11-19T04:53:49.005083Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 4. Dataset Pipeline\n\nThis is the dataset pipeline where all of the images are compressed and saved to disk. Image sizes can be reduced further by resizing them. Images are not perfect squares so they have to be resized while preserving their aspect ratios. That operation prevents any distortion. ","metadata":{}},{"cell_type":"code","source":"def resize_with_aspect_ratio(image, longest_edge):\n\n    \"\"\"\n    Resize image while preserving its aspect ratio\n\n    Parameters\n    ----------\n    image: numpy.ndarray of shape (height, width, 3)\n        Image array\n\n    longest_edge: int\n        Desired number of pixels on the longest edge\n\n    Returns\n    -------\n    image: numpy.ndarray of shape (resized_height, resized_width, 3)\n        Resized image array\n    \"\"\"\n\n    height, width = image.shape[:2]\n    scale = longest_edge / max(height, width)\n    image = cv2.resize(image, dsize=(int(np.ceil(width * scale)), int(np.ceil(height * scale))), interpolation=cv2.INTER_AREA)\n\n    return image\n","metadata":{"execution":{"iopub.status.busy":"2024-11-19T04:55:02.371449Z","iopub.execute_input":"2024-11-19T04:55:02.371951Z","iopub.status.idle":"2024-11-19T04:55:02.37785Z","shell.execute_reply.started":"2024-11-19T04:55:02.37191Z","shell.execute_reply":"2024-11-19T04:55:02.376757Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"`MAX_SIZE` is set to `20000` pixels which is the highest value that an image can have as height and/or width. `JPEG_QUALITY` is set to `80` which is a reasonable rate of compression quality. Those two parameters can be tuned based on the need but they might be enough for modelling purposes at this point.\n\n**This notebook uses the updated data with fixed masks.** ","metadata":{}},{"cell_type":"code","source":"import os\nfrom pathlib import Path\nimport cv2\nfrom tqdm import tqdm\n\nMAX_SIZE = 20000\nJPEG_QUALITY = 80\n\n# Directories for compressed images\ntrain_compressed_image_directory = Path('./train_compressed_images')\ntrain_compressed_image_directory.mkdir(exist_ok=True, parents=True)\n\ntest_compressed_image_directory = Path('./test_compressed_images')\ntest_compressed_image_directory.mkdir(exist_ok=True, parents=True)\n\n# Compress and save train images\nfor idx, row in tqdm(df_train.iterrows(), total=df_train.shape[0]):\n    \n    raw_image_path = str(competition_dataset_directory / 'train_images' / f'{row[\"image_id\"]}.png')\n    compressed_image_path = str(train_compressed_image_directory / f'{row[\"image_id\"]}.jpg')\n    image_type = 'TMA' if row['is_tma'] else 'WSI' \n    \n    image = cv2.imread(raw_image_path)\n    \n    raw_image_shape = image.shape[:2]\n    longest_edge = max(raw_image_shape)\n    if longest_edge > MAX_SIZE:\n        image = resize_with_aspect_ratio(image=image, longest_edge=MAX_SIZE)\n    resized_image_shape = image.shape[:2]\n\n    # Save the compressed image with the same name in the train_compressed_image_directory\n    cv2.imwrite(compressed_image_path, image, [int(cv2.IMWRITE_JPEG_QUALITY), JPEG_QUALITY])\n\n    raw_image_size = os.path.getsize(raw_image_path) / (1 << 20)\n    compressed_image_size = os.path.getsize(compressed_image_path) / (1 << 20)\n    print(f'Image ID: {row[\"image_id\"]} Type: {image_type} Shape: {raw_image_shape[0]}x{raw_image_shape[1]} -> {resized_image_shape[0]}x{resized_image_shape[1]} Size: {raw_image_size:.2f} -> {compressed_image_size:.2f} MBs')\n\n# Compress and save test images\nfor idx, row in tqdm(df_test.iterrows(), total=df_test.shape[0]):\n    \n    raw_image_path = str(competition_dataset_directory / 'test_images' / f'{row[\"image_id\"]}.png')\n    compressed_image_path = str(test_compressed_image_directory / f'{row[\"image_id\"]}.jpg')\n    \n    image = cv2.imread(raw_image_path)\n    \n    raw_image_shape = image.shape[:2]\n    longest_edge = max(raw_image_shape)\n    if longest_edge > MAX_SIZE:\n        image = resize_with_aspect_ratio(image=image, longest_edge=MAX_SIZE)\n    resized_image_shape = image.shape[:2]\n\n    # Save the compressed image with the same name in the test_compressed_image_directory\n    cv2.imwrite(compressed_image_path, image, [int(cv2.IMWRITE_JPEG_QUALITY), JPEG_QUALITY])\n\n    raw_image_size = os.path.getsize(raw_image_path) / (1 << 20)\n    compressed_image_size = os.path.getsize(compressed_image_path) / (1 << 20)\n    print(f'Image ID: {row[\"image_id\"]} Shape: {raw_image_shape[0]}x{raw_image_shape[1]} -> {resized_image_shape[0]}x{resized_image_shape[1]} Size: {raw_image_size:.2f} -> {compressed_image_size:.2f} MBs')\n","metadata":{"execution":{"iopub.status.busy":"2024-11-19T04:55:18.340473Z","iopub.execute_input":"2024-11-19T04:55:18.340885Z","iopub.status.idle":"2024-11-19T05:16:38.315874Z","shell.execute_reply.started":"2024-11-19T04:55:18.340843Z","shell.execute_reply":"2024-11-19T05:16:38.314353Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import shutil\n\n# Specify the paths to your folders\ntrain_folder = '/kaggle/working/train_compressed_images'\ntest_folder = '/kaggle/working/test_compressed_images'\n\n# Create zip files for the folders\nshutil.make_archive('/kaggle/working/train_compressed_images', 'zip', train_folder)\nshutil.make_archive('/kaggle/working/test_compressed_images', 'zip', test_folder)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}