{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"},{"sourceId":6984590,"sourceType":"datasetVersion","datasetId":4014175}],"dockerImageVersionId":30626,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 🟣 **Install Libs**","metadata":{}},{"cell_type":"code","source":"!sudo apt install libvips-dev --no-install-recommends -y -qq\n!pip -q install pyvips","metadata":{"execution":{"iopub.status.busy":"2024-01-01T12:48:15.237027Z","iopub.execute_input":"2024-01-01T12:48:15.237541Z","iopub.status.idle":"2024-01-01T12:48:33.87284Z","shell.execute_reply.started":"2024-01-01T12:48:15.237505Z","shell.execute_reply":"2024-01-01T12:48:33.871014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip -q install patchify","metadata":{"execution":{"iopub.status.busy":"2024-01-01T12:48:39.64493Z","iopub.execute_input":"2024-01-01T12:48:39.645458Z","iopub.status.idle":"2024-01-01T12:48:54.88293Z","shell.execute_reply.started":"2024-01-01T12:48:39.645416Z","shell.execute_reply":"2024-01-01T12:48:54.880908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 🟣 **Import Libs**","metadata":{}},{"cell_type":"code","source":"import os, glob\nfrom pathlib import Path\nimport pandas as pd, gc\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom patchify import patchify\nimport pyvips\nfrom PIL import Image\nimport cv2\nfrom sklearn.model_selection import train_test_split\n\nfrom tqdm import tqdm\nfrom joblib import Parallel, delayed\n\nfrom IPython.display import display\n\nfrom sklearn.preprocessing import LabelEncoder","metadata":{"execution":{"iopub.status.busy":"2024-01-01T12:49:09.170602Z","iopub.execute_input":"2024-01-01T12:49:09.171139Z","iopub.status.idle":"2024-01-01T12:49:11.427913Z","shell.execute_reply.started":"2024-01-01T12:49:09.1711Z","shell.execute_reply":"2024-01-01T12:49:11.426555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 🟣 **Arguments**","metadata":{}},{"cell_type":"code","source":"root = '/kaggle/input/UBC-OCEAN/'\nroot = Path(root)\nmask_root = '/kaggle/input/ubc-ovarian-cancer-competition-supplemental-masks'\nmask_root = Path(mask_root)\ntrain_csv_path = root / 'train.csv'\ntest_csv_path = root / 'test.csv'\ntrain_images_path = root / 'train_images'\ntest_images_path = root / 'test_images'\ntrain_thumbnails_path = root / 'train_thumbnails'\ntest_thumbnails_path = root / 'test_thumbnails'\noutput_train_directory = 'output_train'\noutput_valid_directory = 'output_valid'","metadata":{"execution":{"iopub.status.busy":"2024-01-01T12:49:11.429914Z","iopub.execute_input":"2024-01-01T12:49:11.430434Z","iopub.status.idle":"2024-01-01T12:49:11.438362Z","shell.execute_reply.started":"2024-01-01T12:49:11.430394Z","shell.execute_reply":"2024-01-01T12:49:11.436829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 🟣 **Dataset**","metadata":{}},{"cell_type":"markdown","source":"### 🟡 Load CSV file","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv(train_csv_path)\ndisplay(df_train.head())\ndf_train.shape","metadata":{"execution":{"iopub.status.busy":"2024-01-01T12:49:18.515474Z","iopub.execute_input":"2024-01-01T12:49:18.516038Z","iopub.status.idle":"2024-01-01T12:49:18.569647Z","shell.execute_reply.started":"2024-01-01T12:49:18.515982Z","shell.execute_reply":"2024-01-01T12:49:18.568472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.info()","metadata":{"execution":{"iopub.status.busy":"2024-01-01T12:49:23.597009Z","iopub.execute_input":"2024-01-01T12:49:23.597527Z","iopub.status.idle":"2024-01-01T12:49:23.633247Z","shell.execute_reply.started":"2024-01-01T12:49:23.597472Z","shell.execute_reply":"2024-01-01T12:49:23.632312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_thumbnails_list = glob.glob(str(train_thumbnails_path / '*.png'))\ntest_thumbnails_list = glob.glob(str(test_thumbnails_path / '*.png'))\ntrain_images_list = glob.glob(str(train_images_path / '*.png'))\ntest_images_list = glob.glob(str(test_images_path / '*.png'))\nmask_list = glob.glob(str(mask_root / '*.png'))\nlen(train_images_list), len(train_thumbnails_list), len(mask_list), len(test_images_list), len(test_thumbnails_list)","metadata":{"execution":{"iopub.status.busy":"2024-01-01T12:49:31.754327Z","iopub.execute_input":"2024-01-01T12:49:31.755211Z","iopub.status.idle":"2024-01-01T12:49:31.974757Z","shell.execute_reply.started":"2024-01-01T12:49:31.755172Z","shell.execute_reply":"2024-01-01T12:49:31.973245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 🟡 Get images path","metadata":{}},{"cell_type":"markdown","source":"#### 🔵 Add thumbnails and images path columns to the dataframe ","metadata":{}},{"cell_type":"code","source":"def get_train_thumbnails_path(image_id):\n    thumbnails_path  = train_thumbnails_path / f'{image_id}_thumbnail.png'\n    return thumbnails_path if thumbnails_path.exists() else None\n\ndef get_train_images_path(image_id):\n    images_path  = train_images_path / f'{image_id}.png'\n    return images_path if images_path.exists() else None\n\n# add thumbnails and images path columns to the dataframe \ndf_train['image_path'] = df_train['image_id'].apply(get_train_images_path)\ndf_train['thumbnail_path'] = df_train['image_id'].apply(get_train_thumbnails_path)\n\ndf_train.head(10)","metadata":{"execution":{"iopub.status.busy":"2024-01-01T12:49:37.996635Z","iopub.execute_input":"2024-01-01T12:49:37.998152Z","iopub.status.idle":"2024-01-01T12:49:38.60165Z","shell.execute_reply.started":"2024-01-01T12:49:37.998089Z","shell.execute_reply":"2024-01-01T12:49:38.600584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 🔵 Show an image","metadata":{}},{"cell_type":"code","source":"image = Image.open(df_train['thumbnail_path'].iloc[5])\nwidth, height = image.size\nprint(f'width: {width}, height: {height}')\ndisplay(image)","metadata":{"execution":{"iopub.status.busy":"2024-01-01T12:49:47.573367Z","iopub.execute_input":"2024-01-01T12:49:47.573832Z","iopub.status.idle":"2024-01-01T12:49:50.174015Z","shell.execute_reply.started":"2024-01-01T12:49:47.573796Z","shell.execute_reply":"2024-01-01T12:49:50.172288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 🔵 Add thumbnails width and height columns to the dataframe","metadata":{}},{"cell_type":"code","source":"def get_thumbnail_dimensions(path):\n    if path is not None:\n        image = cv2.imread(str(path))\n        if image is not None:\n            height, width, _ = image.shape\n            return int(width), int(height)\n    return None\n\ndf_train[['thumbnail_width',\n          'thumbnail_height']] = df_train['thumbnail_path'].apply(get_thumbnail_dimensions).apply(pd.Series)\n\ncolumns = ['image_id', 'label', 'image_width', 'thumbnail_width',\n           'image_height', 'thumbnail_height', 'is_tma',\n           'image_path', 'thumbnail_path']\n\ndf_train = df_train[columns]\ndf_train.head(10)","metadata":{"execution":{"iopub.status.busy":"2024-01-01T12:54:01.125179Z","iopub.execute_input":"2024-01-01T12:54:01.125683Z","iopub.status.idle":"2024-01-01T12:55:56.060463Z","shell.execute_reply.started":"2024-01-01T12:54:01.125644Z","shell.execute_reply":"2024-01-01T12:55:56.059582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.to_csv('df_train.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-12-25T08:27:25.397073Z","iopub.execute_input":"2023-12-25T08:27:25.397601Z","iopub.status.idle":"2023-12-25T08:27:25.411798Z","shell.execute_reply.started":"2023-12-25T08:27:25.397559Z","shell.execute_reply":"2023-12-25T08:27:25.410448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 🟡 Class Distribution","metadata":{}},{"cell_type":"markdown","source":"#### 🔵 Counts the occurances of each class, and then creates a bar plot to visualize the distribution.","metadata":{}},{"cell_type":"code","source":"class_counts = df_train['label'].value_counts()\n\n# define color for each class\ncolors = ['blue', 'orange', 'green', 'red', 'purple']\n\n# plot the class distribution\nax = class_counts.plot(kind='bar', rot=0, color=colors)\nplt.xlabel('Class')\nplt.ylabel('Count')\nplt.title('Class Distribution');\n\n# annotate each bar with its count\nfor i, count in enumerate(class_counts):\n    ax.text(i, count + 1, str(count), ha='center', va='bottom', color='black')","metadata":{"execution":{"iopub.status.busy":"2024-01-01T12:56:42.384011Z","iopub.execute_input":"2024-01-01T12:56:42.384453Z","iopub.status.idle":"2024-01-01T12:56:42.762959Z","shell.execute_reply.started":"2024-01-01T12:56:42.384421Z","shell.execute_reply":"2024-01-01T12:56:42.761927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 🔵 Pie chart to visualize the class distribution.","metadata":{}},{"cell_type":"code","source":"# plot the class distribution as a pie chart\nclass_counts.plot(kind='pie', autopct='%1.1f%%', # autopct: to display the percentage labels on each slice\n                 startangle=90)\nplt.axis('equal')\nplt.title('Class Distribution');","metadata":{"execution":{"iopub.status.busy":"2024-01-01T12:56:48.039845Z","iopub.execute_input":"2024-01-01T12:56:48.040353Z","iopub.status.idle":"2024-01-01T12:56:48.28202Z","shell.execute_reply.started":"2024-01-01T12:56:48.040311Z","shell.execute_reply":"2024-01-01T12:56:48.280208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Train dataset is **imblanced**:\n* **HGSC (41.3%):** This class contains the highest percentage of instances, indicating that it is the most prevalent class.\n* **EC (23%):** This class has the second-highest percentage, indicating a significant prresence in dataset, but not as dominant as HGSC.\n* **CC (18.4%):** This class falls in the middle in terms of prevalence.\n* **LGSC (8.7%):** This class has a relatively low percentage (similar to MC), indicating a lower prevalence.\n* **MC (8.6%):** This class has a lower percentage, suggesting that is less commen in the dataset.","metadata":{}},{"cell_type":"markdown","source":"#### 🔵 Create a bar plot for each label based on the counts of the 'is_tma' column.","metadata":{}},{"cell_type":"code","source":"# group by 'label' and count the occurrences of True and False in 'is_tma'\ngrouped_data = df_train.groupby(['label', 'is_tma']).size().unstack()\n\n# plotting the bar chart with labels\nax = grouped_data.plot(kind='bar')\n\n# add labels on top of each bar\nfor p in ax.patches:\n    height = p.get_height()\n    width = p.get_x() + p.get_width() / 2\n    label = f'{height}'\n    ax.text(width, height, label, ha='center', va='bottom')\n\nplt.title('Bar Plot of is_tma Counts for Each Label')\nplt.xlabel('Label')\nplt.ylabel('Count')\nplt.legend(title='is_tma', bbox_to_anchor=(1.05, 1), loc='upper left')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-01T12:56:53.227348Z","iopub.execute_input":"2024-01-01T12:56:53.228681Z","iopub.status.idle":"2024-01-01T12:56:53.598829Z","shell.execute_reply.started":"2024-01-01T12:56:53.228623Z","shell.execute_reply":"2024-01-01T12:56:53.597577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 🟡 Images Info","metadata":{}},{"cell_type":"markdown","source":"#### 🔵 Distribution of Image Width/Height & Thumbnail Width/Height","metadata":{}},{"cell_type":"code","source":"def plot_distribution(ax, data, title, xlabel, color):\n    sns.histplot(data, bins=30, kde=True, color=color, ax=ax)\n    ax.set_title(title)\n    ax.set_xlabel(xlabel)\n    ax.set_ylabel('Frequency')\n\nfig, axs = plt.subplots(2, 2, figsize=(12, 12))\n\n# Distribution of Image Width\nplot_distribution(axs[0, 0], df_train['image_width'],\n                  'Distribution of Image Width', 'Image Width', 'skyblue')\n\n# Distribution of Image Height\nplot_distribution(axs[0, 1], df_train['image_height'],\n                  'Distribution of Image Height', 'Image Height', 'salmon')\n\n# Distribution of Thumbnail Width\nplot_distribution(axs[1, 0], df_train['thumbnail_width'],\n                  'Distribution of Thumbnail Width', 'Thumbnail Width', 'lightseagreen')\n\n# Distribution of Thumbnail Height\nplot_distribution(axs[1, 1], df_train['thumbnail_height'],\n                  'Distribution of Thumbnail Height', 'Thumbnail Height', 'orchid')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-01T12:56:58.482731Z","iopub.execute_input":"2024-01-01T12:56:58.483235Z","iopub.status.idle":"2024-01-01T12:57:00.331381Z","shell.execute_reply.started":"2024-01-01T12:56:58.483199Z","shell.execute_reply":"2024-01-01T12:57:00.329687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 🔵 BoxPlot of image_width/image_height and thumbnail_width/thumbnail_height for each category in the label column","metadata":{}},{"cell_type":"code","source":"def plot_boxplot(ax, data, x_col, y_col, title, xlabel, ylabel):\n    sns.boxplot(x=x_col, y=y_col, data=data, ax=ax)\n    ax.set_title(title)\n    ax.set_xlabel(xlabel)\n    ax.set_ylabel(ylabel)\n\nfig, axs = plt.subplots(2, 2, figsize=(12, 12))\n\n# Boxplot of image_width\nplot_boxplot(axs[0, 0], df_train, 'label', 'image_width', \n             'Boxplot of Image Width for Each Category',\n             'Label', 'Image Width')\n\n# Boxplot of image_height\nplot_boxplot(axs[0, 1], df_train, 'label', 'image_height',\n             'Boxplot of Image Height for Each Category',\n             'Label', 'Image Height')\n\n# Boxplot of thumbnail_width\nplot_boxplot(axs[1, 0], df_train, 'label', 'thumbnail_width',\n             'Boxplot of Thumbnail Width for Each Category', \n             'Label', 'Thumbnail Width')\n\n# Boxplot of thumbnail_height\nplot_boxplot(axs[1, 1], df_train, 'label', 'thumbnail_height',\n             'Boxplot of Thumbnail Height for Each Category',\n             'Label', 'Thumbnail Height')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-01T12:57:05.117949Z","iopub.execute_input":"2024-01-01T12:57:05.118434Z","iopub.status.idle":"2024-01-01T12:57:06.455639Z","shell.execute_reply.started":"2024-01-01T12:57:05.1184Z","shell.execute_reply":"2024-01-01T12:57:06.454293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 🔵 Show random image of thumbnails","metadata":{}},{"cell_type":"code","source":"def show_random_image_grid(dataframe, image_path_column, id_column,\n                           label_column, width_column, height_column, grid_size=(3, 3)):\n\n    # select a random sample of rows from the DataFrame\n    random_rows = dataframe.sample(n=grid_size[0] * grid_size[1])\n\n    # set up the grid for subplots\n    fig, axes = plt.subplots(*grid_size, figsize=(10, 10))\n    fig.suptitle('Randomly Selected Images')\n\n    # iterate through the random rows and plot the images\n    for i, (_, row) in enumerate(random_rows.iterrows()):\n        img_path = row[image_path_column]\n        img = Image.open(img_path)\n        axes[i // grid_size[1], i % grid_size[1]].imshow(img)\n        axes[i // grid_size[1], i % grid_size[1]].set_title(\n            f\"id: {row[id_column]}\\nLabel: {row[label_column]}\\nWidth: {row[width_column]}\\nHeight: {row[height_column]}\",\n        )\n        axes[i // grid_size[1], i % grid_size[1]].axis('off')\n\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-01T12:57:13.671058Z","iopub.execute_input":"2024-01-01T12:57:13.67161Z","iopub.status.idle":"2024-01-01T12:57:13.683984Z","shell.execute_reply.started":"2024-01-01T12:57:13.671569Z","shell.execute_reply":"2024-01-01T12:57:13.681679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_random_image_grid(df_train.dropna(), 'thumbnail_path', 'image_id', 'label', \n                       'thumbnail_width', 'thumbnail_height', grid_size=(3, 3))","metadata":{"execution":{"iopub.status.busy":"2024-01-01T11:30:33.146978Z","iopub.execute_input":"2024-01-01T11:30:33.147534Z","iopub.status.idle":"2024-01-01T11:30:45.397746Z","shell.execute_reply.started":"2024-01-01T11:30:33.14749Z","shell.execute_reply":"2024-01-01T11:30:45.39653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 🟡 Supplemental Masks","metadata":{}},{"cell_type":"markdown","source":"#### 🔵 Create dataframe of supplemental Masks","metadata":{}},{"cell_type":"code","source":"image_ids = []\nfor path in mask_list:\n    img_id = os.path.splitext(os.path.basename(path))[0]\n    image_ids.append(int(img_id))\n    \ndf_mask_train = pd.DataFrame({'image_id': image_ids,\n                              'mask_path': mask_list})\n\ndf_mask_train.to_csv('df_mask_train.csv', index=False)\ndisplay(df_mask_train.head())\ndf_mask_train.shape","metadata":{"execution":{"iopub.status.busy":"2024-01-01T12:57:21.397333Z","iopub.execute_input":"2024-01-01T12:57:21.397833Z","iopub.status.idle":"2024-01-01T12:57:21.421412Z","shell.execute_reply.started":"2024-01-01T12:57:21.397794Z","shell.execute_reply":"2024-01-01T12:57:21.419991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 🔵 Merge `df_train` and `df_mask_train`","metadata":{}},{"cell_type":"code","source":"df_train_merge = df_train.merge(df_mask_train, on='image_id')\ndisplay(df_train_merge.head())\ndf_train_merge.shape","metadata":{"execution":{"iopub.status.busy":"2024-01-01T12:57:30.854367Z","iopub.execute_input":"2024-01-01T12:57:30.85487Z","iopub.status.idle":"2024-01-01T12:57:30.891276Z","shell.execute_reply.started":"2024-01-01T12:57:30.854832Z","shell.execute_reply":"2024-01-01T12:57:30.889782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 🟡 LabelEncoder: Encode target labels with value between 0 and n_classes-1","metadata":{}},{"cell_type":"code","source":"le = LabelEncoder()\nle.fit(df_train_merge['label'])\ndf_train_merge['label'] = le.transform(df_train_merge['label'])\ndf_train_merge.head()","metadata":{"execution":{"iopub.status.busy":"2024-01-01T12:57:38.141065Z","iopub.execute_input":"2024-01-01T12:57:38.141611Z","iopub.status.idle":"2024-01-01T12:57:38.164254Z","shell.execute_reply.started":"2024-01-01T12:57:38.14157Z","shell.execute_reply":"2024-01-01T12:57:38.162595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_merge.to_csv('df_train_merge.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-01-01T12:57:45.929375Z","iopub.execute_input":"2024-01-01T12:57:45.929869Z","iopub.status.idle":"2024-01-01T12:57:45.941399Z","shell.execute_reply.started":"2024-01-01T12:57:45.929832Z","shell.execute_reply":"2024-01-01T12:57:45.939855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 🔵 Split dataset into train and validation","metadata":{}},{"cell_type":"code","source":"df_train_set, df_valid_set = train_test_split(df_train_merge, test_size=0.15,\n                                              stratify=df_train_merge.label, random_state=42)\nlen(df_train_set), len(df_valid_set)","metadata":{"execution":{"iopub.status.busy":"2024-01-01T12:57:51.938797Z","iopub.execute_input":"2024-01-01T12:57:51.939217Z","iopub.status.idle":"2024-01-01T12:57:51.95443Z","shell.execute_reply.started":"2024-01-01T12:57:51.939186Z","shell.execute_reply":"2024-01-01T12:57:51.952971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 🟣 **Patch images and masks**","metadata":{}},{"cell_type":"code","source":"def patch_image_and_mask(image_path, mask_path, images_dir, masks_dir,\n                         patch_size=2048,\n                         background_threshold=0.9, output_dir='.'):\n    \n    # Load image and mask\n    image = np.array(pyvips.Image.new_from_file(image_path))\n    mask = np.array(pyvips.Image.new_from_file(mask_path))\n    \n    # Get the base name for the output folders\n    name, _ = os.path.splitext(os.path.basename(image_path))\n    \n    image_folder = os.path.join(images_dir, name)\n    mask_folder = os.path.join(masks_dir, name)\n    os.makedirs(image_folder, exist_ok=True)\n    os.makedirs(mask_folder, exist_ok=True)\n\n    # Patchify the image and mask\n    image_patches = patchify(image, (patch_size, patch_size, 3), step=patch_size)\n    mask_patches = patchify(mask, (patch_size, patch_size, 3), step=patch_size)\n    \n    del image, mask\n\n    for i in range(image_patches.shape[0]):\n        for j in range(image_patches.shape[1]):\n            patch_img = image_patches[i, j, 0]\n            patch_msk = mask_patches[i, j, 0]\n\n            background_pixels = np.sum((patch_msk == [0, 0, 0]).all(axis=2))\n            total_pixels = patch_msk.size // 3  # 3 channels (R, G, B)\n            background_percentage = (background_pixels / total_pixels) \n\n            # Delete patch if background percentage exceeds the threshold\n            if background_percentage >= background_threshold:\n                continue  # Skip this patch         \n\n            patch_img_path = os.path.join(image_folder, f'{name}_{i}_{j}.png')\n            patch_msk_path = os.path.join(mask_folder, f'{name}_{i}_{j}.png')\n\n            # Resize the patch_img and patch_msk to size 512 and save them\n            Image.fromarray(patch_img).resize((512, 512)).save(patch_img_path)\n            Image.fromarray(patch_msk).resize((512, 512)).save(patch_msk_path)","metadata":{"execution":{"iopub.status.busy":"2024-01-01T12:57:54.681986Z","iopub.execute_input":"2024-01-01T12:57:54.682444Z","iopub.status.idle":"2024-01-01T12:57:54.697416Z","shell.execute_reply.started":"2024-01-01T12:57:54.682409Z","shell.execute_reply":"2024-01-01T12:57:54.696356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 🔵 Create `images` and `masks` directories","metadata":{}},{"cell_type":"code","source":"# Create 'images' and 'masks' directories if they don't exist\ntrain_images_dir = os.path.join(output_train_directory, 'images')\ntrain_masks_dir = os.path.join(output_train_directory, 'masks')\nos.makedirs(train_images_dir, exist_ok=True)\nos.makedirs(train_masks_dir, exist_ok=True)\n\nvalid_images_dir = os.path.join(output_valid_directory, 'images')\nvalid_masks_dir = os.path.join(output_valid_directory, 'masks')\nos.makedirs(valid_images_dir, exist_ok=True)\nos.makedirs(valid_masks_dir, exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2024-01-01T12:58:00.570735Z","iopub.execute_input":"2024-01-01T12:58:00.571194Z","iopub.status.idle":"2024-01-01T12:58:00.581088Z","shell.execute_reply.started":"2024-01-01T12:58:00.571159Z","shell.execute_reply":"2024-01-01T12:58:00.5796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 🟡 Patch the image and corresponding mask and save them to the specified directory","metadata":{}},{"cell_type":"code","source":"# Patch an image and its corresponding mask in the train_set\npatch_image_and_mask(df_train_set['image_path'].iloc[34], \n                     df_train_set['mask_path'].iloc[34], \n                     train_images_dir, train_masks_dir, output_dir=output_train_directory)","metadata":{"execution":{"iopub.status.busy":"2024-01-01T12:58:07.316015Z","iopub.execute_input":"2024-01-01T12:58:07.316451Z","iopub.status.idle":"2024-01-01T13:03:38.520788Z","shell.execute_reply.started":"2024-01-01T12:58:07.31642Z","shell.execute_reply":"2024-01-01T13:03:38.51899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualse a sample patch with its corresponding mask\nimg = np.array(Image.open('/kaggle/working/output_train/images/2666/2666_18_5.png'))\nmsk = np.array(Image.open('/kaggle/working/output_train/masks/2666/2666_18_5.png'))\n\nplt.figure(figsize=(12, 12))\nplt.subplot(1, 2, 1)\nplt.imshow(img);\nplt.subplot(1, 2, 2)\nplt.imshow(msk);","metadata":{"execution":{"iopub.status.busy":"2024-01-01T13:03:42.369178Z","iopub.execute_input":"2024-01-01T13:03:42.369758Z","iopub.status.idle":"2024-01-01T13:03:43.155089Z","shell.execute_reply.started":"2024-01-01T13:03:42.369712Z","shell.execute_reply":"2024-01-01T13:03:43.153449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Patch an image and its corresponding mask in the valid_set\npatch_image_and_mask(df_valid_set['image_path'].iloc[3], \n                     df_valid_set['mask_path'].iloc[3], \n                     valid_images_dir, valid_masks_dir, output_dir=output_valid_directory)","metadata":{"execution":{"iopub.status.busy":"2024-01-01T13:03:51.549386Z","iopub.execute_input":"2024-01-01T13:03:51.54995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualse a sample patch with its corresponding mask\nimg_valid = np.array(Image.open('/kaggle/working/output_valid/images/56861/56861_13_11.png'))\nmsk_valid = np.array(Image.open('/kaggle/working/output_valid/masks/56861/56861_13_11.png'))\n\nplt.figure(figsize=(12, 12))\nplt.subplot(1, 2, 1)\nplt.imshow(img_valid);\nplt.subplot(1, 2, 2)\nplt.imshow(msk_valid);","metadata":{"execution":{"iopub.status.busy":"2024-01-01T11:45:10.409989Z","iopub.execute_input":"2024-01-01T11:45:10.411379Z","iopub.status.idle":"2024-01-01T11:45:11.230827Z","shell.execute_reply.started":"2024-01-01T11:45:10.411323Z","shell.execute_reply":"2024-01-01T11:45:11.22906Z"},"trusted":true},"execution_count":null,"outputs":[]}]}