{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!yes | sudo dpkg -i /kaggle/input/libvips-pyvips-installation-and-getting-started/libvips/*.deb\n!pip install /kaggle/input/libvips-pyvips-installation-and-getting-started/pyvips/pyvips-2.2.1-py2.py3-none-any.whl --no-index --find-links /kaggle/input/libvips-pyvips-installation-and-getting-started/pyvips\n! clear","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-11-05T23:09:45.002523Z","iopub.execute_input":"2023-11-05T23:09:45.002818Z","iopub.status.idle":"2023-11-05T23:10:23.589064Z","shell.execute_reply.started":"2023-11-05T23:09:45.002792Z","shell.execute_reply":"2023-11-05T23:10:23.587797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport shutil\n\nimport numpy as np\nimport pandas as pd\nimport torch\nimport matplotlib.pyplot as plt\nimport glob\nfrom PIL import Image\n\nfrom datasets import load_dataset\n\n# !ls /kaggle/input/pyvips-python-and-deb-package\n# # intall the deb packages\n# !yes | sudo dpkg -i /kaggle/input/pyvips-python-and-deb-package-gpu/linux_packages/archives/*.deb\n# # install the python wrapper\n# !pip install pyvips -f /kaggle/input/pyvips-python-and-deb-package/python_packages/ --no-index \nimport pyvips\n\n","metadata":{"execution":{"iopub.status.busy":"2023-11-05T23:10:23.591459Z","iopub.execute_input":"2023-11-05T23:10:23.591844Z","iopub.status.idle":"2023-11-05T23:10:28.359486Z","shell.execute_reply.started":"2023-11-05T23:10:23.591809Z","shell.execute_reply":"2023-11-05T23:10:28.358664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(\"/kaggle/input/UBC-OCEAN/train.csv\")\ntest_df = pd.read_csv(\"/kaggle/input/UBC-OCEAN/test.csv\")\n\n\nBASE_DIR = [\"/kaggle/input/UBC-OCEAN/train_thumbnails/\", \"/kaggle/input/UBC-OCEAN/test_thumbnails/\"]\nTRAIN_DIR= \"/kaggle/input/UBC-OCEAN/train_images\"","metadata":{"execution":{"iopub.status.busy":"2023-11-05T23:10:28.36101Z","iopub.execute_input":"2023-11-05T23:10:28.361538Z","iopub.status.idle":"2023-11-05T23:10:28.380527Z","shell.execute_reply.started":"2023-11-05T23:10:28.361508Z","shell.execute_reply":"2023-11-05T23:10:28.379795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def organize_images_by_label(df: pd.DataFrame, source_dir: str) -> None:\n    image_paths=[]\n    for _, row in df.iterrows():\n        image_id = row[\"image_id\"]\n        label = row[\"label\"]\n        source_path = os.path.join(source_dir, f\"{image_id}_thumbnail.png\") \n        try:\n            image_paths.append(os.path.join(source_dir, f\"{image_id}_thumbnail.png\"))\n        except FileNotFoundError:\n            image_paths.append(1)\n            continue\n    return image_paths\n\n\nimage_paths = organize_images_by_label(train_df, BASE_DIR[0])\ntrain_df['image_path'] = image_paths\n","metadata":{"execution":{"iopub.status.busy":"2023-11-05T23:10:28.382741Z","iopub.execute_input":"2023-11-05T23:10:28.383409Z","iopub.status.idle":"2023-11-05T23:10:28.768035Z","shell.execute_reply.started":"2023-11-05T23:10:28.383373Z","shell.execute_reply":"2023-11-05T23:10:28.767227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport numpy as np\n\ndef is_informative(patch, threshold=25):\n    \"\"\"\n    Check if a patch is informative based on standard deviation of its pixel values.\n    \n    Parameters:\n    - patch (numpy array): The image patch.\n    - threshold (float): The minimum standard deviation for a patch to be considered informative.\n\n    Returns:\n    - bool: True if the patch is informative, False otherwise.\n    \"\"\"\n    # Convert the patch to grayscale for simplicity\n    gray = cv2.cvtColor(patch, cv2.COLOR_BGR2GRAY)\n    # Check if the standard deviation exceeds the threshold\n    return np.std(gray) > threshold and len(np.unique(np.int16(gray)))>50\n\ndef extract_informative_patches(image_path, patch_size=(256, 256), num_patches=10):\n    \"\"\"\n    Extract informative patches from an image.\n    \n    Parameters:\n    - image_path (str): Path to the input image.\n    - patch_size (tuple): Size of the patches to extract. Default is (256, 256).\n    - num_patches (int): Number of random patches to extract. Default is 10.\n\n    Returns:\n    - list: List of extracted informative patches.\n    \"\"\"\n\n    # Load the image\n    image = np.array(pyvips.Image.new_from_file(image_path))\n\n    # Check if image loaded correctly\n    if image is None:\n        raise ValueError(\"Could not load the image from the provided path.\")\n\n    # Image dimensions\n    img_height, img_width = image.shape[:2]\n\n    # Ensure the image is large enough for the patch size\n    if img_width < patch_size[0] or img_height < patch_size[1]:\n        raise ValueError(\"Image dimensions are smaller than the patch size.\")\n\n    patches = []\n    attempts = 0\n    max_attempts = num_patches * 10  # Max number of tries to extract informative patches\n\n    while len(patches) < num_patches and attempts < max_attempts:\n        # Randomly select top-left corner of the patch\n        x = np.random.randint(0, img_width - patch_size[0])\n        y = np.random.randint(0, img_height - patch_size[1])\n\n        # Extract patch\n        patch = image[y:y+patch_size[1], x:x+patch_size[0]]\n        \n        # Check if the patch is informative\n        if is_informative(patch):\n            patches.append(patch)\n\n        attempts += 1\n    return patches\n\ndef zoom_and_crop(image, zoom_factor=1):\n    # Zoom (resize) the image\n    h, w = image.shape[:2]\n    enlarged = cv2.resize(image, (w*zoom_factor, h*zoom_factor), interpolation=cv2.INTER_LINEAR)\n\n    # Crop the center\n    center_x, center_y = enlarged.shape[1] // 2, enlarged.shape[0] // 2\n    left_x, right_x = center_x - w//2, center_x + w//2\n    top_y, bottom_y = center_y - h//2, center_y + h//2\n    \n    cropped = enlarged[top_y:bottom_y, left_x:right_x]\n    return cropped\n\ndef plot_patches(patches):\n    fig, axes = plt.subplots(1, len(patches), figsize=(20, 5))\n    for ax, patch in zip(axes, patches):\n        ax.imshow(patch)\n        ax.axis('off')\n    plt.show()\n    \ndef random_centered_crop(image_path, patch_size=512, num_patches=10):\n    \"\"\"\n    Generate random patches centered around the image center.\n\n    Args:\n    - image_path (str): Input image path.\n    - patch_size (int): Size of the side for square cropping. Default is 256.\n    - num_patches (int): Number of random patches to generate. Default is 10.\n    \n    Returns:\n    - list: List of randomly cropped patches.\n    \"\"\"\n    image = np.array(pyvips.Image.new_from_file(image_path))\n    height, width = image.shape[:2]\n    center_x, center_y = width // 2, height // 2\n    patches = []\n\n    for _ in range(num_patches):\n        # Generate random offsets\n        offset_x = np.random.randint(-patch_size//2, patch_size//2)\n        offset_y = np.random.randint(-patch_size//2, patch_size//2)\n        startx = center_x + offset_x - patch_size // 2\n        starty = center_y + offset_y - patch_size // 2\n        endx = startx + patch_size\n        endy = starty + patch_size\n\n        # Handle cases where the crop goes out of image boundaries\n        startx = max(0, startx)\n        starty = max(0, starty)\n        endx = min(width, endx)\n        endy = min(height, endy)\n\n        patch = image[starty:endy, startx:endx]\n        patch = cv2.resize(patch, (256, 256), interpolation=cv2.INTER_LINEAR)\n        patches.append(patch)\n\n    return patches\n\n\ndef resize(image):\n    resized_img = cv2.resize(image, (256, 256), interpolation=cv2.INTER_LINEAR)\n    return resized_img\n\ndef save_patches(patches, image_id):\n    source_dir=\"/kaggle/working/train/\"+str(image_id)\n    if not os.path.isdir(source_dir):\n        os.makedirs(source_dir)\n    for idx, patch in enumerate(patches):\n        filename = os.path.join(source_dir, f\"{image_id}_patch_{idx}.jpg\")\n        cv2.imwrite(filename, patch)","metadata":{"execution":{"iopub.status.busy":"2023-11-05T23:10:28.76923Z","iopub.execute_input":"2023-11-05T23:10:28.769517Z","iopub.status.idle":"2023-11-05T23:10:28.972999Z","shell.execute_reply.started":"2023-11-05T23:10:28.769492Z","shell.execute_reply":"2023-11-05T23:10:28.972167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for _, n in train_df.iterrows():\n    print(os.path.join(TRAIN_DIR, str(n['image_id'])+\".png\"), \":\", n['label'])\n    \n    if n['is_tma']:\n        patches = random_centered_crop(os.path.join(TRAIN_DIR, str(n['image_id'])+\".png\"))\n        save_patches(patches, n['image_id'])\n        plot_patches(patches)\n    else:\n        patches = extract_informative_patches(os.path.join(TRAIN_DIR, str(n['image_id'])+\".png\"))\n        save_patches(patches, n['image_id'])\n        plot_patches(patches)    \n    if _==10:\n        break","metadata":{"execution":{"iopub.status.busy":"2023-11-05T23:11:26.256524Z","iopub.execute_input":"2023-11-05T23:11:26.257408Z","iopub.status.idle":"2023-11-05T23:13:22.360404Z","shell.execute_reply.started":"2023-11-05T23:11:26.257375Z","shell.execute_reply":"2023-11-05T23:13:22.359373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}