{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"},{"sourceId":6774400,"sourceType":"datasetVersion","datasetId":3895136},{"sourceId":6774553,"sourceType":"datasetVersion","datasetId":3898019},{"sourceId":6862644,"sourceType":"datasetVersion","datasetId":3944110}],"dockerImageVersionId":30579,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!ls /kaggle/input/pyvips-python-and-deb-package-gpu\n# intall the deb packages\n!yes | dpkg -i --force-depends /kaggle/input/pyvips-python-and-deb-package/linux_packages/archives/*.deb\n# install the python wrapper\n!pip install pyvips -f /kaggle/input/pyvips-python-and-deb-package/python_packages/ --no-index\n!pip list | grep pyvips\n\nfrom IPython import display\ndisplay.clear_output()\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-11-15T11:51:45.087455Z","iopub.execute_input":"2023-11-15T11:51:45.087867Z","iopub.status.idle":"2023-11-15T11:53:06.916475Z","shell.execute_reply.started":"2023-11-15T11:51:45.087836Z","shell.execute_reply":"2023-11-15T11:53:06.914998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os, glob\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport re\n\nDATASET_FOLDER = \"/kaggle/input/UBC-OCEAN/\"\nIMAGES_FOLDER = \"./test_tiles\"\n\nos.environ['VIPS_CONCURRENCY'] = '4'\nos.environ['VIPS_DISC_THRESHOLD'] = '15gb'","metadata":{"execution":{"iopub.status.busy":"2023-11-15T11:53:06.919417Z","iopub.execute_input":"2023-11-15T11:53:06.919832Z","iopub.status.idle":"2023-11-15T11:53:07.347573Z","shell.execute_reply.started":"2023-11-15T11:53:06.919798Z","shell.execute_reply":"2023-11-15T11:53:07.346432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\nimport io\n","metadata":{"execution":{"iopub.status.busy":"2023-11-15T11:53:07.348952Z","iopub.execute_input":"2023-11-15T11:53:07.349437Z","iopub.status.idle":"2023-11-15T11:53:20.648159Z","shell.execute_reply.started":"2023-11-15T11:53:07.349403Z","shell.execute_reply":"2023-11-15T11:53:20.647218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport pyvips\nimport numpy as np\nimport random\nfrom PIL import Image\nfrom IPython import display\nfrom tqdm import tqdm\ndef extract_image_tiles(\n    p_img,label, folder, size: int = 2048, scale: float = 0.5,\n    drop_thr: float = 0.6, white_thr: int = 240, max_samples: int = 50\n) -> list:\n    name, _ = os.path.splitext(os.path.basename(p_img))\n    im = pyvips.Image.new_from_file(p_img)\n    w = h = size\n    # https://stackoverflow.com/a/47581978/4521646\n    idxs = [(y, y + h, x, x + w) for y in range(0, im.height, h) for x in range(0, im.width, w)]\n    # random subsample\n    max_samples = max_samples if isinstance(max_samples, int) else int(len(idxs) * max_samples)\n    random.shuffle(idxs)\n    files = []\n    for y, y_, x, x_ in idxs:        # https://libvips.github.io/pyvips/vimage.html#pyvips.Image.crop\n\n    #for y, y_, x, x_ in tqdm(idxs, total=len(idxs)):        # https://libvips.github.io/pyvips/vimage.html#pyvips.Image.crop\n        tile = im.crop(x, y, min(w, im.width - x), min(h, im.height - y)).numpy()[..., :3]\n        if tile.shape[:2] != (h, w):\n            tile_ = tile\n            tile_size = (h, w) if tile.ndim == 2 else (h, w, tile.shape[2])\n            tile = np.zeros(tile_size, dtype=tile.dtype)\n            tile[:tile_.shape[0], :tile_.shape[1], ...] = tile_\n        black_bg = np.sum(tile, axis=2) == 0\n        tile[black_bg, :] = 255\n        img = np.dot(tile[..., :3], [0.2989, 0.5870, 0.1140]).astype(np.uint8)\n        white_pixels = np.sum(img>220)\n#         mask_bg = np.mean(tile, axis=2) > white_thr\n        if np.sum(white_pixels) >= (np.prod(img.shape) * drop_thr):\n#             display.clear_output()\n#             plt.imshow(tile)\n#             plt.show()\n            continue\n        p_img = os.path.join(folder, f\"label-{label}-{int(x_ / w)}-{int(y_ / h)}.png\")\n        # print(tile.shape, tile.dtype, tile.min(), tile.max())\n#         new_size = int(size * scale), int(size * scale)\n         #Image.fromarray(tile).resize(new_size, Image.LANCZOS).save(p_img)\n        Image.fromarray(tile).save(p_img)\n\n        files.append(p_img)\n        # need to set counter check as some empty tiles could be skipped earlier\n        if len(files) >= max_samples:\n            break\n    return files","metadata":{"execution":{"iopub.status.busy":"2023-11-15T17:55:43.898764Z","iopub.execute_input":"2023-11-15T17:55:43.899211Z","iopub.status.idle":"2023-11-15T17:55:44.367326Z","shell.execute_reply.started":"2023-11-15T17:55:43.899163Z","shell.execute_reply":"2023-11-15T17:55:44.365482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from torchvision import transforms as T\n\nimg_color_mean = [0.8721593659261734, 0.7799686061900686, 0.8644588534918227]\nimg_color_std = [0.08258995918115268, 0.10991684444009092, 0.06839816226731532]\n\nVALID_TRANSFORM = T.Compose([\n    T.CenterCrop(512),\n    T.ToTensor(),\n    #T.Normalize([0.485, 0.456, 0.406], [0.229, 0.224, 0.225]),\n    T.Normalize(img_color_mean, img_color_std),  # custom\n])","metadata":{"execution":{"iopub.status.busy":"2023-11-15T11:53:21.002837Z","iopub.execute_input":"2023-11-15T11:53:21.003302Z","iopub.status.idle":"2023-11-15T11:53:24.907077Z","shell.execute_reply.started":"2023-11-15T11:53:21.00327Z","shell.execute_reply":"2023-11-15T11:53:24.905851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nfrom PIL import Image\nfrom torch.utils.data import Dataset\n\nclass TilesFolderDataset(Dataset):\n\n    def __init__(\n        self,\n        folder: str,\n        image_ext: str =  '.png',\n        transforms = None\n    ):\n        assert os.path.isdir(folder)\n        self.transforms = transforms\n        self.imgs = glob.glob(os.path.join(folder, \"*\" + image_ext))\n\n    def __getitem__(self, idx: int) -> tuple:\n        img_path = self.imgs[idx]\n        assert os.path.isfile(img_path), f\"missing: {img_path}\"\n        img = np.array(Image.open(img_path))[..., :3]\n        # filter background\n        mask = np.sum(img, axis=2) == 0\n        img[mask, :] = 255\n        if np.max(img) < 1.5:\n            img = np.clip(img * 255, 0, 255).astype(np.uint8)\n        # augmentation\n        if self.transforms:\n            img = self.transforms(Image.fromarray(img))\n        #print(f\"img dim: {img.shape}\")\n        return img\n\n    def __len__(self) -> int:\n        return len(self.imgs)","metadata":{"execution":{"iopub.status.busy":"2023-11-15T11:53:24.90874Z","iopub.execute_input":"2023-11-15T11:53:24.909575Z","iopub.status.idle":"2023-11-15T11:53:24.922985Z","shell.execute_reply.started":"2023-11-15T11:53:24.909533Z","shell.execute_reply":"2023-11-15T11:53:24.921706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def extract_prune_tiles(\n    path_img: str,label:str, folder: str, size: int = 2048, scale: float = 0.25,\n    drop_thr: float = 0.6,white_thr: int=0.5, max_samples: int = 12000\n) -> str:\n    print(f\"processing: {path_img}\")\n    name, _ = os.path.splitext(os.path.basename(path_img))\n    folder = os.path.join(folder, name)\n    os.makedirs(folder, exist_ok=True)\n    tiles = extract_image_tiles(\n        path_img,label, folder, size=size, scale=scale,\n        drop_thr=drop_thr,white_thr=225 ,max_samples=max_samples)\n    return folder","metadata":{"execution":{"iopub.status.busy":"2023-11-15T11:53:24.925022Z","iopub.execute_input":"2023-11-15T11:53:24.925418Z","iopub.status.idle":"2023-11-15T11:53:24.937582Z","shell.execute_reply.started":"2023-11-15T11:53:24.925387Z","shell.execute_reply":"2023-11-15T11:53:24.936309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nfull_df=pd.read_csv(\"/kaggle/input/cancerdatasetwithpath/cancerdf.csv\")\nfull_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-15T11:53:24.939287Z","iopub.execute_input":"2023-11-15T11:53:24.939665Z","iopub.status.idle":"2023-11-15T11:53:24.99167Z","shell.execute_reply.started":"2023-11-15T11:53:24.939634Z","shell.execute_reply":"2023-11-15T11:53:24.990409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import shutil\n\ndef record(id):\n    folderpath='/kaggle/working/test_tiles'\n    folder = glob.glob(os.path.join(folderpath, \"*\"))\n    image_list=[]\n    for path in folder:\n        image_list= image_list+(glob.glob(os.path.join(path,\"*.png\")))\n        print(len(image_list))\n\n    random.shuffle(image_list)\n    path=f\"{id}xtwentyimages.tfrecords\"\n    with tf.io.TFRecordWriter(path) as writer:\n        for imagepath in image_list:\n\n            pattern = r'label-(\\d+)-\\d+-\\d+\\.png'\n\n            match = re.search(pattern, imagepath)\n            label = match.group(1)\n            label=int(label)\n            image = Image.open(imagepath)\n\n            bytes_buffer = io.BytesIO()\n            image.convert(\"RGB\").save(bytes_buffer, \"JPEG\")\n            image_bytes = bytes_buffer.getvalue()\n\n            bytes_feature = tf.train.Feature(bytes_list=tf.train.BytesList(value=[image_bytes]))\n            class_feature = tf.train.Feature(int64_list=tf.train.Int64List(value=[label]))\n\n            example = tf.train.Example(\n              features=tf.train.Features(feature={\n                  \"image\": bytes_feature,\n                  \"class\": class_feature\n              })\n            )\n\n            writer.write(example.SerializeToString())\n\n            image.close()\n    for directory in folder:\n        try:\n            shutil.rmtree(directory)\n            print(f\"Directory '{directory}' removed successfully.\")\n        except OSError as e:\n            print(f\"Error removing directory '{directory}': {e}\")\n","metadata":{"execution":{"iopub.status.busy":"2023-11-15T11:53:24.99339Z","iopub.execute_input":"2023-11-15T11:53:24.994792Z","iopub.status.idle":"2023-11-15T11:53:25.009236Z","shell.execute_reply.started":"2023-11-15T11:53:24.994732Z","shell.execute_reply":"2023-11-15T11:53:25.007768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\npreviousbatch=150\nbatch_size = 10\ndf=full_df[previousbatch:200]\ngrouped = df.groupby(df.index // batch_size)\nfor idx,(group_id, group_data) in enumerate(grouped):\n    for idx,row in group_data.iterrows():\n        path=row['Image_path']\n        label=row['label']\n        folder_tiles = extract_prune_tiles(path,label, IMAGES_FOLDER, size=256, scale=1,drop_thr=0.4,white_thr=222)\n    record(idx+previousbatch)\n# print(f\"found tiles: {len(dataset)}\")\n\n# # quick view\n# fig, axes = plt.subplots(nrows=3, ncols=3, figsize=(10, 10))\n# for i in range(9):\n#     img = dataset[i]\n#     axes[i // 3, i % 3].imshow(img)\n# fig.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2023-11-15T17:59:48.091025Z","iopub.execute_input":"2023-11-15T17:59:48.091519Z","iopub.status.idle":"2023-11-15T17:59:48.195263Z","shell.execute_reply.started":"2023-11-15T17:59:48.091481Z","shell.execute_reply":"2023-11-15T17:59:48.194075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# dataset = TilesFolderDataset(\"/kaggle/working/test_tiles/1952\")\n# print(f\"found tiles: {len(dataset)}\")\n\n# # quick view\n# fig, axes = plt.subplots(nrows=10, ncols=10, figsize=(10, 10))\n# axes=axes.flatten()\n# for i in range(100):\n#     ax=axes[i]\n#     img = dataset[i+2000]\n#     ax.imshow(img)\n#     ax.axis('off')\n# fig.tight_layout()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"hello\")","metadata":{"execution":{"iopub.status.busy":"2023-11-15T18:00:11.239117Z","iopub.execute_input":"2023-11-15T18:00:11.239615Z","iopub.status.idle":"2023-11-15T18:00:11.246492Z","shell.execute_reply.started":"2023-11-15T18:00:11.239578Z","shell.execute_reply":"2023-11-15T18:00:11.244857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# import os\n# import glob\n# import random\n# import re\n# import io\n# from PIL import Image\n# import tensorflow as tf\n\n# def record():\n#     folderpath = '/kaggle/working/test_tiles'\n#     folder = glob.glob(os.path.join(folderpath, \"*\"))\n#     image_list = []\n\n#     for path in folder:\n#         image_list = image_list + (glob.glob(os.path.join(path, \"*.png\")))\n#         print(len(image_list))\n\n#     random.shuffle(image_list)\n\n#     # Open the TFRecord file in append mode ('ab')\n#     with tf.io.TFRecordWriter(\"caltech_dataset.tfrecords\", options='ab') as writer:\n#         for imagepath in image_list:\n#             pattern = r'label-(\\d+)-\\d+-\\d+\\.png'\n#             match = re.search(pattern, imagepath)\n#             label = match.group(1)\n#             label = int(label)\n#             image = Image.open(imagepath)\n\n#             bytes_buffer = io.BytesIO()\n#             image.convert(\"RGB\").save(bytes_buffer, \"JPEG\")\n#             image_bytes = bytes_buffer.getvalue()\n\n#             bytes_feature = tf.train.Feature(bytes_list=tf.train.BytesList(value=[image_bytes]))\n#             class_feature = tf.train.Feature(int64_list=tf.train.Int64List(value=[label]))\n\n#             example = tf.train.Example(\n#                 features=tf.train.Features(feature={\n#                     \"image\": bytes_feature,\n#                     \"class\": class_feature\n#                 })\n#             )\n\n#             writer.write(example.SerializeToString())\n\n#             image.close()\n\n#     # Removing directories after processing\n#     for directory in folder:\n#         try:\n#             shutil.rmtree(directory)\n#             print(f\"Directory '{directory}' removed successfully.\")\n#         except OSError as e:\n#             print(f\"Error removing directory '{directory}': {e}\")\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## code to extract the images from tfr record\n","metadata":{}},{"cell_type":"code","source":"# image_feature_description = {\n#     \"image\": tf.io.FixedLenFeature([], tf.string), \n#     \"class\": tf.io.FixedLenFeature([], tf.int64), \n#     }\n# def _parse_data(unparsed_example):\n#     return tf.io.parse_single_example(unparsed_example, image_feature_description)\n\n# def _bytestring_to_pixels(parsed_example):\n#     byte_string = parsed_example['image']\n#     image = tf.io.decode_image(byte_string)\n#     image = tf.reshape(image, [256, 256, 3])\n#     return image, parsed_example[\"class\"]\n# AUTOTUNE = tf.data.AUTOTUNE\n# def load_and_extract_images(filepath):\n#     dataset = tf.data.TFRecordDataset(filepath)\n#     dataset = dataset.map(_parse_data, num_parallel_calls=AUTOTUNE)\n#     dataset = dataset.map(_bytestring_to_pixels, num_parallel_calls=AUTOTUNE) # .cache()\n#     return dataset\n\n# caltech_dataset = load_and_extract_images(\"caltech_dataset.tfrecords\")\n\n# train_dataset = caltech_dataset.take(90)\n\n# cropsize=256\n# def _train_data_preprocess_and_augment(image, label):\n# #     image = tf.cast(image, tf.float32)\n# #     image = tf.image.random_crop(image, size=[crop_size, crop_size, 3])\n    \n#     return image, label\n# train_preprocessed_augmented = train_dataset.map(_train_data_preprocess_and_augment)\n# fig,axes=plt.subplots(8,4,figsize=(10,10))\n# axes=axes.flatten()\n# for(images, label_batch) in train_preprocessed_augmented.batch(32):\n#     for i in range(32):\n#         ax=axes[i] \n#         ax.imshow(images[i])\n#         ax.set_title(f\"label-{label_batch[i]}\")\n\n#     plt.axis(\"off\")\n#     plt.show()\n#     break\n        \n    ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}