{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"},{"sourceId":6774400,"sourceType":"datasetVersion","datasetId":3895136},{"sourceId":6774553,"sourceType":"datasetVersion","datasetId":3898019},{"sourceId":6862644,"sourceType":"datasetVersion","datasetId":3944110},{"sourceId":7143890,"sourceType":"datasetVersion","datasetId":4123551}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!ls /kaggle/input/pyvips-python-and-deb-package-gpu\n# intall the deb packages\n!yes | dpkg -i --force-depends /kaggle/input/pyvips-python-and-deb-package/linux_packages/archives/*.deb\n# install the python wrapper\n!pip install pyvips -f /kaggle/input/pyvips-python-and-deb-package/python_packages/ --no-index\n!pip list | grep pyvips\n\nfrom IPython import display\ndisplay.clear_output()\n","metadata":{"execution":{"iopub.status.busy":"2023-12-07T06:20:41.547406Z","iopub.execute_input":"2023-12-07T06:20:41.547874Z","iopub.status.idle":"2023-12-07T06:22:11.636451Z","shell.execute_reply.started":"2023-12-07T06:20:41.547841Z","shell.execute_reply":"2023-12-07T06:22:11.634635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os, glob\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport re\n\nDATASET_FOLDER = \"/kaggle/input/UBC-OCEAN/\"\nIMAGES_FOLDER = \"./test_tiles\"\n\nos.environ['VIPS_CONCURRENCY'] = '4'\nos.environ['VIPS_DISC_THRESHOLD'] = '15gb'","metadata":{"execution":{"iopub.status.busy":"2023-12-07T06:22:11.640472Z","iopub.execute_input":"2023-12-07T06:22:11.64113Z","iopub.status.idle":"2023-12-07T06:22:11.64967Z","shell.execute_reply.started":"2023-12-07T06:22:11.641078Z","shell.execute_reply":"2023-12-07T06:22:11.648539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\nimport io\n","metadata":{"execution":{"iopub.status.busy":"2023-12-07T06:22:11.650845Z","iopub.execute_input":"2023-12-07T06:22:11.65213Z","iopub.status.idle":"2023-12-07T06:22:11.666233Z","shell.execute_reply.started":"2023-12-07T06:22:11.652078Z","shell.execute_reply":"2023-12-07T06:22:11.665121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch \nmodel_Net = torch.jit.load('/kaggle/input/stainnet/model_scripted.pt')\n","metadata":{"execution":{"iopub.status.busy":"2023-12-07T06:22:11.668953Z","iopub.execute_input":"2023-12-07T06:22:11.670218Z","iopub.status.idle":"2023-12-07T06:22:11.695626Z","shell.execute_reply.started":"2023-12-07T06:22:11.670156Z","shell.execute_reply":"2023-12-07T06:22:11.694149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n#model_Net = StainNet()\n\nmodel_Net.load_state_dict(torch.load(\"/kaggle/input/stainnet/StainNet-Public_layer3_ch32 (1).pth\",map_location=torch.device('cpu')))\nmodel_Net.eval()\n\ndef norm(image):\n    image = np.array(image).astype(np.float32)\n    image = image.transpose((2, 0, 1))\n    image = ((image / 255) - 0.5) / 0.5\n    image=image[np.newaxis, ...]\n    image=torch.from_numpy(image)\n    return image\n\ndef un_norm(image):\n    image = image.cpu().detach().numpy()[0]\n    image = ((image * 0.5 + 0.5) * 255).astype(np.uint8).transpose((1,2,0))\n    return image\n\ndef stain_normalize1(source, verbose=False):\n    with torch.no_grad():\n        img_net=model_Net(norm(source))\n        img_net=un_norm(img_net)\n        if verbose: plt.imshow(img_net); plt.show()\n        return img_net","metadata":{"execution":{"iopub.status.busy":"2023-12-07T06:22:11.697555Z","iopub.execute_input":"2023-12-07T06:22:11.69923Z","iopub.status.idle":"2023-12-07T06:22:11.716318Z","shell.execute_reply.started":"2023-12-07T06:22:11.699186Z","shell.execute_reply":"2023-12-07T06:22:11.714744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport pyvips\nimport numpy as np\nimport random\nfrom PIL import Image\nfrom IPython import display\nfrom tqdm import tqdm\ndef extract_image_tiles(\n    p_img,label, folder, size: int = 2048, scale: float = 0.5,\n    drop_thr: float = 0.6, white_thr: int = 240, max_samples: int = 50\n) -> list:\n    name, _ = os.path.splitext(os.path.basename(p_img))\n    im = pyvips.Image.new_from_file(p_img)\n    w = h = size\n    # https://stackoverflow.com/a/47581978/4521646\n    idxs = [(y, y + h, x, x + w) for y in range(0, im.height, h) for x in range(0, im.width, w)]\n    # random subsample\n    max_samples = max_samples if isinstance(max_samples, int) else int(len(idxs) * max_samples)\n    random.shuffle(idxs)\n    files = []\n    for y, y_, x, x_ in idxs:        # https://libvips.github.io/pyvips/vimage.html#pyvips.Image.crop\n\n    #for y, y_, x, x_ in tqdm(idxs, total=len(idxs)):        # https://libvips.github.io/pyvips/vimage.html#pyvips.Image.crop\n        tile = im.crop(x, y, min(w, im.width - x), min(h, im.height - y)).numpy()[..., :3]\n        if tile.shape[:2] != (h, w):\n            tile_ = tile\n            tile_size = (h, w) if tile.ndim == 2 else (h, w, tile.shape[2])\n            tile = np.zeros(tile_size, dtype=tile.dtype)\n            tile[:tile_.shape[0], :tile_.shape[1], ...] = tile_\n        black_bg = np.sum(tile, axis=2) == 0\n        tile[black_bg, :] = 255\n        tile=stain_normalize1(tile)\n        img = np.dot(tile[..., :3], [0.2989, 0.5870, 0.1140]).astype(np.uint8)\n        white_pixels = np.sum(img>220)\n#         mask_bg = np.mean(tile, axis=2) > white_thr\n        if np.sum(white_pixels) >= (np.prod(img.shape) * drop_thr):\n#             display.clear_output()\n#             plt.imshow(tile)\n#             plt.show()\n            continue\n        p_img = os.path.join(folder, f\"label-{label}-{int(x_ / w)}-{int(y_ / h)}.png\")\n        # print(tile.shape, tile.dtype, tile.min(), tile.max())\n#         new_size = int(size * scale), int(size * scale)\n         #Image.fromarray(tile).resize(new_size, Image.LANCZOS).save(p_img)\n        Image.fromarray(tile).save(p_img)\n\n        files.append(p_img)\n        # need to set counter check as some empty tiles could be skipped earlier\n        if len(files) >= max_samples:\n            break\n    return files","metadata":{"execution":{"iopub.status.busy":"2023-12-07T06:22:11.718639Z","iopub.execute_input":"2023-12-07T06:22:11.719241Z","iopub.status.idle":"2023-12-07T06:22:12.538454Z","shell.execute_reply.started":"2023-12-07T06:22:11.719196Z","shell.execute_reply":"2023-12-07T06:22:12.53716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from torchvision import transforms as T\n\nimg_color_mean = [0.8721593659261734, 0.7799686061900686, 0.8644588534918227]\nimg_color_std = [0.08258995918115268, 0.10991684444009092, 0.06839816226731532]\n\nVALID_TRANSFORM = T.Compose([\n    T.CenterCrop(512),\n    T.ToTensor(),\n    #T.Normalize([0.485, 0.456, 0.406], [0.229, 0.224, 0.225]),\n    T.Normalize(img_color_mean, img_color_std),  # custom\n])","metadata":{"execution":{"iopub.status.busy":"2023-12-07T06:22:12.540244Z","iopub.execute_input":"2023-12-07T06:22:12.540742Z","iopub.status.idle":"2023-12-07T06:22:12.926618Z","shell.execute_reply.started":"2023-12-07T06:22:12.540695Z","shell.execute_reply":"2023-12-07T06:22:12.925168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nfrom PIL import Image\nfrom torch.utils.data import Dataset\n\nclass TilesFolderDataset(Dataset):\n\n    def __init__(\n        self,\n        folder: str,\n        image_ext: str =  '.png',\n        transforms = None\n    ):\n        assert os.path.isdir(folder)\n        self.transforms = transforms\n        self.imgs = glob.glob(os.path.join(folder, \"*\" + image_ext))\n\n    def __getitem__(self, idx: int) -> tuple:\n        img_path = self.imgs[idx]\n        assert os.path.isfile(img_path), f\"missing: {img_path}\"\n        img = np.array(Image.open(img_path))[..., :3]\n        # filter background\n        mask = np.sum(img, axis=2) == 0\n        img[mask, :] = 255\n        if np.max(img) < 1.5:\n            img = np.clip(img * 255, 0, 255).astype(np.uint8)\n        # augmentation\n        if self.transforms:\n            img = self.transforms(Image.fromarray(img))\n        #print(f\"img dim: {img.shape}\")\n        return img\n\n    def __len__(self) -> int:\n        return len(self.imgs)","metadata":{"execution":{"iopub.status.busy":"2023-12-07T06:22:12.928212Z","iopub.execute_input":"2023-12-07T06:22:12.928589Z","iopub.status.idle":"2023-12-07T06:22:12.940731Z","shell.execute_reply.started":"2023-12-07T06:22:12.928557Z","shell.execute_reply":"2023-12-07T06:22:12.939539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def extract_prune_tiles(\n    path_img: str,label:str, folder: str, size: int = 2048, scale: float = 0.25,\n    drop_thr: float = 0.6,white_thr: int=0.5, max_samples: int = 12000\n) -> str:\n    print(f\"processing: {path_img}\")\n    name, _ = os.path.splitext(os.path.basename(path_img))\n    folder = os.path.join(folder, name)\n    os.makedirs(folder, exist_ok=True)\n    tiles = extract_image_tiles(\n        path_img,label, folder, size=size, scale=scale,\n        drop_thr=drop_thr,white_thr=225 ,max_samples=max_samples)\n    return folder","metadata":{"execution":{"iopub.status.busy":"2023-12-07T06:22:12.942182Z","iopub.execute_input":"2023-12-07T06:22:12.943486Z","iopub.status.idle":"2023-12-07T06:22:12.958996Z","shell.execute_reply.started":"2023-12-07T06:22:12.943448Z","shell.execute_reply":"2023-12-07T06:22:12.957559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nfull_df=pd.read_csv(\"/kaggle/input/cancerdatasetwithpath/cancerdf.csv\")\nfull_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-07T06:22:46.373269Z","iopub.execute_input":"2023-12-07T06:22:46.373793Z","iopub.status.idle":"2023-12-07T06:22:46.400684Z","shell.execute_reply.started":"2023-12-07T06:22:46.373753Z","shell.execute_reply":"2023-12-07T06:22:46.399073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import shutil\n\ndef record(id):\n    folderpath='/kaggle/working/test_tiles'\n    folder = glob.glob(os.path.join(folderpath, \"*\"))\n    image_list=[]\n    for path in folder:\n        image_list= image_list+(glob.glob(os.path.join(path,\"*.png\")))\n        print(len(image_list))\n\n    random.shuffle(image_list)\n    path=f\"{id}xtwentyimages.tfrecords\"\n    with tf.io.TFRecordWriter(path) as writer:\n        for imagepath in image_list:\n\n            pattern = r'label-(\\d+)-\\d+-\\d+\\.png'\n\n            match = re.search(pattern, imagepath)\n            label = match.group(1)\n            label=int(label)\n            image = Image.open(imagepath)\n\n            bytes_buffer = io.BytesIO()\n            image.convert(\"RGB\").save(bytes_buffer, \"JPEG\")\n            image_bytes = bytes_buffer.getvalue()\n\n            bytes_feature = tf.train.Feature(bytes_list=tf.train.BytesList(value=[image_bytes]))\n            class_feature = tf.train.Feature(int64_list=tf.train.Int64List(value=[label]))\n\n            example = tf.train.Example(\n              features=tf.train.Features(feature={\n                  \"image\": bytes_feature,\n                  \"class\": class_feature\n              })\n            )\n\n            writer.write(example.SerializeToString())\n\n            image.close()\n    for directory in folder:\n        try:\n            shutil.rmtree(directory)\n            print(f\"Directory '{directory}' removed successfully.\")\n        except OSError as e:\n            print(f\"Error removing directory '{directory}': {e}\")\n","metadata":{"execution":{"iopub.status.busy":"2023-12-07T06:22:13.025998Z","iopub.execute_input":"2023-12-07T06:22:13.026605Z","iopub.status.idle":"2023-12-07T06:22:13.042036Z","shell.execute_reply.started":"2023-12-07T06:22:13.026557Z","shell.execute_reply":"2023-12-07T06:22:13.040444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\npreviousbatch=350\nbatch_size = 10\ndf=full_df[previousbatch:400]\ngrouped = df.groupby(df.index // batch_size)\nfor idx,(group_id, group_data) in enumerate(grouped):\n    for idx,row in group_data.iterrows():\n        path=row['Image_path']\n        label=row['label']\n        if row[\"is_tma\"]==True:\n            print(\"tma\")\n            continue\n            #folder_tiles = extract_prune_tiles(path,label, IMAGES_FOLDER, size=2048, scale=1,drop_thr=0.4,white_thr=222,max_samples=100)\n        else :\n            print(\"nontma\")\n            folder_tiles = extract_prune_tiles(path,label, IMAGES_FOLDER, size=1024, scale=1,drop_thr=0.4,white_thr=222,max_samples=10000)\n\n    record(idx+previousbatch)\n# print(f\"found tiles: {len(dataset)}\")\n\n# # quick view\n# fig, axes = plt.subplots(nrows=3, ncols=3, figsize=(10, 10))\n# for i in range(9):\n#     img = dataset[i]\n#     axes[i // 3, i % 3].imshow(img)\n# fig.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2023-12-07T08:18:25.165594Z","iopub.execute_input":"2023-12-07T08:18:25.166076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# record(idx+previousbatch)","metadata":{"execution":{"iopub.status.busy":"2023-12-07T06:18:34.209957Z","iopub.execute_input":"2023-12-07T06:18:34.210733Z","iopub.status.idle":"2023-12-07T06:18:34.244885Z","shell.execute_reply.started":"2023-12-07T06:18:34.210704Z","shell.execute_reply":"2023-12-07T06:18:34.243308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# dataset = TilesFolderDataset(\"/kaggle/working/test_tiles/1952\")\n# print(f\"found tiles: {len(dataset)}\")\n\n# # quick view\n# fig, axes = plt.subplots(nrows=10, ncols=10, figsize=(10, 10))\n# axes=axes.flatten()\n# for i in range(100):\n#     ax=axes[i]\n#     img = dataset[i+2000]\n#     ax.imshow(img)\n#     ax.axis('off')\n# fig.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2023-12-07T06:18:34.245811Z","iopub.status.idle":"2023-12-07T06:18:34.246676Z","shell.execute_reply.started":"2023-12-07T06:18:34.246255Z","shell.execute_reply":"2023-12-07T06:18:34.246301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# import os\n# import glob\n# import random\n# import re\n# import io\n# from PIL import Image\n# import tensorflow as tf\n\n# def record():\n#     folderpath = '/kaggle/working/test_tiles'\n#     folder = glob.glob(os.path.join(folderpath, \"*\"))\n#     image_list = []\n\n#     for path in folder:\n#         image_list = image_list + (glob.glob(os.path.join(path, \"*.png\")))\n#         print(len(image_list))\n\n#     random.shuffle(image_list)\n\n#     # Open the TFRecord file in append mode ('ab')\n#     with tf.io.TFRecordWriter(\"caltech_dataset.tfrecords\", options='ab') as writer:\n#         for imagepath in image_list:\n#             pattern = r'label-(\\d+)-\\d+-\\d+\\.png'\n#             match = re.search(pattern, imagepath)\n#             label = match.group(1)\n#             label = int(label)\n#             image = Image.open(imagepath)\n\n#             bytes_buffer = io.BytesIO()\n#             image.convert(\"RGB\").save(bytes_buffer, \"JPEG\")\n#             image_bytes = bytes_buffer.getvalue()\n\n#             bytes_feature = tf.train.Feature(bytes_list=tf.train.BytesList(value=[image_bytes]))\n#             class_feature = tf.train.Feature(int64_list=tf.train.Int64List(value=[label]))\n\n#             example = tf.train.Example(\n#                 features=tf.train.Features(feature={\n#                     \"image\": bytes_feature,\n#                     \"class\": class_feature\n#                 })\n#             )\n\n#             writer.write(example.SerializeToString())\n\n#             image.close()\n\n#     # Removing directories after processing\n#     for directory in folder:\n#         try:\n#             shutil.rmtree(directory)\n#             print(f\"Directory '{directory}' removed successfully.\")\n#         except OSError as e:\n#             print(f\"Error removing directory '{directory}': {e}\")\n","metadata":{"execution":{"iopub.status.busy":"2023-12-07T06:18:34.248572Z","iopub.status.idle":"2023-12-07T06:18:34.249023Z","shell.execute_reply.started":"2023-12-07T06:18:34.248828Z","shell.execute_reply":"2023-12-07T06:18:34.248847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## code to extract the images from tfr record\n","metadata":{}},{"cell_type":"code","source":"# image_feature_description = {\n#     \"image\": tf.io.FixedLenFeature([], tf.string), \n#     \"class\": tf.io.FixedLenFeature([], tf.int64), \n#     }\n# def _parse_data(unparsed_example):\n#     return tf.io.parse_single_example(unparsed_example, image_feature_description)\n\n# def _bytestring_to_pixels(parsed_example):\n#     byte_string = parsed_example['image']\n#     image = tf.io.decode_image(byte_string)\n# #     image = tf.reshape(image, [256, 256, 3])\n#     return image, parsed_example[\"class\"]\n# AUTOTUNE = tf.data.AUTOTUNE\n# def load_and_extract_images(filepath):\n#     dataset = tf.data.TFRecordDataset(filepath)\n#     dataset = dataset.map(_parse_data, num_parallel_calls=AUTOTUNE)\n#     dataset = dataset.map(_bytestring_to_pixels, num_parallel_calls=AUTOTUNE) # .cache()\n#     return dataset\n\n# caltech_dataset = load_and_extract_images(\"/kaggle/working/89xtwentyimages.tfrecords\")\n\n# train_dataset = caltech_dataset.take(90)\n\n# cropsize=256\n# def _train_data_preprocess_and_augment(image, label):\n# #     image = tf.cast(image, tf.float32)\n# #     image = tf.image.random_crop(image, size=[crop_size, crop_size, 3])\n    \n#     return image, label\n# train_preprocessed_augmented = train_dataset.map(_train_data_preprocess_and_augment)\n# fig,axes=plt.subplots(8,4,figsize=(10,10))\n# axes=axes.flatten()\n# for(images, label_batch) in train_preprocessed_augmented.batch(32):\n#     for i in range(32):\n#         ax=axes[i] \n#         ax.imshow(images[i])\n#         ax.set_title(f\"label-{label_batch[i]}\")\n\n#     plt.axis(\"off\")\n#     plt.show()\n#     break\n        \n    ","metadata":{"execution":{"iopub.status.busy":"2023-12-07T07:43:22.756359Z","iopub.execute_input":"2023-12-07T07:43:22.756836Z","iopub.status.idle":"2023-12-07T07:43:31.686397Z","shell.execute_reply.started":"2023-12-07T07:43:22.756803Z","shell.execute_reply":"2023-12-07T07:43:31.684787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# image=plt.imread(\"/kaggle/working/test_tiles/66/label-3-15-9.png\")\n# plt.imshow(image)","metadata":{"execution":{"iopub.status.busy":"2023-12-07T07:19:20.276963Z","iopub.execute_input":"2023-12-07T07:19:20.277573Z","iopub.status.idle":"2023-12-07T07:19:21.046699Z","shell.execute_reply.started":"2023-12-07T07:19:20.277535Z","shell.execute_reply":"2023-12-07T07:19:21.045422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}