{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport cv2\nimport tifffile as tifi\nimport openslide\n# from PIL import Image\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# img = tifi.imread(\"../input/mayo-clinic-strip-ai/test/008e5c_0.tif\")\n# plt.imshow(cv2.flip(img, 0))\n# plt.show()\n\n# slide = openslide.open_slide(\"../input/mayo-clinic-strip-ai/test/008e5c_0.tif\")\n\nfrom openslide.deepzoom import DeepZoomGenerator\n\n# tiles = DeepZoomGenerator(slide, tile_size=512, overlap=0, limit_bounds=False)\n# print(\"Number of tiles:\", tiles.tile_count)\n# print(\"Number of tile levels:\", tiles.level_count)\n# print(\"Tiles levels:\", tiles.level_tiles)\n# print(\"Tiles level dimensions:\", tiles.level_dimensions)\n# NUM_LEVELS = 13\n# # print(tiles.get_tile_dimensions(NUM_LEVELS, (5,28)))\n# for i in range(tiles.level_dimensions[NUM_LEVELS][0]//512 + 1):\n#     for j in range(tiles.level_dimensions[NUM_LEVELS][1]//512 + 1):\n#         tile = tiles.get_tile(NUM_LEVELS, (i, j))\n#         print(\"Tile mean value:\", np.array(tile).mean())\n#         print(\"Tile standard deviation:\", np.array(tile).std())\n#         plt.imshow(tile)\n#         plt.show()\n# # plt.imshow(tiles.get_tile(14, (0,0)))\n# # plt.show()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-10-05T22:00:23.094028Z","iopub.execute_input":"2022-10-05T22:00:23.095152Z","iopub.status.idle":"2022-10-05T22:00:23.556954Z","shell.execute_reply.started":"2022-10-05T22:00:23.095007Z","shell.execute_reply":"2022-10-05T22:00:23.555833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pip install pqdm","metadata":{"execution":{"iopub.status.busy":"2022-10-05T22:00:23.55878Z","iopub.execute_input":"2022-10-05T22:00:23.559119Z","iopub.status.idle":"2022-10-05T22:00:23.56375Z","shell.execute_reply.started":"2022-10-05T22:00:23.559089Z","shell.execute_reply":"2022-10-05T22:00:23.562899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from pqdm.processes import pqdm\nfrom tqdm import tqdm\n\ntrain_csv_path = \"../input/mayo-clinic-strip-ai/train.csv\"\ntrain_images_path = \"../input/mayo-clinic-strip-ai/train/\"\n\ntest_csv_path = \"../input/mayo-clinic-strip-ai/test.csv\"\ntest_images_path = \"../input/mayo-clinic-strip-ai/test/\"\n\nIMG_SIZE = 300","metadata":{"execution":{"iopub.status.busy":"2022-10-05T22:00:23.565103Z","iopub.execute_input":"2022-10-05T22:00:23.565692Z","iopub.status.idle":"2022-10-05T22:00:23.57805Z","shell.execute_reply.started":"2022-10-05T22:00:23.565655Z","shell.execute_reply":"2022-10-05T22:00:23.576799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv = pd.read_csv(train_csv_path)\ntrain_csv.info()\n\ntest_csv = pd.read_csv(test_csv_path)","metadata":{"execution":{"iopub.status.busy":"2022-10-05T22:00:23.579384Z","iopub.execute_input":"2022-10-05T22:00:23.579761Z","iopub.status.idle":"2022-10-05T22:00:23.63665Z","shell.execute_reply.started":"2022-10-05T22:00:23.579727Z","shell.execute_reply":"2022-10-05T22:00:23.635395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv[\"label\"].value_counts().plot(kind=\"bar\")","metadata":{"execution":{"iopub.status.busy":"2022-10-05T22:00:23.639952Z","iopub.execute_input":"2022-10-05T22:00:23.640343Z","iopub.status.idle":"2022-10-05T22:00:23.855645Z","shell.execute_reply.started":"2022-10-05T22:00:23.640306Z","shell.execute_reply":"2022-10-05T22:00:23.854399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The dataset is unbalanced","metadata":{}},{"cell_type":"code","source":"pip install imagesize","metadata":{"execution":{"iopub.status.busy":"2022-10-05T22:00:23.856932Z","iopub.execute_input":"2022-10-05T22:00:23.857303Z","iopub.status.idle":"2022-10-05T22:00:38.198877Z","shell.execute_reply.started":"2022-10-05T22:00:23.857269Z","shell.execute_reply":"2022-10-05T22:00:38.19746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install tabulate","metadata":{"execution":{"iopub.status.busy":"2022-10-05T22:00:38.200704Z","iopub.execute_input":"2022-10-05T22:00:38.20111Z","iopub.status.idle":"2022-10-05T22:00:49.546923Z","shell.execute_reply.started":"2022-10-05T22:00:38.201063Z","shell.execute_reply":"2022-10-05T22:00:49.545659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import imagesize\nfrom tabulate import tabulate\n\nmin_w, min_h = imagesize.get(\"../input/mayo-clinic-strip-ai/train/006388_0.tif\")\nmax_w, max_h = min_w, min_h\nmin_size = max_size = min_w * min_h\nnames = [\"\" for i in range(6)]\n\nfor image_id in tqdm(train_csv[\"image_id\"][1:]):\n    w, h = imagesize.get(train_images_path + image_id + \".tif\")\n    if w < min_w:\n        min_w = w\n        names[0] = image_id\n    if w > max_w:\n        max_w = w\n        names[1] = image_id\n    if h < min_h:\n        min_h = h\n        names[2] = image_id\n    if h > max_h:\n        max_h = h\n        names[3] = image_id\n    if w * h < min_size:\n        min_size = w * h\n        names[4] = image_id\n    if w * h > max_size:\n        max_size = w * h\n        names[5] = image_id\n\ndata = [[\"size\", min_size, max_size], [\"width\", min_w, max_w], [\"hight\", min_h, max_h]]\nprint(tabulate(data, headers=[\"min\", \"max\"]))\nprint(\"Smallest image id:\", names[4])\nprint(\"Largest image id:\", names[5])","metadata":{"execution":{"iopub.status.busy":"2022-10-05T22:00:49.548231Z","iopub.execute_input":"2022-10-05T22:00:49.548631Z","iopub.status.idle":"2022-10-05T22:01:07.914561Z","shell.execute_reply.started":"2022-10-05T22:00:49.548585Z","shell.execute_reply":"2022-10-05T22:01:07.913378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf","metadata":{"execution":{"iopub.status.busy":"2022-10-05T22:01:07.916334Z","iopub.execute_input":"2022-10-05T22:01:07.91708Z","iopub.status.idle":"2022-10-05T22:01:13.618127Z","shell.execute_reply.started":"2022-10-05T22:01:07.917033Z","shell.execute_reply":"2022-10-05T22:01:13.616975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Create tfrecords","metadata":{}},{"cell_type":"code","source":"def _bytes_feature(value):\n    \"\"\"Returns a bytes_list from a string / byte.\"\"\"\n    if isinstance(value, type(tf.constant(0))): # if value ist tensor\n        value = value.numpy() # get value of tensor\n    return tf.train.Feature(bytes_list=tf.train.BytesList(value=[value]))\n\ndef _float_feature(value):\n  \"\"\"Returns a floast_list from a float / double.\"\"\"\n  return tf.train.Feature(float_list=tf.train.FloatList(value=[value]))\n\ndef _int64_feature(value):\n    \"\"\"Returns an int64_list from a bool / enum / int / uint.\"\"\"\n    return tf.train.Feature(int64_list=tf.train.Int64List(value=[value]))\n\ndef serialize_array(array):\n    array = tf.io.serialize_tensor(array)\n    return array\n\ndef parse_single_image(image, label):\n    #define the dictionary -- the structure -- of our single example\n    data = {\n        'raw_image' : _bytes_feature(serialize_array(image)),\n        'label' : _int64_feature(label)\n    }\n    #create an Example, wrapping the single features\n    out = tf.train.Example(features=tf.train.Features(feature=data))\n    return out\n\ndef parse_tfr_element(element):\n    #use the same structure as above; it's kinda an outline of the structure we now want to create\n    data = {\n      'label':tf.io.FixedLenFeature([], tf.int64),\n      'raw_image' : tf.io.FixedLenFeature([], tf.string)\n    }\n    content = tf.io.parse_single_example(element, data)\n    \n    label = content['label']\n    raw_image = content['raw_image']\n    \n    #get our 'feature'-- our image -- and reshape it appropriately\n    feature = tf.io.parse_tensor(raw_image, out_type=tf.uint8)\n    feature = tf.reshape(feature, shape=[IMG_SIZE, IMG_SIZE,3])\n    return (feature, label)\n\ndef write_images_to_tfr(images, labels, image_id:str, filename:str=\"train_records\", max_files:int=10, start:int=0):\n    #determine the number of shards (single TFRecord files) we need:\n    splits = (len(images)//max_files) + 1 #determine how many tfr shards are needed\n    if len(images) % max_files == 0:\n        splits -= 1\n        print(f\"\\nUsing {splits} shard(s) for {len(images)} files, with up to {max_files} samples per shard\")\n\n    file_count = 0\n    for i in tqdm.tqdm(range(splits)):\n        current_shard_name = \"{}_{}{}.tfrecords\".format(image_id, start+i+1, filename)\n        writer = tf.io.TFRecordWriter(current_shard_name)\n\n        current_shard_count = 0\n        while current_shard_count < max_files: #as long as our shard is not full\n            #get the index of the file that we want to parse now\n            index = i*max_files+current_shard_count\n            if index == len(images): #when we have consumed the whole data, preempt generation\n                break\n            current_image = images[index]\n            current_label = labels[index]\n\n            #create the required Example representation\n            out = parse_single_image(image=current_image, label=current_label)\n            writer.write(out.SerializeToString())\n            current_shard_count+=1\n            file_count += 1\n\n    writer.close()\n    print(f\"\\nWrote {file_count} elements to TFRecord\")\n    return file_count","metadata":{"execution":{"iopub.status.busy":"2022-10-05T22:01:13.619511Z","iopub.execute_input":"2022-10-05T22:01:13.620258Z","iopub.status.idle":"2022-10-05T22:01:13.641328Z","shell.execute_reply.started":"2022-10-05T22:01:13.620217Z","shell.execute_reply":"2022-10-05T22:01:13.639772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from PIL import Image\n\ntry:\n    os.makedirs(\"./white\")\n    os.makedirs(\"./usable\")\nexcept:\n    pass\nwhite_images_count = 0\nuseful_images_count = 0\n# for image_id in tqdm(test_csv[\"image_id\"]):\nfor i in range(len(test_csv[\"image_id\"])):\n    image_id = test_csv[\"image_id\"][i]\n    patient_id = test_csv[\"patient_id\"][i]\n    print(patient_id)\n    slide = openslide.open_slide(test_images_path + image_id + \".tif\")\n    tiles = DeepZoomGenerator(slide, tile_size=IMG_SIZE, overlap=0, limit_bounds=False)\n    NUM_LEVELS = tiles.level_count - 2\n    useful_images = []\n    try:\n        os.makedirs(\"./white/\" + patient_id + \"/\" + patient_id)\n        os.makedirs(\"./usable/\" + patient_id + \"/\" + patient_id)\n    except:\n        pass\n    for i in range(tiles.level_dimensions[NUM_LEVELS][0]//IMG_SIZE + 1):\n        for j in range(tiles.level_dimensions[NUM_LEVELS][1]//IMG_SIZE + 1):\n            tile = tiles.get_tile(NUM_LEVELS, (i, j))\n            ##seperate white tiles\n            if np.array(tile).mean() > 230 and np.array(tile).std() < 35:\n#                 tile.save(\"./white/\" + patient_id + \"/\" + patient_id + \"/\" + image_id + \"_\" + str(i) + \",\" + str(j) + \".jpg\")\n                white_images_count += 1\n            else:\n                useful_images_count += 1\n                useful_images.append(np.array(tile))\n                tile.save(\"./usable/\" + patient_id + \"/\" + patient_id + \"/\" + image_id + \"_\" + str(i) + \",\" + str(j) + \".jpg\")","metadata":{"execution":{"iopub.status.busy":"2022-10-05T22:01:13.644694Z","iopub.execute_input":"2022-10-05T22:01:13.64525Z","iopub.status.idle":"2022-10-05T22:11:33.060881Z","shell.execute_reply.started":"2022-10-05T22:01:13.64519Z","shell.execute_reply":"2022-10-05T22:11:33.057549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##print(len(os.listdir(\"./usable\")))\n##print(len(os.listdir(\"./white\")))","metadata":{"execution":{"iopub.status.busy":"2022-10-05T22:11:33.067155Z","iopub.execute_input":"2022-10-05T22:11:33.067882Z","iopub.status.idle":"2022-10-05T22:11:33.078626Z","shell.execute_reply.started":"2022-10-05T22:11:33.067797Z","shell.execute_reply":"2022-10-05T22:11:33.077191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Decode images from tfrecords","metadata":{}},{"cell_type":"code","source":"def decode_image(image):\n    image = tf.image.decode_jpeg(image, channels=3)\n    image = tf.cast(image, tf.float32)\n    image = tf.reshape(image, [*IMAGE_SIZE, 3])\n    return image\ndef read_tfrecord(example, labeled):\n    tfrecord_format = (\n        {\n          'label':tf.io.FixedLenFeature([], tf.int64),\n          'raw_image' : tf.io.FixedLenFeature([], tf.string)\n        }\n        if labeled\n        else {\"raw_image\": tf.io.FixedLenFeature([], tf.string),}\n    )\n    example = tf.io.parse_single_example(example, tfrecord_format)\n    image = decode_image(example[\"raw_image\"])\n    if labeled:\n        label = tf.cast(example[\"label\"], tf.int32)\n        return image, label\n    return image\ndef load_dataset(filenames, labeled=True):\n    ignore_order = tf.data.Options()\n    ignore_order.experimental_deterministic = False  # disable order, increase speed\n    dataset = tf.data.TFRecordDataset(\n        filenames\n    )  # automatically interleaves reads from multiple files\n    dataset = dataset.with_options(\n        ignore_order\n    )  # uses data as soon as it streams in, rather than in its original order\n    dataset = dataset.map(\n        parse_tfr_element, num_parallel_calls=AUTOTUNE\n    )\n    # returns a dataset of (image, label) pairs if labeled=True or just images if labeled=False\n    return dataset\ndef get_dataset(filenames, labeled=True):\n    dataset = load_dataset(filenames, labeled=labeled)\n    dataset = dataset.shuffle(2048)\n    dataset = dataset.prefetch(buffer_size=AUTOTUNE)\n    dataset = dataset.batch(BATCH_SIZE)\n    return dataset","metadata":{"execution":{"iopub.status.busy":"2022-10-05T22:11:33.080666Z","iopub.execute_input":"2022-10-05T22:11:33.081054Z","iopub.status.idle":"2022-10-05T22:11:33.097743Z","shell.execute_reply.started":"2022-10-05T22:11:33.081018Z","shell.execute_reply":"2022-10-05T22:11:33.096061Z"},"trusted":true},"execution_count":null,"outputs":[]}]}