{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"pip install --upgrade efficientnet-pytorch","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport sys\nsys.path = [\n    '../input/efficientnet-pytorch/EfficientNet-PyTorch/EfficientNet-PyTorch-master',\n] + sys.path\nimport numpy as np\nimport pandas as pd\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data import DataLoader, Dataset\nfrom efficientnet_pytorch import model as enet\n\nfrom tqdm import tqdm_notebook as tqdm\nimport skimage.io","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_dir = '../input/prostate-cancer-grade-assessment'\ndf_train = pd.read_csv(os.path.join(data_dir, 'train.csv'))\ndf_test = pd.read_csv(os.path.join(data_dir, 'test.csv'))\ndf_sub = pd.read_csv(os.path.join(data_dir, 'sample_submission.csv'))\n\nmodel_dir = '../input/panda-base-model'\nimage_folder = os.path.join(data_dir, 'test_images')\nis_test = os.path.exists(image_folder) \nimage_folder = image_folder if is_test else os.path.join(data_dir, 'train_images')\n\ndf = df_test if is_test else df_train.loc[:3]\nprint(df.shape)\n\ntile_size = 256\nimage_size = 256\nn_tiles = 36\nbatch_size = 4\nnum_workers = 4\n\ndevice = torch.device('cuda')\n\nprint(image_folder)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class enetv2(nn.Module):\n    def __init__(self, backbone, out_dim):\n        super(enetv2, self).__init__()\n        self.enet = enet.EfficientNet.from_name(backbone)\n        self.myfc = nn.Linear(self.enet._fc.in_features, out_dim)\n        self.enet._fc = nn.Identity()\n\n    def extract(self, x):\n        return self.enet(x)\n\n    def forward(self, x):\n        x = self.extract(x)\n        x = self.myfc(x)\n        return x\n    \n    \ndef load_models(model_files):\n    models = []\n    for model_dict in model_files:\n        model_f = model_dict['name']\n        model_f = os.path.join(model_dir, model_f)\n        backbone = 'efficientnet-b0'\n        model = enetv2(backbone, out_dim=model_dict['out_dim'])\n        if model_dict['parallel']:\n            model = nn.DataParallel(model)\n        model.load_state_dict(torch.load(model_f, map_location=lambda storage, loc: storage), strict=True)\n        model.eval()\n        model.to(device)\n        models.append(model)\n        print(f'{model_f} loaded!')\n    return models\n\n\nmodel_files = [\n    {\n        'name': 'model_2_bce_0.881.pth',\n#         Trained on several GPUs in parallel\n        'parallel': True,\n#         Misconfigured while training, it should have predicted 5 bins\n        'out_dim': 6,\n        'n_tiles': 36\n    },\n    {\n        'name': 'model_0_bce_4x4_0.884.pth',\n        'parallel': False,\n        'out_dim': 5,\n        'n_tiles': 16\n    },\n    {\n        'name': 'model_2_bce_5x5_total_0.883.pth',\n        'parallel': False,\n        'out_dim': 5,\n        'n_tiles': 25\n    },\n    {\n        'name': 'model_3_bce_6x6_total_0.883.pth',\n        'parallel': False,\n        'out_dim': 5,\n        'n_tiles': 36\n    }\n]\n\nmodels = load_models(model_files)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_tiles(img, n_tiles, mode=0):\n    result = []\n    h, w, c = img.shape\n    pad_h = (tile_size - h % tile_size) % tile_size\n    pad_w = (tile_size - w % tile_size) % tile_size\n    mode_pad_h = (tile_size * mode) // 4\n    mode_pad_w = (tile_size * mode) // 4\n    img2 = np.pad(img,[[pad_h // 2 + mode_pad_h, pad_h - pad_h // 2 + tile_size - mode_pad_h], \n                       [pad_w // 2 + mode_pad_w, pad_w - pad_w // 2 + tile_size - mode_pad_w], [0,0]], constant_values=255)\n    img3 = img2.reshape(\n        img2.shape[0] // tile_size,\n        tile_size,\n        img2.shape[1] // tile_size,\n        tile_size,\n        3\n    )\n\n    img3 = img3.transpose(0,2,1,3,4).reshape(-1, tile_size, tile_size,3)\n    n_tiles_with_info = (img3.reshape(img3.shape[0],-1).sum(1) < tile_size ** 2 * 3 * 255).sum()\n    if len(img3) < n_tiles:\n        img3 = np.pad(img3, [[0, n_tiles - len(img3)], [0, 0] , [0, 0] , [0, 0]], constant_values=255)\n    idxs = np.argsort(img3.reshape(img3.shape[0],-1).sum(-1))[:n_tiles]    \n    img3 = img3[idxs]\n    for i in range(len(img3)):\n        result.append({'img':img3[i], 'idx':i})\n    return result, n_tiles_with_info >= n_tiles\n\n\nclass PANDADataset(Dataset):\n    def __init__(self,\n                 df,\n                 image_size,\n                 n_tiles=n_tiles,\n                 tile_mode=0,\n                 rotate=False,\n                 rand=False,\n                 sub_imgs=False\n                ):\n\n        self.df = df.reset_index(drop=True)\n        self.image_size = image_size\n        self.n_tiles = n_tiles\n        self.tile_mode = tile_mode\n        self.rotate = rotate\n        self.rand = rand\n        self.sub_imgs = sub_imgs\n\n    def __len__(self):\n        return self.df.shape[0]\n\n    def __getitem__(self, index):\n        row = self.df.iloc[index]\n        img_id = row.image_id\n        \n        tiff_file = os.path.join(image_folder, f'{img_id}.tiff')\n        image = skimage.io.MultiImage(tiff_file)[1]\n        tiles, OK = get_tiles(image, n_tiles=self.n_tiles, mode=self.tile_mode)\n\n        if self.rand:\n            idxes = np.random.choice(list(range(self.n_tiles)), self.n_tiles, replace=False)\n        else:\n            idxes = list(range(self.n_tiles))\n        idxes = np.asarray(idxes) + self.n_tiles if self.sub_imgs else idxes\n\n        n_row_tiles = int(np.sqrt(self.n_tiles))\n        images = np.zeros((image_size * n_row_tiles, image_size * n_row_tiles, 3))\n        for h in range(n_row_tiles):\n            for w in range(n_row_tiles):\n                i = h * n_row_tiles + w\n    \n                if len(tiles) > idxes[i]:\n                    this_img = tiles[idxes[i]]['img']\n                else:\n                    this_img = np.ones((self.image_size, self.image_size, 3)).astype(np.uint8) * 255\n                this_img = 255 - this_img\n                h1 = h * image_size\n                w1 = w * image_size\n                \n                if self.rotate:\n                    # 90 degrees counter-clockwise rotation\n                    this_img = this_img.transpose(1, 0, 2)[::-1]\n                    \n                images[h1:h1+image_size, w1:w1+image_size] = this_img\n\n        images = images.astype(np.float32)\n        images /= 255\n        images = images.transpose(2, 0, 1)\n\n        return torch.tensor(images)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_loaders = []\n\nfor model_dict, model in zip(model_files, models):\n    loaders = []\n    for mode in [0, 1, 2, 3]:\n        for rotate in [False, True]:\n            dataset = PANDADataset(df, image_size, n_tiles=model_dict['n_tiles'], tile_mode=mode, rotate=rotate)\n            loader = DataLoader(dataset, batch_size=batch_size, num_workers=num_workers, shuffle=False)\n            loaders.append(loader)\n    \n    model_loaders.append((model, loaders))\nprint(f'{len(model_loaders) * len(model_loaders[0][1])} predictions in total')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LOGITSS = []\nwith torch.no_grad():\n    for model, loaders in model_loaders:\n        for loader in loaders:\n            LOGITS = []\n            for data in tqdm(loader):\n                data = data.to(device)\n                logits = model(data)\n                LOGITS.append(logits)\n            LOGITS = torch.cat(LOGITS).sigmoid().cpu().numpy().sum(1)\n            LOGITSS.append(LOGITS)\n\nLOGITS = np.array(LOGITSS).mean(0)\nPREDS = LOGITS.round()\n\ndf['isup_grade'] = PREDS.astype(int)\ndf[['image_id', 'isup_grade']].to_csv('submission.csv', index=False)\nprint(df.head())\nprint()\nprint(df.isup_grade.value_counts())","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install warmup-scheduler","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport torch\nfrom torch.nn import functional\nimport cv2 as cv\n\n\ndef tile(img, sz=128, N=16, transform=None, random=False):\n    shape = img.shape\n    pad0, pad1 = (sz - shape[0] % sz) % sz, (sz - shape[1] % sz) % sz\n\n    img = np.pad(img, [[pad0 // 2, pad0 - pad0 // 2], [pad1 // 2, pad1 - pad1 // 2], [0, 0]],\n                 constant_values=255)\n\n    img = img.reshape(img.shape[0] // sz, sz, img.shape[1] // sz, sz, 3)\n    img = img.transpose(0, 2, 1, 3, 4).reshape(-1, sz, sz, 3)\n\n    if len(img) < N:\n        img = np.pad(img, [[0, N - len(img)], [0, 0], [0, 0], [0, 0]], constant_values=255)\n\n    idxs = np.argsort(img.reshape(img.shape[0], -1).sum(-1))[:N]\n\n    img = img[idxs]\n    s = int(np.sqrt(N))\n    result = np.full([sz * s, sz * s, 3], 255, dtype=np.uint8)\n\n    indexes = np.arange(N)\n    if random:\n        np.random.shuffle(indexes)\n\n    for i, j in enumerate(indexes):\n        if i >= len(img):\n            break\n        x = j % s\n        y = j // s\n        img_i = img[i] if transform is None else transform(image=img[i])[\"image\"]\n        if result.dtype != img_i.dtype:\n            result = result.astype(img_i.dtype)\n        result[y * sz:(y + 1) * sz, x * sz:(x + 1) * sz] = img_i\n\n    return result\n\n\ndef __rgb_to_hsv(image):\n    if len(image.shape) < 3 or image.shape[-3] != 3:\n        raise ValueError(\"Input size must have a shape of (3, H, W). Got {}\".format(image.shape))\n\n    r, g, b = image\n\n    maxc = image.max(-3)[0]\n    minc = image.min(-3)[0]\n\n    v = maxc  # brightness\n\n    deltac: torch.Tensor = maxc - minc\n\n    s = deltac / v\n    s = torch.where(torch.isnan(s), torch.zeros_like(s), s)\n\n    # avoid division by zero\n    deltac = torch.where(deltac == 0, torch.ones_like(deltac), deltac)\n\n    maxg = g == maxc\n    maxr = r == maxc\n\n    r, g, b = (torch.stack([maxc] * 3) - image) * (1 / deltac)\n\n    h = 4.0 + g - r\n    h = torch.where(maxg, 2.0 + r - b, h)\n    h = torch.where(maxr, b - g, h)\n    h = torch.where(minc == maxc, torch.zeros_like(h), h)\n\n    h *= 60.0\n    h %= 360.0\n\n    return torch.stack([h, s, v], dim=-3)\n\n\ndef get_tile(img, boxes, sz=256, num=36):\n    shape = img.shape\n    pad0, pad1 = (sz - shape[0] % sz) % sz, (sz - shape[1] % sz) % sz\n\n    img = np.pad(img, [[pad0 // 2, pad0 - pad0 // 2], [pad1 // 2, pad1 - pad1 // 2], [0, 0]], constant_values=255)\n\n    s = int(np.sqrt(num))\n    tile = np.full([sz * s, sz * s, 3], 255, dtype=img.dtype)\n    for i, box in enumerate(boxes):\n        x = i % s\n        y = i // s\n\n        try:\n            tile[y * sz:(y + 1) * sz, x * sz:(x + 1) * sz] = img[box[1]:box[3], box[0]:box[2]]\n        except Exception:\n            pass\n    return tile\n\n\ndef tile_hsv_boxes(img, sz=256, N=36, default_val=255):\n    shape = img.shape\n    pad0, pad1 = (sz - shape[0] % sz) % sz, (sz - shape[1] % sz) % sz\n    num_channels = 1\n\n    img = cv.cvtColor(img, cv.COLOR_RGB2HSV)[..., 1]\n    img = 255 - torch.from_numpy(img).cuda()\n    img = torch.unsqueeze(img, -1)\n\n    img = functional.pad(img, [0, 0, pad1 // 2, pad1 - pad1 // 2, pad0 // 2, pad0 - pad0 // 2], value=default_val)\n\n    shape = img.shape\n\n    img = img.reshape(img.shape[0] // sz, sz, img.shape[1] // sz, sz, num_channels)\n    img = img.permute([0, 2, 1, 3, 4]).reshape(-1, sz, sz, num_channels)\n\n    num = len(img)\n\n    if len(img) < N:\n        img = functional.pad(img, [0, 0, 0, 0, 0, 0, 0, N - len(img)], value=default_val)\n\n    idxs = torch.argsort(img.reshape(img.shape[0], -1).float().sum(-1))[:N]\n\n    s = int(np.sqrt(N))\n    res_boxes = []\n    idxs = idxs.detach().cpu().numpy()\n\n    indexes = np.arange(N)\n\n    for i, j in enumerate(indexes):\n        if i >= num:\n            break\n\n        x = idxs[i] % (shape[1] // sz)\n        y = idxs[i] // (shape[1] // sz)\n        x1 = x * sz\n        x2 = (x + 1) * sz\n        y1 = y * sz\n        y2 = (y + 1) * sz\n        res_boxes.append([int(x1), int(y1), int(x2), int(y2)])\n\n    return res_boxes\n\n\ndef _find_bounding_boxes(image, preprocessed=False):\n    if not preprocessed:\n        if image.ndim > 2:\n            image = cv.cvtColor(image, cv.COLOR_RGB2GRAY)\n        image = (image != 255).astype(np.uint8)\n        image = cv.morphologyEx(image, cv.MORPH_CLOSE, cv.getStructuringElement(cv.MORPH_RECT, (10, 10)))\n    countours, hierarchy = cv.findContours(image, cv.RETR_LIST, cv.CHAIN_APPROX_SIMPLE)\n\n    min_rects = []\n    for c in countours:\n        rect = cv.boundingRect(c)\n        min_rects.append(rect)\n\n    return [[x, y, x + w, y + h] for x, y, w, h in min_rects]\n\n\ndef moving_sum(a, n=3) :\n    ret = np.cumsum(a)\n    ret[n:] = ret[n:] - ret[:-n]\n    return ret[n - 1:]\n\n\ndef _find_new_tile_boxes(img, sz=64, threshold=0, max_overlap=0.0):\n    if img.ndim == 3 and img.shape[-1] == 3:\n        img = cv.cvtColor(img, cv.COLOR_RGB2GRAY)\n    img = 255 - img\n\n    shape = img.shape\n    pad0, pad1 = (sz - shape[0] % sz) % sz, (sz - shape[1] % sz) % sz\n    img = cv.copyMakeBorder(img,\n                            top=0, left=sz,\n                            bottom=pad0 + sz, right=pad1 + sz,\n                            borderType=cv.BORDER_CONSTANT, value=0)\n\n    sums = []\n    bboxes = []\n    for i in range(0, img.shape[0], sz):\n        y1 = i\n        y2 = i + sz\n\n        sub_img = img[y1:y2]\n        m_sum = moving_sum(sub_img.sum(axis=0), sz)\n\n        indexes = np.argsort(m_sum)\n\n        cond = m_sum[indexes] > threshold\n        indexes = indexes[cond]\n\n        while len(indexes):\n            x1 = indexes[-1]\n            x2 = x1 + sz\n            cond1 = indexes > (x1 - int(sz * (1 - max_overlap)))\n            cond2 = indexes <= (x2 - int(sz * max_overlap))\n            indexes = indexes[~(cond1 & cond2)]\n\n            sums.append(int(m_sum[x1]))\n\n            x1 -= sz\n            x2 = x2 - sz\n            bboxes.append([x1, y1, x2, y2])\n    return np.array(bboxes), np.array(sums)\n\n\ndef get_new_tile_boxes(img, size):\n    h, w = img.shape[:2]\n\n    boxes = _find_bounding_boxes(img)\n\n    tile_boxes = []\n    tile_sums = []\n    for i, box in enumerate(boxes):\n        sub_img = img[box[1]:box[3], box[0]:box[2]]\n\n        b, s = _find_new_tile_boxes(sub_img, size, max_overlap=0.0)\n        b = b.astype(np.float32)\n        if not len(b):\n            continue\n\n        try:\n            b[:, 0::2] = (b[:, 0::2] + box[0]) / w\n            b[:, 1::2] = (b[:, 1::2] + box[1]) / h\n        except Exception as err:\n            print(b.shape, box)\n            raise err\n        tile_boxes.append(b)\n        tile_sums.append(s)\n\n    if len(tile_sums):\n        tile_sums = np.concatenate(tile_sums)\n        tile_boxes = np.concatenate(tile_boxes)\n    else:\n        tile_sums = np.empty([0])\n        tile_boxes = np.empty([0, 4])\n\n    return tile_boxes, tile_sums\n\n\ndef get_tiles_new(img, boxes, sums, size=64, num=36, pad_value=(255, 255, 255), random=False):\n    h, w = img.shape[:2]\n\n    s = int(np.sqrt(num))\n    result = np.full([s * size, s * size, 3], pad_value, dtype=img.dtype)\n\n    boxes = boxes[np.argsort(sums)[::-1]][:num]\n    if random:\n        np.random.shuffle(boxes)\n    for i, box in enumerate(boxes):\n        x = i % s\n        y = i // s\n\n        box = box.copy()\n        box[0::2] *= w\n        box[1::2] *= h\n        x1, y1, x2, y2 = box.astype(int)\n\n        tile = img[y1:y2, x1:x2]\n        th, tw = tile.shape[:2]\n        if th > size or tw > size:\n            tile = tile[:size, :size]\n        if th < size or tw < size:\n            th, tw = tile.shape[:2]\n            pad_h = size - th\n            pad_w = size - tw\n            tile = cv.copyMakeBorder(tile, top=0, left=0, bottom=pad_h, right=pad_w,\n                                     borderType=cv.BORDER_CONSTANT, value=pad_value)\n\n        result[y * size:(y + 1) * size, x * size:(x + 1) * size] = tile\n\n    return result\n\n\ndef iafoss_tile_boxes_on_original_image(img, sz=256, N=100):\n    shape = img.shape\n    pad0, pad1 = (sz - shape[0] % sz) % sz, (sz - shape[1] % sz) % sz\n\n    img = np.pad(img, [[pad0 // 2, pad0 - pad0 // 2], [pad1 // 2, pad1 - pad1 // 2], [0, 0]],\n                 constant_values=255)\n\n    shape = img.shape\n\n    img = img.reshape(img.shape[0] // sz, sz, img.shape[1] // sz, sz, 3)\n    img = img.transpose(0, 2, 1, 3, 4).reshape(-1, sz, sz, 3)\n\n    num = len(img)\n\n    if len(img) < N:\n        img = np.pad(img, [[0, N - len(img)], [0, 0], [0, 0], [0, 0]], constant_values=255)\n\n    idxs = np.argsort(img.reshape(img.shape[0], -1).sum(-1))[:N]\n\n    img = img[idxs]\n    s = int(np.sqrt(N))\n\n    indexes = np.arange(N)\n\n    boxes = []\n    for i, j in enumerate(indexes):\n        if i >= len(img):\n            break\n        if i >= num:\n            break\n\n        x = idxs[i] % (shape[1] // sz)\n        y = idxs[i] // (shape[1] // sz)\n        x1 = x * sz\n        x2 = (x + 1) * sz\n        y1 = y * sz\n        y2 = (y + 1) * sz\n        x1 -= pad1\n        x2 -= pad1\n        y1 -= pad0\n        y2 -= pad0\n        boxes.append([x1, y1, x2, y2])\n\n    return np.array(boxes)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DEBUG = False\n\nimport os\nimport time\nimport skimage.io\nimport numpy as np\nimport pandas as pd\nimport cv2\nimport PIL.Image\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torch.optim as optim\nfrom torch.optim import lr_scheduler\nfrom torch.utils.data import DataLoader, Dataset\nfrom torch.utils.data.sampler import SubsetRandomSampler, RandomSampler, SequentialSampler\nfrom warmup_scheduler import GradualWarmupScheduler\nfrom efficientnet_pytorch import model as enet\nimport albumentations\nfrom sklearn.model_selection import StratifiedKFold\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import cohen_kappa_score\nfrom tqdm import tqdm_notebook as tqdm\nfrom pytorch_lightning.core.lightning import LightningModule\nimport random\nfrom pytorch_lightning import Trainer\nfrom pytorch_lightning.callbacks import ModelCheckpoint\nfrom torchvision.models import resnet101\n\nimport sys\nsys.path.append(\"../\")\n# from utils.data_utils import get_tiles_new\nimport json\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install utils","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"resnet101(True)\n\nSAVE_NAME = \"effnetb4_256_36_new\"\nFP16 = True\nbatch_size = 2\nnum_workers = 8\nACCUM_STEPS = 1\n\ndata_dir = '../input/prostate-cancer-grade-assessment'\ndf_train = pd.read_csv(os.path.join(data_dir, 'train.csv'))\nimage_folder = os.path.join(data_dir, 'train_images')\n\nwith open(\"../notebooks/new_boxes_256_36.json\", \"r\") as file:\n    boxes_info = json.load(file)\n\nkernel_type = SAVE_NAME\n\nenet_type = 'efficientnet-b4'\nfold = 0\ntile_size = 256\nimage_size = 256\nn_tiles = 36\nout_dim = 5\ninit_lr = 3e-4\nwarmup_factor = 10\n\nwarmup_epo = 1\nn_epochs = 1 if DEBUG else 30\ndf_train = df_train.sample(100).reset_index(drop=True) if DEBUG else df_train\n\ndevice = torch.device('cuda')\n\ntransforms_train = albumentations.Compose([\n    albumentations.Transpose(p=0.5),\n    albumentations.VerticalFlip(p=0.5),\n    albumentations.HorizontalFlip(p=0.5),\n])\ntransforms_val = albumentations.Compose([])\n\nprint(image_folder)\n\nskf = StratifiedKFold(5, shuffle=True, random_state=42)\ndf_train['fold'] = -1\nfor i, (train_idx, valid_idx) in enumerate(skf.split(df_train, df_train['isup_grade'])):\n    df_train.loc[valid_idx, 'fold'] = i\n\n\ndef seed_everything(seed):\n    random.seed(seed)\n    os.environ[\"PYTHONHASHSEED\"] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = True\n\n\nseed_everything(0)\n\n\nclass PANDADataset(Dataset):\n    def __init__(self,\n                 df,\n                 image_size,\n                 n_tiles=n_tiles,\n                 tile_mode=0,\n                 rand=False,\n                 transform=None,\n                 ):\n\n        self.df = df.reset_index(drop=True)\n        self.image_size = image_size\n        self.n_tiles = n_tiles\n        self.tile_mode = tile_mode\n        self.rand = rand\n        self.transform = transform\n\n    def __len__(self):\n        return self.df.shape[0]\n\n    def __getitem__(self, index):\n        row = self.df.iloc[index]\n        img_id = row.image_id\n\n        tiff_file = os.path.join(image_folder, f'{img_id}.tiff')\n        image = skimage.io.MultiImage(tiff_file)[1]\n\n        data = boxes_info[img_id]\n\n        tiles = get_tiles_new(image, np.array(data['boxes']), np.array(data['sums']),\n                              self.image_size, self.n_tiles, random=self.rand)\n\n        n_row_tiles = int(np.sqrt(self.n_tiles))\n        images = np.zeros((image_size * n_row_tiles, image_size * n_row_tiles, 3))\n        for h in range(n_row_tiles):\n            for w in range(n_row_tiles):\n                h1 = h * image_size\n                w1 = w * image_size\n                this_img = tiles[h1:h1 + image_size, w1:w1 + image_size]\n\n                this_img = 255 - this_img\n                if self.transform is not None:\n                    this_img = self.transform(image=this_img)['image']\n                images[h1:h1 + image_size, w1:w1 + image_size] = this_img\n\n        if self.transform is not None:\n            images = self.transform(image=images)['image']\n        images = images.astype(np.float32)\n        images *= 1 / 255\n        images = images.transpose(2, 0, 1)\n\n        label = np.zeros(5).astype(np.float32)\n        label[:row.isup_grade] = 1.\n        return torch.tensor(images), torch.tensor(label)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import json\nimport math\nimport cv2\nimport PIL\nfrom PIL import Image\nimport numpy as np\nfrom keras import layers\nfrom keras.applications import DenseNet121\nfrom keras.callbacks import Callback, ModelCheckpoint\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.models import Sequential\nfrom keras.optimizers import Adam\nimport matplotlib.pyplot as plt\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import cohen_kappa_score, accuracy_score\nimport scipy\nimport tensorflow as tf\nfrom tqdm import tqdm\n%matplotlib inline","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install --upgrade keras keras-applications","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Densenet","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport os","metadata":{"execution":{"iopub.status.busy":"2022-04-28T14:25:15.731463Z","iopub.execute_input":"2022-04-28T14:25:15.732031Z","iopub.status.idle":"2022-04-28T14:25:15.74292Z","shell.execute_reply.started":"2022-04-28T14:25:15.73193Z","shell.execute_reply":"2022-04-28T14:25:15.741926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.getcwd()","metadata":{"execution":{"iopub.status.busy":"2022-04-28T14:25:16.944959Z","iopub.execute_input":"2022-04-28T14:25:16.945522Z","iopub.status.idle":"2022-04-28T14:25:16.953118Z","shell.execute_reply.started":"2022-04-28T14:25:16.945481Z","shell.execute_reply":"2022-04-28T14:25:16.952311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv(\"../input/prostate-cancer-grade-assessment/train.csv\")\ndf_train.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-04-28T14:25:18.675857Z","iopub.execute_input":"2022-04-28T14:25:18.676271Z","iopub.status.idle":"2022-04-28T14:25:18.717134Z","shell.execute_reply.started":"2022-04-28T14:25:18.676233Z","shell.execute_reply":"2022-04-28T14:25:18.716309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_img_path = \"/kaggle/input/prostate-cancer-grade-assessment/train_images\"\nlabel_path = \"/kaggle/input/prostate-cancer-grade-assessment/train_label_masks\"\n\ntrain_img = [img for img in os.listdir(train_img_path)]\ntrain_label = [label for label in os.listdir(label_path)]\n\ntrain_img = list(sorted(train_img))\ntrain_label = list(sorted(train_label))","metadata":{"execution":{"iopub.status.busy":"2022-04-28T14:25:45.899751Z","iopub.execute_input":"2022-04-28T14:25:45.900048Z","iopub.status.idle":"2022-04-28T14:25:46.419835Z","shell.execute_reply.started":"2022-04-28T14:25:45.900019Z","shell.execute_reply":"2022-04-28T14:25:46.419101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from matplotlib import rcParams\nimport openslide\nimport cv2\nfrom IPython.display import display\n\n# rcParams[\"figure.figsize\"] = 15, 15\n\nfor i in range(22, 25):\n    img = openslide.OpenSlide(train_img_path + \"/\" + train_img[i])\n    display(img.get_thumbnail(size=(600, 400)))\n    img.close()\n    d = df_train.loc[df_train[\"image_id\"] == train_img[i][:-5]]\n    print(f\"PROVIDED BY: {d.data_provider.values[0]}\")\n    print(f\"ISUP Grade: {d.isup_grade.values[0]}, Gleason Grade: {d.gleason_score.values[0]}\")","metadata":{"execution":{"iopub.status.busy":"2022-04-28T14:26:30.579486Z","iopub.execute_input":"2022-04-28T14:26:30.58009Z","iopub.status.idle":"2022-04-28T14:26:31.255153Z","shell.execute_reply.started":"2022-04-28T14:26:30.579875Z","shell.execute_reply":"2022-04-28T14:26:31.254194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib\n\nrcParams[\"figure.figsize\"] = 7, 6\n\nfor i in range(22, 25):\n    mask = openslide.OpenSlide(label_path + \"/\" + train_label[i])\n    mask = mask.get_thumbnail(size=(600, 400))\n    mask = np.asarray(mask)\n    mask = mask[:,:,0]\n    cmap = matplotlib.colors.ListedColormap(['black', 'gray', 'green', 'yellow', 'orange', 'red'])\n\n    plt.imshow(mask, cmap=cmap, interpolation='nearest', vmin=0, vmax=5)\n    plt.axis('off')\n    plt.show()\n#     mask.close()","metadata":{"execution":{"iopub.status.busy":"2022-04-28T14:26:35.848987Z","iopub.execute_input":"2022-04-28T14:26:35.849615Z","iopub.status.idle":"2022-04-28T14:26:36.581346Z","shell.execute_reply.started":"2022-04-28T14:26:35.849577Z","shell.execute_reply":"2022-04-28T14:26:36.58054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\n\nSIZE = 200\n\nresized_imgs_path = \"../input/panda-resized-train-data-512x512/train_images/train_images\"\nimg_array = []\nfor i in os.listdir(resized_imgs_path):\n    img = resized_imgs_path + \"/\" + i\n    img = cv2.resize(cv2.imread(img), (SIZE, SIZE))\n    img_array.append(img)","metadata":{"execution":{"iopub.status.busy":"2022-04-28T14:26:44.672979Z","iopub.execute_input":"2022-04-28T14:26:44.67353Z","iopub.status.idle":"2022-04-28T14:28:27.082708Z","shell.execute_reply.started":"2022-04-28T14:26:44.673488Z","shell.execute_reply":"2022-04-28T14:28:27.081849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(img_array[1])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-04-28T14:28:34.029511Z","iopub.execute_input":"2022-04-28T14:28:34.029777Z","iopub.status.idle":"2022-04-28T14:28:34.268693Z","shell.execute_reply.started":"2022-04-28T14:28:34.029748Z","shell.execute_reply":"2022-04-28T14:28:34.267967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelBinarizer\n\ntrain_y = list(df_train['isup_grade'].values)\nlb = LabelBinarizer()\ntrain_y = lb.fit_transform(train_y)","metadata":{"execution":{"iopub.status.busy":"2022-04-28T14:28:38.082514Z","iopub.execute_input":"2022-04-28T14:28:38.083583Z","iopub.status.idle":"2022-04-28T14:28:38.741408Z","shell.execute_reply.started":"2022-04-28T14:28:38.08353Z","shell.execute_reply":"2022-04-28T14:28:38.740649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_y_multi = np.empty(train_y.shape, dtype=train_y.dtype)\ntrain_y_multi[:, 5] = train_y[:, 5]\n\nfor i in range(4, -1, -1):\n    train_y_multi[:, i] = np.logical_or(train_y[:, i], train_y_multi[:,i+1])","metadata":{"execution":{"iopub.status.busy":"2022-04-28T14:28:40.114294Z","iopub.execute_input":"2022-04-28T14:28:40.114733Z","iopub.status.idle":"2022-04-28T14:28:40.122409Z","shell.execute_reply.started":"2022-04-28T14:28:40.114693Z","shell.execute_reply":"2022-04-28T14:28:40.121308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_y_multi.sum(axis=0)","metadata":{"execution":{"iopub.status.busy":"2022-04-28T14:28:42.65805Z","iopub.execute_input":"2022-04-28T14:28:42.658946Z","iopub.status.idle":"2022-04-28T14:28:42.665334Z","shell.execute_reply.started":"2022-04-28T14:28:42.658897Z","shell.execute_reply":"2022-04-28T14:28:42.664582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_x = np.reshape(img_array, (len(img_array), SIZE, SIZE, 3))","metadata":{"execution":{"iopub.status.busy":"2022-04-28T14:28:43.89095Z","iopub.execute_input":"2022-04-28T14:28:43.891569Z","iopub.status.idle":"2022-04-28T14:28:44.299511Z","shell.execute_reply.started":"2022-04-28T14:28:43.891524Z","shell.execute_reply":"2022-04-28T14:28:44.298704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = pd.read_csv(\"../input/prostate-cancer-grade-assessment/test.csv\")\ndf_test.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-04-28T14:28:45.37742Z","iopub.execute_input":"2022-04-28T14:28:45.378082Z","iopub.status.idle":"2022-04-28T14:28:45.394069Z","shell.execute_reply.started":"2022-04-28T14:28:45.378038Z","shell.execute_reply":"2022-04-28T14:28:45.393313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_img = [img+\".tiff\" for img in df_test['image_id'].values]","metadata":{"execution":{"iopub.status.busy":"2022-04-28T14:28:46.588712Z","iopub.execute_input":"2022-04-28T14:28:46.58951Z","iopub.status.idle":"2022-04-28T14:28:46.594937Z","shell.execute_reply.started":"2022-04-28T14:28:46.58947Z","shell.execute_reply":"2022-04-28T14:28:46.593983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_img_path = '../input/prostate-cancer-grade-assessment/test_images'\ntest_x = []\n\nif os.path.exists(test_img_path):\n    for i in range(len(test_img)):\n        img = test_img_path + \"/\" + test_img[i]\n        img = preprocessing_img(img)\n        test_x.append(img)\n    test_x = np.reshape(train_img, (len(test_img), SIZE, SIZE, 3))\nelse:\n    test_x = np.random.rand(len(test_img),SIZE, SIZE, 3)","metadata":{"execution":{"iopub.status.busy":"2022-04-28T14:28:47.473995Z","iopub.execute_input":"2022-04-28T14:28:47.474756Z","iopub.status.idle":"2022-04-28T14:28:47.485935Z","shell.execute_reply.started":"2022-04-28T14:28:47.47472Z","shell.execute_reply":"2022-04-28T14:28:47.485173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\ntrain_x, val_x, train_y, val_y = train_test_split(\n    train_x, train_y_multi, train_size=0.8, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-04-28T14:28:48.688571Z","iopub.execute_input":"2022-04-28T14:28:48.688848Z","iopub.status.idle":"2022-04-28T14:28:49.103871Z","shell.execute_reply.started":"2022-04-28T14:28:48.688816Z","shell.execute_reply":"2022-04-28T14:28:49.103046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.applications import DenseNet121\n\ndensenet = DenseNet121(\n    weights = '../input/densenet-p/DenseNet-BC-121-32-no-top.h5',\n    include_top=False,\n    input_shape=(SIZE, SIZE, 3)\n)","metadata":{"execution":{"iopub.status.busy":"2022-04-28T14:28:50.164588Z","iopub.execute_input":"2022-04-28T14:28:50.16551Z","iopub.status.idle":"2022-04-28T14:28:57.870071Z","shell.execute_reply.started":"2022-04-28T14:28:50.165455Z","shell.execute_reply":"2022-04-28T14:28:57.869198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.models import Sequential\nfrom tensorflow.keras import layers as ly\nfrom tensorflow.keras.optimizers import Adam\n\nmodel = Sequential([\n    densenet,\n    ly.GlobalAveragePooling2D(),\n    ly.Dropout(0.8),\n    ly.Dense(6, activation=\"sigmoid\")\n])\n\nmodel.compile(loss=\"binary_crossentropy\", optimizer=Adam(lr=0.01), metrics=[\"accuracy\"])","metadata":{"execution":{"iopub.status.busy":"2022-04-28T14:28:57.871708Z","iopub.execute_input":"2022-04-28T14:28:57.87198Z","iopub.status.idle":"2022-04-28T14:28:58.663215Z","shell.execute_reply.started":"2022-04-28T14:28:57.871941Z","shell.execute_reply":"2022-04-28T14:28:58.661458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2022-04-28T14:29:03.384875Z","iopub.execute_input":"2022-04-28T14:29:03.385612Z","iopub.status.idle":"2022-04-28T14:29:03.426459Z","shell.execute_reply.started":"2022-04-28T14:29:03.385556Z","shell.execute_reply":"2022-04-28T14:29:03.425379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.preprocessing.image import ImageDataGenerator\n\ndata = ImageDataGenerator(\n    zoom_range = 0.15,\n    fill_mode=\"nearest\",\n    cval=0.,\n    horizontal_flip=True,\n    vertical_flip=True\n)\n\ndata = data.flow(train_x, train_y, batch_size=10, seed=42)","metadata":{"execution":{"iopub.status.busy":"2022-04-28T14:29:10.349878Z","iopub.execute_input":"2022-04-28T14:29:10.350202Z","iopub.status.idle":"2022-04-28T14:29:11.322925Z","shell.execute_reply.started":"2022-04-28T14:29:10.350166Z","shell.execute_reply":"2022-04-28T14:29:11.322137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit_generator(\n    data, steps_per_epoch=train_x.shape[0] / 10,\n    epochs=50,\n    validation_data=(val_x, val_y)\n)","metadata":{"execution":{"iopub.status.busy":"2022-04-28T14:29:17.409935Z","iopub.execute_input":"2022-04-28T14:29:17.410214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"acc = history.history[\"accuracy\"]\nval_acc = history.history[\"val_accuracy\"]\nloss = history.history[\"loss\"]\nval_loss = history.history[\"val_loss\"]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(acc)\nplt.plot(val_acc)\nplt.legend([\"accuracy\", \"val_accuracy\"])\nplt.title(\"Accuracy and Validation Accuracy throughout epochs\")\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(loss)\nplt.plot(val_loss)\nplt.title(\"Loss and Validation Loss throughout epochs\")\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from random import randint\n\nif os.path.exists(test_img_path):\n    test_y = model.predict(test_x)\n    test_y = test_y > 0.37757874193797547\n    test_y = test_y.astype(int).sum(axis=1) - 1\nelse:\n    test_y = [randint(0, 5) for i in range(3)]\n\ndf_test['isup_grade'] = test_y\ndf_test = df_test[[\"image_id\", \"isup_grade\"]]\ndf_test.to_csv(\"submission.csv\", index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}