{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\n\nimport numpy as np\nimport pandas as pd\n\nfrom matplotlib import pyplot as plt\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2022-07-20T03:38:12.101559Z","iopub.execute_input":"2022-07-20T03:38:12.102962Z","iopub.status.idle":"2022-07-20T03:38:12.82197Z","shell.execute_reply.started":"2022-07-20T03:38:12.102832Z","shell.execute_reply":"2022-07-20T03:38:12.820483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('../input/mayo-clinic-strip-ai/test.csv')\ndf","metadata":{"execution":{"iopub.status.busy":"2022-07-20T03:38:12.824196Z","iopub.execute_input":"2022-07-20T03:38:12.824587Z","iopub.status.idle":"2022-07-20T03:38:12.855201Z","shell.execute_reply.started":"2022-07-20T03:38:12.824548Z","shell.execute_reply":"2022-07-20T03:38:12.854153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['path'] = '../input/mayo-clinic-strip-ai/test/' + df['image_id'] + '.tif'\ndf","metadata":{"execution":{"iopub.status.busy":"2022-07-20T03:38:12.85977Z","iopub.execute_input":"2022-07-20T03:38:12.860129Z","iopub.status.idle":"2022-07-20T03:38:12.883188Z","shell.execute_reply.started":"2022-07-20T03:38:12.860094Z","shell.execute_reply":"2022-07-20T03:38:12.881958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['pred'] = 0.5\ndf","metadata":{"execution":{"iopub.status.busy":"2022-07-20T03:38:12.889482Z","iopub.execute_input":"2022-07-20T03:38:12.889835Z","iopub.status.idle":"2022-07-20T03:38:12.90949Z","shell.execute_reply.started":"2022-07-20T03:38:12.8898Z","shell.execute_reply":"2022-07-20T03:38:12.908204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nfrom torch import nn, optim\nimport torch.nn.functional as F\nfrom torch.utils.data import DataLoader","metadata":{"execution":{"iopub.status.busy":"2022-07-20T03:38:12.91169Z","iopub.execute_input":"2022-07-20T03:38:12.91409Z","iopub.status.idle":"2022-07-20T03:38:14.428838Z","shell.execute_reply.started":"2022-07-20T03:38:12.914044Z","shell.execute_reply":"2022-07-20T03:38:14.427873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import sys\nsys.path.append('../input/timm-pytorch-image-models/pytorch-image-models-master')\nimport timm","metadata":{"execution":{"iopub.status.busy":"2022-07-20T03:38:14.430508Z","iopub.execute_input":"2022-07-20T03:38:14.431069Z","iopub.status.idle":"2022-07-20T03:38:16.398619Z","shell.execute_reply.started":"2022-07-20T03:38:14.431033Z","shell.execute_reply":"2022-07-20T03:38:16.397477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_size = 224 # for ViT\n\nfrom albumentations import Compose, OneOf, Normalize, Resize\nfrom albumentations.pytorch import ToTensorV2\nfrom albumentations import ImageOnlyTransform\n# ====================================================\n# Transforms\n# ====================================================\ndef get_transforms(*, data):\n    if data == 'valid':\n        return Compose([\n            Resize(img_size, img_size),\n            Normalize(\n                mean=[0.485, 0.456, 0.406],\n                std=[0.229, 0.224, 0.225],\n            ),\n            ToTensorV2(),\n        ])","metadata":{"execution":{"iopub.status.busy":"2022-07-20T03:38:16.402064Z","iopub.execute_input":"2022-07-20T03:38:16.402372Z","iopub.status.idle":"2022-07-20T03:38:17.501527Z","shell.execute_reply.started":"2022-07-20T03:38:16.402347Z","shell.execute_reply":"2022-07-20T03:38:17.500453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2 # cv2 can't read the whole image\nfrom torch.utils.data import DataLoader, Dataset\nLABEL = 'label_CE'\n# ====================================================\n# Dataset\n# ====================================================\n\nclass TestDataset(Dataset):\n    def __init__(self, files, transform=None):\n        self.files = files\n        self.transform = transform\n        \n    def __len__(self):\n        return len(self.files)\n\n    def __getitem__(self, idx):\n        file_path = self.files[idx]\n        image = cv2.imread(file_path)\n        if self.transform:\n            augmented = self.transform(image=image)\n            image = augmented['image']\n        return image","metadata":{"execution":{"iopub.status.busy":"2022-07-20T03:38:17.503197Z","iopub.execute_input":"2022-07-20T03:38:17.503599Z","iopub.status.idle":"2022-07-20T03:38:17.512039Z","shell.execute_reply.started":"2022-07-20T03:38:17.50355Z","shell.execute_reply":"2022-07-20T03:38:17.511026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import skimage.io as io\n\ndef read_tiff(image_path):\n    image = io.imread(image_path)\n    image = np.squeeze(image) # some images have unnecessary axes with shape 1 --> remove\n    if image.shape[0] == 3: # some images have color as first axis -> swap axes\n        image = image.swapaxes(0,1)\n        image = image.swapaxes(1,2)\n    return image","metadata":{"execution":{"iopub.status.busy":"2022-07-20T03:38:17.513656Z","iopub.execute_input":"2022-07-20T03:38:17.514303Z","iopub.status.idle":"2022-07-20T03:38:17.727999Z","shell.execute_reply.started":"2022-07-20T03:38:17.514262Z","shell.execute_reply":"2022-07-20T03:38:17.727113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMG_SIZE = 1024\n\ndef slice_images(image_id, image):\n    print('Slicing Image ' + image_id + ' ...')\n    possible_slices_x = image.shape[0] // IMG_SIZE\n    possible_slices_y = image.shape[1] // IMG_SIZE\n    for x in range(possible_slices_x):\n        for y in range(possible_slices_y):\n            image_slice = image[x * IMG_SIZE : (x+1) * IMG_SIZE, y * IMG_SIZE : (y+1) * IMG_SIZE]\n            intensity = image_slice.mean()\n            if (intensity > 100) & (intensity < 200):\n                cv2.imwrite(f\"./{image_id}-imgslice.{x}.{y}.jpg\", image_slice)\n    del image, image_slice\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-20T03:38:17.731796Z","iopub.execute_input":"2022-07-20T03:38:17.73206Z","iopub.status.idle":"2022-07-20T03:38:17.7389Z","shell.execute_reply.started":"2022-07-20T03:38:17.732035Z","shell.execute_reply":"2022-07-20T03:38:17.737783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\nimport glob\nimport shutil\n\nfrom tqdm import tqdm\nimport itertools\nmodelname = 'swin_tiny_patch4_window7_224'","metadata":{"execution":{"iopub.status.busy":"2022-07-20T03:38:17.741201Z","iopub.execute_input":"2022-07-20T03:38:17.742218Z","iopub.status.idle":"2022-07-20T03:38:17.747912Z","shell.execute_reply.started":"2022-07-20T03:38:17.74218Z","shell.execute_reply":"2022-07-20T03:38:17.746933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"torch.cuda.is_available()","metadata":{"execution":{"iopub.status.busy":"2022-07-20T03:38:17.749396Z","iopub.execute_input":"2022-07-20T03:38:17.749964Z","iopub.status.idle":"2022-07-20T03:38:17.808328Z","shell.execute_reply.started":"2022-07-20T03:38:17.749928Z","shell.execute_reply":"2022-07-20T03:38:17.807384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def prediction(image_path):\n    slice_images('image', read_tiff(image_path))\n    imgs = glob.glob('./*.jpg')\n    test_dataset = TestDataset(imgs, transform=get_transforms(data='valid'))\n    test_loader = DataLoader(test_dataset, batch_size=64, shuffle=False, num_workers=2, pin_memory=False, drop_last=False)\n    paths = [f'../input/mayo-timm-model-0717/model{modelname}_fold{i}.pth' for i in range(5)]\n    preds = []\n    for path in tqdm(paths):\n        model = timm.create_model(modelname, pretrained = False, in_chans=3, num_classes=1).cuda()\n        model.load_state_dict(torch.load(path))\n        model.eval()\n        pred = []\n        with torch.no_grad():\n            for image in test_loader:\n                output = model(image.cuda())\n                outputnp = output.sigmoid().cpu().detach().numpy()\n                pred.append(outputnp)\n                del image\n                gc.collect()\n        pred = np.array(list(itertools.chain.from_iterable(pred)))\n        preds.append(pred)\n        del model, pred\n        gc.collect()\n    pred = np.mean(preds)\n    del imgs, test_dataset, test_loader, preds\n    gc.collect()\n    torch.cuda.empty_cache()\n    return pred","metadata":{"execution":{"iopub.status.busy":"2022-07-20T03:38:17.810564Z","iopub.execute_input":"2022-07-20T03:38:17.811269Z","iopub.status.idle":"2022-07-20T03:38:17.822941Z","shell.execute_reply.started":"2022-07-20T03:38:17.811232Z","shell.execute_reply":"2022-07-20T03:38:17.82209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import resource\nresource.setrlimit(resource.RLIMIT_DATA, (12 * 1024 ** 3, -1)) # 12GB\nresource.getrlimit(resource.RLIMIT_DATA)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T03:38:17.824558Z","iopub.execute_input":"2022-07-20T03:38:17.824959Z","iopub.status.idle":"2022-07-20T03:38:17.837836Z","shell.execute_reply.started":"2022-07-20T03:38:17.824923Z","shell.execute_reply":"2022-07-20T03:38:17.836677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(len(df.index)):\n    gc.collect()\n    torch.cuda.empty_cache()\n    imgs = glob.glob('./*.jpg')\n    for p in imgs:\n        os.remove(p)\n    try:\n        image_path = df['path'][i]\n        pred = prediction(image_path)\n    except Exception as e:\n        print(type(e), e)\n        pred = 0.5\n    df.loc[i, 'pred'] = pred","metadata":{"execution":{"iopub.status.busy":"2022-07-20T03:38:17.839475Z","iopub.execute_input":"2022-07-20T03:38:17.83987Z","iopub.status.idle":"2022-07-20T03:38:56.502162Z","shell.execute_reply.started":"2022-07-20T03:38:17.839838Z","shell.execute_reply":"2022-07-20T03:38:56.500846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imgs = glob.glob('./*.jpg')\nfor p in imgs:\n    os.remove(p)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_agg = df.groupby('patient_id')[['pred']].agg(['mean', 'min', 'max', 'median']).reset_index()\ndf_agg.columns = ['patient_id', 'mean', 'min', 'max', 'median']\ndf_agg","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val = 'mean'\ndf_agg['CE'] = df_agg[val]\ndf_agg['LAA']= 1 - df_agg['CE']\ndf_agg = df_agg.drop(columns=['mean', 'min', 'max', 'median'])\ndf_agg","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv(\"../input/mayo-clinic-strip-ai/sample_submission.csv\")\nsubmission","metadata":{"execution":{"iopub.status.busy":"2022-07-20T03:38:56.504267Z","iopub.execute_input":"2022-07-20T03:38:56.504755Z","iopub.status.idle":"2022-07-20T03:38:56.518477Z","shell.execute_reply.started":"2022-07-20T03:38:56.504708Z","shell.execute_reply":"2022-07-20T03:38:56.517211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = submission.drop(columns=['CE', 'LAA']).merge(df_agg, on='patient_id', how='left') \nsub.fillna(0.5).to_csv('submission.csv', index=False)\nsub","metadata":{"execution":{"iopub.status.busy":"2022-07-20T03:38:56.520483Z","iopub.execute_input":"2022-07-20T03:38:56.520732Z","iopub.status.idle":"2022-07-20T03:38:56.544233Z","shell.execute_reply.started":"2022-07-20T03:38:56.520708Z","shell.execute_reply":"2022-07-20T03:38:56.543198Z"},"trusted":true},"execution_count":null,"outputs":[]}]}