{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"},{"sourceId":7272321,"sourceType":"datasetVersion","datasetId":4215881}],"dockerImageVersionId":30559,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport sys\nimport gc\nimport ast\nimport cv2\nimport time\nimport timm\nimport pickle\nimport random\nimport pydicom\nimport argparse\nimport warnings\nimport threading\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\nfrom tqdm import tqdm\nfrom glob import glob\nimport albumentations\nimport matplotlib.pyplot as plt\n\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nimport torch.cuda.amp as amp\nimport torch.nn.functional as F\nfrom torch.utils.data import DataLoader, Dataset\nfrom pylab import rcParams\nfrom albumentations.pytorch import ToTensorV2\n\ndevice = torch.device('cuda')\ntorch.backends.cudnn.benchmark = True","metadata":{"execution":{"iopub.status.busy":"2023-12-24T13:51:45.977901Z","iopub.execute_input":"2023-12-24T13:51:45.978565Z","iopub.status.idle":"2023-12-24T13:51:45.988456Z","shell.execute_reply.started":"2023-12-24T13:51:45.978534Z","shell.execute_reply":"2023-12-24T13:51:45.98766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Define some configurations for running","metadata":{}},{"cell_type":"code","source":"class CFG:\n    kernel_type = '231223_resnet101d_2048_3ch_augv2_mixupp5_drl3_rov1p2_bs2_lr23e6_eta23e6_75ep_ubc_cancer_classiication'\n    model_dir = '../input/resnet101-100-1224/resnet101_100'\n    load_kernel = None\n    load_last = True\n    autocast = True\n\n    n_folds = 5\n    backbone = 'resnet101d'\n    image_size = 2048\n    in_chans = 3\n\n    lr = 2e-4\n    wd = 1e-6\n    n_warmup_steps = 0\n    eta_min = 23e-6\n    lw = [15, 1]\n    batch_size = 8\n    drop_rate = 0.\n    drop_rate_last = 0.3\n    drop_path_rate = 0.\n    p_mixup = 0.5\n    p_rand_order = 0.2\n    p_rand_order_v1 = 0.2\n    seed = 3407\n\n    data_dir = '../input/UBC-OCEAN'\n    data_train_dir = '../input/UBC-OCEAN/train_images'\n    data_train_thumbnails_dir = '../input/UBC-OCEAN/train_thumbnails'\n    data_test_dir = '../input/UBC-OCEAN/test_images'\n    data_test_thumbnails_dir = '../input/UBC-OCEAN/test_thumbnails'\n    \n    use_amp = True\n    num_workers = 4\n    out_dim = 5\n\n    device = torch.device('cuda')\n    ce = nn.CrossEntropyLoss(reduction='none')\n    n_epochs = 100\n\n    transform_train = albumentations.Compose([\n        albumentations.Resize(image_size, image_size),\n        albumentations.Normalize(\n            mean=[0.485, 0.456, 0.406],\n            std=[0.229, 0.224, 0.225],\n            max_pixel_value=255.0,\n            p=1.0\n        ),\n        albumentations.Perspective(p=0.5),\n        albumentations.HorizontalFlip(p=0.5),\n        albumentations.VerticalFlip(p=0.5),\n        albumentations.RandomBrightnessContrast(p=0.75),\n        albumentations.ShiftScaleRotate(p=0.75),\n        albumentations.OneOf([\n            albumentations.GaussNoise(var_limit=[10, 50]),\n            albumentations.GaussianBlur(),\n            albumentations.MotionBlur(),\n        ], p=0.4),\n        albumentations.GridDistortion(num_steps=5, distort_limit=0.3, p=0.5),\n        albumentations.CoarseDropout(max_holes=1, max_width=int(image_size * 0.3), max_height=int(image_size * 0.3),\n                        mask_fill_value=0, p=0.5),\n        ToTensorV2(),\n    ])\n\n    transform_valid = albumentations.Compose([\n        albumentations.Resize(image_size, image_size),\n        albumentations.Normalize(\n            mean=[0.485, 0.456, 0.406],\n            std=[0.229, 0.224, 0.225],\n            max_pixel_value=255.0,\n            p=1.0\n        ),\n        ToTensorV2(),\n    ])","metadata":{"execution":{"iopub.status.busy":"2023-12-24T13:51:49.063854Z","iopub.execute_input":"2023-12-24T13:51:49.064203Z","iopub.status.idle":"2023-12-24T13:51:49.077103Z","shell.execute_reply.started":"2023-12-24T13:51:49.064176Z","shell.execute_reply":"2023-12-24T13:51:49.076131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_train_file_path(image_id):\n    if os.path.exists(f\"{CFG.data_train_thumbnails_dir}/{image_id}_thumbnail.png\"):\n        return f\"{CFG.data_train_thumbnails_dir}/{image_id}_thumbnail.png\"\n    else:\n        return f\"{CFG.data_train_dir}/{image_id}.png\"\n\ndef get_test_file_path(image_id):\n    if os.path.exists(f\"{CFG.data_test_thumbnails_dir}/{image_id}_thumbnail.png\"):\n        return f\"{CFG.data_test_thumbnails_dir}/{image_id}_thumbnail.png\"\n    else:\n        return f\"{CFG.data_test_dir}/{image_id}.png\"\n\n\n\nmode = 'submit' #submit #local\n\nif mode =='local':\n    df = pd.read_csv(f\"{CFG.data_dir}/train.csv\")\n    df['file_path'] = df['image_id'].apply(get_train_file_path)\n\n    \nif mode =='submit':\n    df = pd.read_csv(f\"{CFG.data_dir}/test.csv\")\n    df['file_path'] = df['image_id'].apply(get_test_file_path)\n    \n\n# print(len(df))\n\n# df = df.head(20)\n\n# print(df)\n    \n\n","metadata":{"execution":{"iopub.status.busy":"2023-12-24T13:54:37.15288Z","iopub.execute_input":"2023-12-24T13:54:37.153577Z","iopub.status.idle":"2023-12-24T13:54:37.167444Z","shell.execute_reply.started":"2023-12-24T13:54:37.153547Z","shell.execute_reply":"2023-12-24T13:54:37.166558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prepare dataset","metadata":{}},{"cell_type":"code","source":"class UBCDataset(Dataset):\n    def __init__(self, df, transform=None):\n        self.df = df\n        self.file_names = df['file_path'].values\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, index):\n        row = self.df.iloc[index]\n        img_path = self.file_names[index]\n        img = cv2.imread(img_path)\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n\n        if self.transform:\n            images = self.transform(image=img)[\"image\"]\n\n        return images, row.image_id\n\ndataset_seg = UBCDataset(df, transform=CFG.transform_valid)\nloader_seg = torch.utils.data.DataLoader(dataset_seg, batch_size=CFG.batch_size, shuffle=False, num_workers=CFG.num_workers)\n    ","metadata":{"execution":{"iopub.status.busy":"2023-12-24T13:54:39.434874Z","iopub.execute_input":"2023-12-24T13:54:39.435494Z","iopub.status.idle":"2023-12-24T13:54:39.443799Z","shell.execute_reply.started":"2023-12-24T13:54:39.435464Z","shell.execute_reply":"2023-12-24T13:54:39.442808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load model","metadata":{}},{"cell_type":"code","source":"class Model(nn.Module):\n    def __init__(self, backbone, pretrained=False):\n        super(Model, self).__init__()\n\n        self.encoder = timm.create_model(\n            backbone,\n            in_chans=CFG.in_chans,\n            num_classes=CFG.out_dim,\n            features_only=False,\n            drop_rate=CFG.drop_rate,\n            drop_path_rate=CFG.drop_path_rate,\n            pretrained=pretrained\n        )\n\n        if 'efficient' in backbone:\n            hdim = self.encoder.conv_head.out_channels\n            self.encoder.classifier = nn.Identity()\n        elif 'convnext' in backbone:\n            hdim = self.encoder.head.fc.in_features\n            self.encoder.head.fc = nn.Identity()\n        elif 'resnet' in backbone:\n            hdim = self.encoder.fc.in_features\n            self.encoder.fc = nn.Identity()\n\n        self.lstm = nn.LSTM(hdim, hdim//2, num_layers=1, dropout=CFG.drop_rate, bidirectional=True, batch_first=True)\n        self.head = nn.Sequential(\n            nn.Dropout(CFG.drop_rate_last),\n            nn.Linear(hdim, CFG.out_dim),\n        )\n\n    def forward(self, x):  # (bs, ch, sz, sz)\n        bs = x.shape[0]\n        x = x.view(bs, CFG.in_chans, CFG.image_size, CFG.image_size)\n        feat = self.encoder(x)\n        feat = feat.view(bs, -1)\n        feat, _ = self.lstm(feat)\n        feat = feat.contiguous().view(bs, -1)\n        feat = self.head(feat)\n        feat = feat.view(bs, CFG.out_dim).contiguous()\n        return feat","metadata":{"execution":{"iopub.status.busy":"2023-12-24T13:54:41.95337Z","iopub.execute_input":"2023-12-24T13:54:41.954191Z","iopub.status.idle":"2023-12-24T13:54:41.964699Z","shell.execute_reply.started":"2023-12-24T13:54:41.954159Z","shell.execute_reply":"2023-12-24T13:54:41.963602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models_ubc_ocean = []\n\nfor fold in range(5):\n    model = Model(CFG.backbone, pretrained=False)\n    load_model_file = os.path.join(CFG.model_dir, f'{CFG.kernel_type}_fold{fold}_best.pth')\n    sd = torch.load(load_model_file, map_location='cpu')\n    if 'model_state_dict' in sd.keys():\n        sd = sd['model_state_dict']\n    sd = {k[7:] if k.startswith('module.') else k: sd[k] for k in sd.keys()}\n    model.load_state_dict(sd, strict=True)\n    model = model.to(device)\n    model.eval()\n    models_ubc_ocean.append(model)\n","metadata":{"execution":{"iopub.status.busy":"2023-12-24T13:54:45.048511Z","iopub.execute_input":"2023-12-24T13:54:45.049351Z","iopub.status.idle":"2023-12-24T13:54:50.870975Z","shell.execute_reply.started":"2023-12-24T13:54:45.049318Z","shell.execute_reply":"2023-12-24T13:54:50.869937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Start predicting","metadata":{}},{"cell_type":"code","source":"submit_df_data = []\n\nbar = tqdm(loader_seg)\nwith torch.no_grad():\n    for batch_id, (images, image_id) in enumerate(bar):\n        images = images.cuda()\n        \n        pred = []\n        for _, model in enumerate(models_ubc_ocean):\n            logits = model(images)\n            pred.append(logits.sigmoid())\n        pred = torch.stack(pred, 0).mean(0).cpu()\n        \n\n        for i in range(pred.shape[0]):\n            for j in range(5):\n                pred[i][j] = pred[i][j].clamp(0.0001, 0.9999)\n            max_value, max_index = pred[i].max(dim=0)\n            max_index = max_index.item()\n            label = \"xxx\"\n            if max_index == 0:\n                label = \"CC\"\n            elif max_index == 1:\n                label = \"EC\"\n            elif max_index == 2:\n                label = \"HGSC\"\n            elif max_index == 3:\n                label = \"LGSC\"\n            elif max_index == 4:\n                label = \"MC\" \n            submit_df_data.append({\n                    'image_id': image_id[i].item(),\n                    'label' : label,\n                })\n\nsubmit_df_data","metadata":{"execution":{"iopub.status.busy":"2023-12-24T13:54:55.842046Z","iopub.execute_input":"2023-12-24T13:54:55.842908Z","iopub.status.idle":"2023-12-24T13:54:59.313159Z","shell.execute_reply.started":"2023-12-24T13:54:55.842875Z","shell.execute_reply":"2023-12-24T13:54:59.312024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Generate output file.","metadata":{}},{"cell_type":"code","source":"submit_df = pd.DataFrame(submit_df_data)\nsubmit_df.to_csv('submission.csv',index=False)\nsubmit_df","metadata":{"execution":{"iopub.status.busy":"2023-12-24T13:55:03.466657Z","iopub.execute_input":"2023-12-24T13:55:03.467571Z","iopub.status.idle":"2023-12-24T13:55:03.480579Z","shell.execute_reply.started":"2023-12-24T13:55:03.467536Z","shell.execute_reply":"2023-12-24T13:55:03.479529Z"},"trusted":true},"execution_count":null,"outputs":[]}]}