{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport gc\nimport cv2\nimport math\nimport copy\nimport time\nimport random\nimport glob\nfrom matplotlib import pyplot as plt\n\n# For data manipulation\nimport numpy as np\nimport pandas as pd\n\n# Pytorch Imports\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nimport torch.nn.functional as F\nfrom torch.optim import lr_scheduler\nfrom torch.utils.data import Dataset, DataLoader\nfrom torch.cuda import amp\nimport torchvision\n\n# Utils\nimport joblib\nfrom tqdm import tqdm\nfrom collections import defaultdict\n\n# Sklearn Imports\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import StratifiedKFold\n\n# For Image Models\nimport timm\n\n# Albumentations for augmentations\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\n\n# For colored terminal text\nfrom colorama import Fore, Back, Style\nb_ = Fore.BLUE\nsr_ = Style.RESET_ALL\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\n# For descriptive error messages\nos.environ['CUDA_LAUNCH_BLOCKING'] = \"1\"","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-10-20T08:08:19.323021Z","iopub.execute_input":"2023-10-20T08:08:19.323328Z","iopub.status.idle":"2023-10-20T08:08:26.417443Z","shell.execute_reply.started":"2023-10-20T08:08:19.323306Z","shell.execute_reply":"2023-10-20T08:08:26.416343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CONFIG = {\n    \"seed\": 3407,\n    \"epochs\": 60,\n    \"img_size\": 1024,\n    \"model_name\": \"resnet50\",\n    \"num_classes\": 8,\n    \"train_batch_size\": 8,\n    \"valid_batch_size\": 8,\n    \"learning_rate\": 1e-4,\n    \"scheduler\": 'CosineAnnealingLR',\n    \"min_lr\": 1e-6,\n    \"T_max\": 500,\n    \"weight_decay\": 1e-6,\n    \"fold\" : 0,\n    \"n_fold\": 5,\n    \"n_accumulate\": 1,\n    \"device\": torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\"),\n}","metadata":{"execution":{"iopub.status.busy":"2023-10-20T08:08:26.419207Z","iopub.execute_input":"2023-10-20T08:08:26.419777Z","iopub.status.idle":"2023-10-20T08:08:26.450616Z","shell.execute_reply.started":"2023-10-20T08:08:26.41974Z","shell.execute_reply":"2023-10-20T08:08:26.449771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def set_seed(seed=42):\n    '''Sets the seed of the entire notebook so results are the same every time we run.\n    This is for REPRODUCIBILITY.'''\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    # When running on the CuDNN backend, two further options must be set\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False\n    # Set a fixed value for the hash seed\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    \nset_seed(CONFIG['seed'])","metadata":{"execution":{"iopub.status.busy":"2023-10-20T08:08:36.468047Z","iopub.execute_input":"2023-10-20T08:08:36.468713Z","iopub.status.idle":"2023-10-20T08:08:36.477607Z","shell.execute_reply.started":"2023-10-20T08:08:36.46868Z","shell.execute_reply":"2023-10-20T08:08:36.476959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROOT_DIR = '/kaggle/input/UBC-OCEAN'\nTEST_DIR = '/kaggle/input/UBC-OCEAN/test_thumbnails'\n\nLABEL_ENCODER_BIN = \"/kaggle/input/resnet50/label_encoder.pkl\"\nBEST_WEIGHT = \"/kaggle/input/resnet50/rn50-Acc0.76_Loss0.5396_epoch27.bin/rn50-Acc0.76_Loss0.5396_epoch27.bin\"\nBEST_WEIGHT2 = \"/kaggle/input/resnet50/rn50-Acc0.75_Loss0.9691_epoch50.bin/Acc0.75_Loss0.9691_epoch50.bin\"","metadata":{"execution":{"iopub.status.busy":"2023-10-20T08:14:43.368629Z","iopub.execute_input":"2023-10-20T08:14:43.36893Z","iopub.status.idle":"2023-10-20T08:14:43.37322Z","shell.execute_reply.started":"2023-10-20T08:14:43.368907Z","shell.execute_reply":"2023-10-20T08:14:43.372265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_test_file_path(image_id):\n    return f\"{TEST_DIR}/{image_id}_thumbnail.png\"","metadata":{"execution":{"iopub.status.busy":"2023-10-20T08:11:37.622247Z","iopub.execute_input":"2023-10-20T08:11:37.622903Z","iopub.status.idle":"2023-10-20T08:11:37.626643Z","shell.execute_reply.started":"2023-10-20T08:11:37.622874Z","shell.execute_reply":"2023-10-20T08:11:37.625712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(f\"{ROOT_DIR}/test.csv\")\ndf['file_path'] = df['image_id'].apply(get_test_file_path)\ndf['label'] = 0 # dummy\ndf","metadata":{"execution":{"iopub.status.busy":"2023-10-20T08:11:47.52249Z","iopub.execute_input":"2023-10-20T08:11:47.523259Z","iopub.status.idle":"2023-10-20T08:11:47.554504Z","shell.execute_reply.started":"2023-10-20T08:11:47.523228Z","shell.execute_reply":"2023-10-20T08:11:47.553775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sub = pd.read_csv(f\"{ROOT_DIR}/sample_submission.csv\")\ndf_sub","metadata":{"execution":{"iopub.status.busy":"2023-10-20T08:12:05.238492Z","iopub.execute_input":"2023-10-20T08:12:05.239307Z","iopub.status.idle":"2023-10-20T08:12:05.251228Z","shell.execute_reply.started":"2023-10-20T08:12:05.239276Z","shell.execute_reply":"2023-10-20T08:12:05.25033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"encoder = joblib.load( LABEL_ENCODER_BIN )","metadata":{"execution":{"iopub.status.busy":"2023-10-20T08:12:15.75101Z","iopub.execute_input":"2023-10-20T08:12:15.751727Z","iopub.status.idle":"2023-10-20T08:12:15.760585Z","shell.execute_reply.started":"2023-10-20T08:12:15.751699Z","shell.execute_reply":"2023-10-20T08:12:15.759965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class UBCDataset(Dataset):\n    def __init__(self, df, transforms=None):\n        self.df = df\n        self.file_names = df['file_path'].values\n        self.labels = df['label'].values\n        self.transforms = transforms\n        \n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, index):\n        img_path = self.file_names[index]\n        img = cv2.imread(img_path)\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        label = self.labels[index]\n        \n        if self.transforms:\n            img = self.transforms(image=img)[\"image\"]\n            \n        return {\n            'image': img,\n            'label': torch.tensor(label, dtype=torch.long)\n        }","metadata":{"execution":{"iopub.status.busy":"2023-10-20T08:12:30.110847Z","iopub.execute_input":"2023-10-20T08:12:30.111199Z","iopub.status.idle":"2023-10-20T08:12:30.117424Z","shell.execute_reply.started":"2023-10-20T08:12:30.111173Z","shell.execute_reply":"2023-10-20T08:12:30.116478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_transforms = {\n    \"valid\": A.Compose([\n        A.Resize(CONFIG['img_size'], CONFIG['img_size']),      \n        A.Normalize(\n                mean=[0.485, 0.456, 0.406], \n                std=[0.229, 0.224, 0.225], \n                max_pixel_value=255.0, \n                p=1.0\n            ),\n        ToTensorV2()], p=1.)\n}","metadata":{"execution":{"iopub.status.busy":"2023-10-20T08:12:54.672483Z","iopub.execute_input":"2023-10-20T08:12:54.673184Z","iopub.status.idle":"2023-10-20T08:12:54.678242Z","shell.execute_reply.started":"2023-10-20T08:12:54.673157Z","shell.execute_reply":"2023-10-20T08:12:54.677156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class UBCModel(nn.Module):\n    def __init__(self, model_name, num_classes, pretrained=False, checkpoint_path=None):\n        super(UBCModel, self).__init__()\n        self.model = timm.create_model(model_name, pretrained=pretrained, checkpoint_path=checkpoint_path)\n        out_features = self.model.fc.out_features\n        self.linear0 = nn.Linear(out_features, out_features//2)\n        self.linear = nn.Linear(out_features//2, num_classes)\n        self.softmax = nn.Softmax(dim=1)\n    def forward(self, images):\n        features = self.model(images)\n        output = self.linear0(features)\n        output = self.linear(output)\n        return output\n    \nmodel = UBCModel(CONFIG['model_name'], CONFIG['num_classes'])\nmodel.load_state_dict(torch.load( BEST_WEIGHT ))\nmodel.to(CONFIG['device']);","metadata":{"execution":{"iopub.status.busy":"2023-10-20T08:14:49.494073Z","iopub.execute_input":"2023-10-20T08:14:49.494428Z","iopub.status.idle":"2023-10-20T08:14:54.112737Z","shell.execute_reply.started":"2023-10-20T08:14:49.494402Z","shell.execute_reply":"2023-10-20T08:14:54.112045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset = UBCDataset(df, transforms=data_transforms[\"valid\"])\ntest_loader = DataLoader(test_dataset, batch_size=CONFIG['valid_batch_size'], \n                          num_workers=2, shuffle=False, pin_memory=True)","metadata":{"execution":{"iopub.status.busy":"2023-10-20T08:15:04.441611Z","iopub.execute_input":"2023-10-20T08:15:04.441941Z","iopub.status.idle":"2023-10-20T08:15:04.447365Z","shell.execute_reply.started":"2023-10-20T08:15:04.441913Z","shell.execute_reply":"2023-10-20T08:15:04.446329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = []\nwith torch.no_grad():\n    bar = tqdm(enumerate(test_loader), total=len(test_loader))\n    for step, data in bar:        \n        images = data['image'].to(CONFIG[\"device\"], dtype=torch.float)        \n        batch_size = images.size(0)\n        outputs = model(images)\n        preds.append(model.softmax(outputs))\n\nmodel.load_state_dict(torch.load( BEST_WEIGHT2 ))\nmodel.to(CONFIG['device']);\n\npreds1 = []\nwith torch.no_grad():\n    bar = tqdm(enumerate(test_loader), total=len(test_loader))\n    for step, data in bar:        \n        images = data['image'].to(CONFIG[\"device\"], dtype=torch.float)        \n        batch_size = images.size(0)\n        outputs = model(images)\n        preds1.append(model.softmax(outputs))\n# weighted\nfinal_preds = []\nfor i,j in zip(preds, preds1):\n    outputs = i * 0.7 + j * 0.3\n    _, predicted = torch.max(outputs, 1)\n    final_preds.append( predicted.detach().cpu().numpy() )\n    \nfinal_preds = np.concatenate(final_preds).flatten()\npred_labels = encoder.inverse_transform( final_preds )","metadata":{"execution":{"iopub.status.busy":"2023-10-20T08:15:53.523901Z","iopub.execute_input":"2023-10-20T08:15:53.524248Z","iopub.status.idle":"2023-10-20T08:15:55.209939Z","shell.execute_reply.started":"2023-10-20T08:15:53.524222Z","shell.execute_reply":"2023-10-20T08:15:55.2089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sub[\"label\"] = pred_labels\ndf_sub.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-10-20T08:15:57.602301Z","iopub.execute_input":"2023-10-20T08:15:57.602636Z","iopub.status.idle":"2023-10-20T08:15:57.613016Z","shell.execute_reply.started":"2023-10-20T08:15:57.602608Z","shell.execute_reply":"2023-10-20T08:15:57.612237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sub","metadata":{"execution":{"iopub.status.busy":"2023-10-20T08:16:08.91584Z","iopub.execute_input":"2023-10-20T08:16:08.916169Z","iopub.status.idle":"2023-10-20T08:16:08.924656Z","shell.execute_reply.started":"2023-10-20T08:16:08.916146Z","shell.execute_reply":"2023-10-20T08:16:08.923667Z"},"trusted":true},"execution_count":null,"outputs":[]}]}