{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"},{"sourceId":7029169,"sourceType":"datasetVersion","datasetId":4042935},{"sourceId":7030378,"sourceType":"datasetVersion","datasetId":4043770}],"dockerImageVersionId":30587,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport random\nfrom sklearn.model_selection import train_test_split\nimport numpy as np\nfrom torch.utils.data import Dataset, DataLoader\nfrom tqdm import tqdm\nfrom sklearn.metrics import balanced_accuracy_score, accuracy_score\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torchvision import transforms\nimport torch.optim as optim\nimport torch.nn as nn\nimport torchvision\nfrom PIL import Image\nfrom torch.utils.data import DataLoader, Dataset\nimport torchvision.models as models\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.optim.lr_scheduler import CosineAnnealingLR\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-11-22T22:55:18.785185Z","iopub.execute_input":"2023-11-22T22:55:18.785441Z","iopub.status.idle":"2023-11-22T22:55:23.731036Z","shell.execute_reply.started":"2023-11-22T22:55:18.785416Z","shell.execute_reply":"2023-11-22T22:55:23.730068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Transfer Learning with MobileNetV2 Model\n\nThis code snippet involves modifying a pre-trained MobileNetV2 model for a custom classification task and loading previously trained weights.\n\n#### Model Modification","metadata":{}},{"cell_type":"code","source":"import torch\nfrom torchvision import models\nimport torch.nn as nn\n\nmodel = models.mobilenet_v2(pretrained=False)\nnum_classes = 5\nmodel.classifier[1] = nn.Linear(model.last_channel, num_classes)\n\n# Define the path where the model was saved\nsaved_model_path = '/kaggle/input/mobilenetv2-finetuned-with-ucb/epoch_9_mobile2.pth'\n\n# Load the state dictionary\nstate_dict = torch.load(saved_model_path, map_location=device)\n\n# Load state_dict into the model\nmodel.load_state_dict(state_dict)","metadata":{"execution":{"iopub.status.busy":"2023-11-22T22:55:23.732708Z","iopub.execute_input":"2023-11-22T22:55:23.733169Z","iopub.status.idle":"2023-11-22T22:55:27.184009Z","shell.execute_reply.started":"2023-11-22T22:55:23.733141Z","shell.execute_reply":"2023-11-22T22:55:27.183214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Image Processing Functions\n\n#### `get_tiles(img, tile_size=256, n_tiles=30, mode=0)`\n\nDivides an image into tiles, pads if necessary, and sorts them by content, returning specified number of tiles.\n\n#### `concat_tiles(tiles, n_tiles, image_size)`\n\nReconstructs an image from tiles and returns a single concatenated image.\n\n#### `to_tensor(x)`\n\nConverts a NumPy image array to a PyTorch tensor.\n\nThese functions facilitate image tiling, reconstruction, and conversion for preprocessing in machine learning tasks.","metadata":{}},{"cell_type":"code","source":"def get_tiles(img, tile_size=256, n_tiles=30, mode=0):\n    h, w, c = img.shape\n    pad_h = (tile_size - h % tile_size) % tile_size + ((tile_size * mode) // 2)\n    pad_w = (tile_size - w % tile_size) % tile_size + ((tile_size * mode) // 2)\n\n    img = np.pad(\n        img,\n        [[pad_h // 2, pad_h - pad_h // 2], [pad_w // 2, pad_w - pad_w // 2], [0, 0]],\n        constant_values=0,\n    )\n    img = img.reshape(\n        img.shape[0] // tile_size, tile_size, img.shape[1] // tile_size, tile_size, 3\n    )\n    img = img.transpose(0, 2, 1, 3, 4).reshape(-1, tile_size, tile_size, 3)\n    \n    idxs = np.argsort(img.reshape(img.shape[0], -1).sum(-1))\n    if len(img) < n_tiles:\n        img = np.pad(\n            img, [[0, n_tiles - len(img)], [0,0], [0,0], [0,0]], constant_values=255\n        )\n    # idxs = np.argsort(-img.reshape(img.shape[0], -1).sum(-1))[:n_tiles]\n    # print(type(idxs))\n    if idxs.shape[0]>n_tiles:\n        idxs = idxs[-n_tiles:]\n    img = img[idxs]\n    \n    return img\n\ndef concat_tiles(tiles, n_tiles, image_size):\n    idxes = list(range(n_tiles))\n    \n    n_row_tiles = int(np.sqrt(n_tiles))\n    img = np.zeros(\n        (image_size*n_row_tiles, image_size*n_row_tiles, 3), dtype=\"uint8\"\n    )\n    \n    for h in range(n_row_tiles):\n        for w in range(n_row_tiles):\n            i = h * n_row_tiles + w\n            if len(tiles) > idxes[i]:\n                this_img = tiles[idxes[i]]\n            else:\n                this_img = np.ones((image_size, image_size, 3), dtype=\"uint8\") * 255\n                \n            h1 = h * image_size\n            w1 = w * image_size\n            img[h1 : h1 + image_size, w1 : w1 + image_size] = this_img\n    return img\n\ndef to_tensor(x):\n    x = x.astype(\"float32\") / 255\n\n    return torch.from_numpy(x).permute(2, 0, 1)","metadata":{"execution":{"iopub.status.busy":"2023-11-22T22:55:29.751437Z","iopub.execute_input":"2023-11-22T22:55:29.7521Z","iopub.status.idle":"2023-11-22T22:55:29.765349Z","shell.execute_reply.started":"2023-11-22T22:55:29.752068Z","shell.execute_reply":"2023-11-22T22:55:29.764398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### UCBDataset Class\n\n#### `__init__(self, metadata_df, image_folder, transform=None)`\n\nInitializes a custom dataset with metadata and image folder paths. Allows for an optional transformation parameter.\n\n#### `__len__(self)`\n\nReturns the length of the dataset based on metadata.\n\n#### `__getitem__(self, idx)`\n\nFetches an image from the dataset using its index. Processes the image by dividing it into tiles, concatenating, and converting to a PyTorch tensor before returning.\n\nThis class enables data loading and preprocessing for image-based tasks, particularly designed for the UBC dataset.","metadata":{}},{"cell_type":"code","source":"class UCBDataset(Dataset):\n    def __init__(self, metadata_df, image_folder, transform=None):\n        self.metadata_df = metadata_df\n        self.image_folder = image_folder\n        self.transform = transform  # Use the provided transform\n    def __len__(self):\n        return len(self.metadata_df)\n    def __getitem__(self, idx):\n        image_ids = self.metadata_df.image_id[idx]  \n        image_name = os.path.join(self.image_folder, \"{}_thumbnail.png\".format(image_ids))\n        image = Image.open(image_name)\n        img = get_tiles(\n            np.array(image),\n            mode=0,\n        )\n        img = concat_tiles(\n            img, 16, 256\n        )\n        img = to_tensor(img)\n\n        return img","metadata":{"execution":{"iopub.status.busy":"2023-11-22T22:55:32.981226Z","iopub.execute_input":"2023-11-22T22:55:32.981978Z","iopub.status.idle":"2023-11-22T22:55:32.988827Z","shell.execute_reply.started":"2023-11-22T22:55:32.981943Z","shell.execute_reply":"2023-11-22T22:55:32.9879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This code sets up a testing data loader (test_loader) for the provided test data using a custom dataset (UCBDataset). It loads test data from a CSV file and creates a DataLoader to handle batches during testing.","metadata":{}},{"cell_type":"code","source":"test_df = pd.read_csv('/kaggle/input/UBC-OCEAN/test.csv')\ntest_dataset = UCBDataset(metadata_df=test_df, image_folder='/kaggle/input/UBC-OCEAN/test_thumbnails', transform=None)","metadata":{"execution":{"iopub.status.busy":"2023-11-22T22:55:35.026372Z","iopub.execute_input":"2023-11-22T22:55:35.026838Z","iopub.status.idle":"2023-11-22T22:55:35.047607Z","shell.execute_reply.started":"2023-11-22T22:55:35.026795Z","shell.execute_reply":"2023-11-22T22:55:35.046536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size = 4\ntest_loader = DataLoader(test_dataset, batch_size=batch_size)","metadata":{"execution":{"iopub.status.busy":"2023-11-22T22:55:36.99078Z","iopub.execute_input":"2023-11-22T22:55:36.991442Z","iopub.status.idle":"2023-11-22T22:55:36.995709Z","shell.execute_reply.started":"2023-11-22T22:55:36.991408Z","shell.execute_reply":"2023-11-22T22:55:36.994809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Testing with UCBDataset and DataLoader\n\nLoads the test dataset using `UCBDataset` and evaluates a model's predictions using a DataLoader.\n","metadata":{}},{"cell_type":"code","source":"model.eval()\nall_preds = []\nmodel.to(device)\nwith torch.no_grad():\n    for inputs in tqdm(test_loader, desc='testing', leave=False):\n        inputs = inputs.to(device)\n        outputs = model(inputs)\n        _, preds = torch.max(outputs, 1)\n        all_preds.extend(preds.cpu().numpy())\n    print('Testing done!')","metadata":{"execution":{"iopub.status.busy":"2023-11-22T22:56:17.789686Z","iopub.execute_input":"2023-11-22T22:56:17.790402Z","iopub.status.idle":"2023-11-22T22:56:22.482875Z","shell.execute_reply.started":"2023-11-22T22:56:17.790369Z","shell.execute_reply":"2023-11-22T22:56:22.481944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Decoding Predictions and Creating Submission DataFrame\n\nDecodes numerical predictions to their corresponding class labels using a mapping dictionary. Creates a submission DataFrame.creates a mapping (label_mapping) that associates numeric labels (0, 1, 2, 3, 4) with their corresponding string representations ('CC', 'EC', 'HGSC', 'LGSC', 'MC'). Then, it uses this mapping to decode the predicted numeric labels (all_preds) into their respective string labels (decoded_preds).","metadata":{}},{"cell_type":"code","source":"label_mapping = {0: 'CC', 1: 'EC', 2: 'HGSC', 3: 'LGSC', 4: 'MC'}  # Adjust this based on your column order\n\ndecoded_preds = [label_mapping[pred] for pred in all_preds]","metadata":{"execution":{"iopub.status.busy":"2023-11-22T22:56:24.962243Z","iopub.execute_input":"2023-11-22T22:56:24.962602Z","iopub.status.idle":"2023-11-22T22:56:24.968164Z","shell.execute_reply.started":"2023-11-22T22:56:24.962571Z","shell.execute_reply":"2023-11-22T22:56:24.966702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### SUBMIT!!!!!!!!!!!!!!!!!!!!!","metadata":{}},{"cell_type":"code","source":"sub_df = pd.DataFrame({'image_id':test_df.image_id, 'label':decoded_preds})","metadata":{"execution":{"iopub.status.busy":"2023-11-22T22:56:26.337542Z","iopub.execute_input":"2023-11-22T22:56:26.337911Z","iopub.status.idle":"2023-11-22T22:56:26.342822Z","shell.execute_reply.started":"2023-11-22T22:56:26.337879Z","shell.execute_reply":"2023-11-22T22:56:26.341884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-11-22T22:43:17.33358Z","iopub.execute_input":"2023-11-22T22:43:17.334861Z","iopub.status.idle":"2023-11-22T22:43:17.34742Z","shell.execute_reply.started":"2023-11-22T22:43:17.334817Z","shell.execute_reply":"2023-11-22T22:43:17.346326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}