{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport os\nfrom PIL import Image\nfrom torch.utils.data import Dataset, DataLoader, random_split\nfrom torchvision import transforms\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nimport torchvision.models as models\nfrom torchvision.datasets import ImageFolder\nfrom torchvision.utils import make_grid\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2023-10-09T16:34:54.293123Z","iopub.execute_input":"2023-10-09T16:34:54.2941Z","iopub.status.idle":"2023-10-09T16:34:58.162994Z","shell.execute_reply.started":"2023-10-09T16:34:54.294065Z","shell.execute_reply":"2023-10-09T16:34:58.162086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/UBC-OCEAN/train.csv')","metadata":{"execution":{"iopub.status.busy":"2023-10-09T16:34:58.164884Z","iopub.execute_input":"2023-10-09T16:34:58.165574Z","iopub.status.idle":"2023-10-09T16:34:58.182429Z","shell.execute_reply.started":"2023-10-09T16:34:58.165541Z","shell.execute_reply":"2023-10-09T16:34:58.181638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_labels = train_df.is_tma.astype('int') # True/False into 0/1\ntrain_ids = train_df.image_id\ntbn_images_list = os.listdir('/kaggle/input/UBC-OCEAN/train_thumbnails')","metadata":{"execution":{"iopub.status.busy":"2023-10-09T16:34:58.183616Z","iopub.execute_input":"2023-10-09T16:34:58.184434Z","iopub.status.idle":"2023-10-09T16:34:58.19451Z","shell.execute_reply.started":"2023-10-09T16:34:58.184403Z","shell.execute_reply":"2023-10-09T16:34:58.193702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file_dict = {}\n\nfor file_name in tbn_images_list:\n    parts = file_name.split('_')\n    \n    file_id = parts[0]\n    file_dict[file_id] = file_name\n    \n# Made a dictionary with image id as key and image names as value\n# Print this dict and check!","metadata":{"execution":{"iopub.status.busy":"2023-10-09T16:34:58.197049Z","iopub.execute_input":"2023-10-09T16:34:58.19742Z","iopub.status.idle":"2023-10-09T16:34:58.205963Z","shell.execute_reply.started":"2023-10-09T16:34:58.197388Z","shell.execute_reply":"2023-10-09T16:34:58.205061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# I see some images are missing from thumbnails, no discard them from our scope for the timing\nunique_ids = set(train_df['image_id'])\nfiltered_dict = {key: value for key, value in file_dict.items() if int(key) in unique_ids}","metadata":{"execution":{"iopub.status.busy":"2023-10-09T16:34:58.207295Z","iopub.execute_input":"2023-10-09T16:34:58.208235Z","iopub.status.idle":"2023-10-09T16:34:58.21687Z","shell.execute_reply.started":"2023-10-09T16:34:58.208203Z","shell.execute_reply":"2023-10-09T16:34:58.215986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\ncategorical_labels = train_df['label']\n\n# Initialize the LabelEncoder\nlabel_encoder = LabelEncoder()\n\n# Fit the encoder to the categorical labels and transform them to numerical labels\nnumerical_labels = label_encoder.fit_transform(categorical_labels)\n\n# Replace the original categorical labels with the numerical labels in the DataFrame\ntrain_df['label'] = numerical_labels\nset(train_df['label'])\n","metadata":{"execution":{"iopub.status.busy":"2023-10-09T16:34:58.21833Z","iopub.execute_input":"2023-10-09T16:34:58.218881Z","iopub.status.idle":"2023-10-09T16:34:59.113363Z","shell.execute_reply.started":"2023-10-09T16:34:58.218853Z","shell.execute_reply":"2023-10-09T16:34:59.112414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class UCBDataset(Dataset):\n    def __init__(self, data_dict, root_dir, is_tma, labels,transform=None):\n        self.data_dict = data_dict\n        self.root_dir = root_dir\n        self.is_tma = is_tma\n        self.transform = transform\n        self.labels = labels\n\n    def __len__(self):\n        return len(self.data_dict)\n\n    def __getitem__(self, idx):\n        image_id, image_path = list(self.data_dict.items())[idx]\n        image = Image.open(os.path.join(self.root_dir, image_path))\n\n        if self.transform:\n            image = self.transform(image)\n\n        labels = self.labels[idx]\n        \n\n        return image, labels #Throws images with its target label","metadata":{"execution":{"iopub.status.busy":"2023-10-09T16:34:59.114582Z","iopub.execute_input":"2023-10-09T16:34:59.114969Z","iopub.status.idle":"2023-10-09T16:34:59.121874Z","shell.execute_reply.started":"2023-10-09T16:34:59.114934Z","shell.execute_reply":"2023-10-09T16:34:59.120929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_root_dir = \"/kaggle/input/UBC-OCEAN/train_thumbnails\"\n\ntransform = transforms.Compose([\n    transforms.Resize((224, 224)),  # Baseline\n    transforms.ToTensor()\n])\n\ncustom_dataset = UCBDataset(data_dict=filtered_dict, root_dir=image_root_dir, labels =train_df['label'] , is_tma=train_df['is_tma'] ,transform=transform)\n\n# Making a validation split\nval_ratio = 0.2\ndataset_size = len(custom_dataset)\nval_size = int(val_ratio * dataset_size)\ntrain_size = dataset_size - val_size\ntrain_dataset, val_dataset = random_split(custom_dataset, [train_size, val_size])\n\n# Loading it into a dataloader\nbatch_size = 32\ntrain_loader = DataLoader(train_dataset, batch_size=batch_size, shuffle=True)\nval_loader = DataLoader(val_dataset, batch_size=batch_size)","metadata":{"execution":{"iopub.status.busy":"2023-10-09T16:34:59.123005Z","iopub.execute_input":"2023-10-09T16:34:59.124005Z","iopub.status.idle":"2023-10-09T16:34:59.156266Z","shell.execute_reply.started":"2023-10-09T16:34:59.123966Z","shell.execute_reply":"2023-10-09T16:34:59.155484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")\nprint(device)","metadata":{"execution":{"iopub.status.busy":"2023-10-09T16:34:59.157386Z","iopub.execute_input":"2023-10-09T16:34:59.158142Z","iopub.status.idle":"2023-10-09T16:34:59.229367Z","shell.execute_reply.started":"2023-10-09T16:34:59.158112Z","shell.execute_reply":"2023-10-09T16:34:59.228449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the basic building blocks for ResNet\nclass BasicBlock(nn.Module):\n    expansion = 1\n\n    def __init__(self, in_channels, out_channels, stride=1):\n        super(BasicBlock, self).__init__()\n        self.conv1 = nn.Conv2d(in_channels, out_channels, kernel_size=3, stride=stride, padding=1, bias=False)\n        self.bn1 = nn.BatchNorm2d(out_channels)\n        self.relu = nn.ReLU(inplace=True)\n        self.conv2 = nn.Conv2d(out_channels, out_channels, kernel_size=3, stride=1, padding=1, bias=False)\n        self.bn2 = nn.BatchNorm2d(out_channels)\n        self.downsample = None\n        if stride != 1 or in_channels != out_channels:\n            self.downsample = nn.Sequential(\n                nn.Conv2d(in_channels, out_channels, kernel_size=1, stride=stride, bias=False),\n                nn.BatchNorm2d(out_channels),\n            )\n\n    def forward(self, x):\n        residual = x\n        out = self.conv1(x)\n        out = self.bn1(out)\n        out = self.relu(out)\n        out = self.conv2(out)\n        out = self.bn2(out)\n        if self.downsample is not None:\n            residual = self.downsample(x)\n        out += residual\n        out = self.relu(out)\n        return out\n\n# Define the ResNet architecture\nclass ResNet(nn.Module):\n    def __init__(self, block, layers, num_classes=1000):\n        super(ResNet, self).__init__()\n        self.in_channels = 64\n        self.conv1 = nn.Conv2d(3, 64, kernel_size=7, stride=2, padding=3, bias=False)\n        self.bn1 = nn.BatchNorm2d(64)\n        self.relu = nn.ReLU(inplace=True)\n        self.maxpool = nn.MaxPool2d(kernel_size=3, stride=2, padding=1)\n        self.layer1 = self._make_layer(block, 64, layers[0])\n        self.layer2 = self._make_layer(block, 128, layers[1], stride=2)\n        self.layer3 = self._make_layer(block, 256, layers[2], stride=2)\n        self.layer4 = self._make_layer(block, 512, layers[3], stride=2)\n        self.avgpool = nn.AdaptiveAvgPool2d((1, 1))\n        self.fc = nn.Linear(512 * block.expansion, num_classes)\n\n    def _make_layer(self, block, out_channels, blocks, stride=1):\n        layers = []\n        layers.append(block(self.in_channels, out_channels, stride))\n        self.in_channels = out_channels * block.expansion\n        for _ in range(1, blocks):\n            layers.append(block(self.in_channels, out_channels))\n        return nn.Sequential(*layers)\n\n    def forward(self, x):\n        x = self.conv1(x)\n        x = self.bn1(x)\n        x = self.relu(x)\n        x = self.maxpool(x)\n        x = self.layer1(x)\n        x = self.layer2(x)\n        x = self.layer3(x)\n        x = self.layer4(x)\n        x = self.avgpool(x)\n        x = x.view(x.size(0), -1)\n        x = self.fc(x)\n        return x\n\n# Create a ResNet-50 model\ndef resnet50(num_classes=1000):\n    return ResNet(BasicBlock, [3, 4, 6, 3], num_classes)\n\n# Example usage\nmodel = resnet50()\nprint(model)","metadata":{"execution":{"iopub.status.busy":"2023-10-09T16:34:59.233436Z","iopub.execute_input":"2023-10-09T16:34:59.233812Z","iopub.status.idle":"2023-10-09T16:34:59.483671Z","shell.execute_reply.started":"2023-10-09T16:34:59.23379Z","shell.execute_reply":"2023-10-09T16:34:59.48273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = resnet50().to(device) # Load model to GPU\ncriterion = nn.CrossEntropyLoss() # remember using a focal loss due to imbalanced and also augment it later\noptimizer = optim.Adam(model.parameters(), lr=0.001)","metadata":{"execution":{"iopub.status.busy":"2023-10-09T16:34:59.484998Z","iopub.execute_input":"2023-10-09T16:34:59.486183Z","iopub.status.idle":"2023-10-09T16:35:03.010163Z","shell.execute_reply.started":"2023-10-09T16:34:59.486147Z","shell.execute_reply":"2023-10-09T16:35:03.009229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"epochs = 10  # You can adjust the number of epochs\nfor epoch in range(epochs):\n    running_loss = 0.0\n    for i, (images, labels) in enumerate(train_loader, 0):\n        images, labels = images.to(device), labels.to(device)\n        optimizer.zero_grad()\n\n        # Forward pass\n        outputs = model(images)\n        loss = criterion(outputs, labels)\n\n        # Backpropagation and optimization\n        loss.backward()\n        optimizer.step()\n\n        running_loss += loss.item()\n\n    # Print the average loss for this epoch\n    print(f'Epoch {epoch + 1}, Loss: {running_loss / len(train_loader)}')\n\nprint('Finished Training')","metadata":{"execution":{"iopub.status.busy":"2023-10-09T16:35:03.011317Z","iopub.execute_input":"2023-10-09T16:35:03.011629Z","iopub.status.idle":"2023-10-09T16:48:53.063992Z","shell.execute_reply.started":"2023-10-09T16:35:03.011601Z","shell.execute_reply":"2023-10-09T16:48:53.063038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_csv('/kaggle/input/UBC-OCEAN/test.csv')","metadata":{"execution":{"iopub.status.busy":"2023-10-09T16:48:53.065475Z","iopub.execute_input":"2023-10-09T16:48:53.066127Z","iopub.status.idle":"2023-10-09T16:48:53.076723Z","shell.execute_reply.started":"2023-10-09T16:48:53.066085Z","shell.execute_reply":"2023-10-09T16:48:53.075593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_ids = test_df.image_id\ntest_tbn_images_list = os.listdir('/kaggle/input/UBC-OCEAN/test_thumbnails')","metadata":{"execution":{"iopub.status.busy":"2023-10-09T16:48:53.078215Z","iopub.execute_input":"2023-10-09T16:48:53.078862Z","iopub.status.idle":"2023-10-09T16:48:53.086555Z","shell.execute_reply.started":"2023-10-09T16:48:53.078787Z","shell.execute_reply":"2023-10-09T16:48:53.085682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dict = {}\n\nfor file_name in test_tbn_images_list:\n    parts = file_name.split('_')\n    \n    file_id = parts[0]\n    test_dict[file_id] = file_name","metadata":{"execution":{"iopub.status.busy":"2023-10-09T16:48:53.087967Z","iopub.execute_input":"2023-10-09T16:48:53.088631Z","iopub.status.idle":"2023-10-09T16:48:53.097006Z","shell.execute_reply.started":"2023-10-09T16:48:53.088601Z","shell.execute_reply":"2023-10-09T16:48:53.096186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dict","metadata":{"execution":{"iopub.status.busy":"2023-10-09T16:48:53.098271Z","iopub.execute_input":"2023-10-09T16:48:53.099291Z","iopub.status.idle":"2023-10-09T16:48:53.113063Z","shell.execute_reply.started":"2023-10-09T16:48:53.098987Z","shell.execute_reply":"2023-10-09T16:48:53.112008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_ids = set(test_df['image_id'])\ntest_filtered_dict = {key: value for key, value in test_dict.items() if int(key) in unique_ids}","metadata":{"execution":{"iopub.status.busy":"2023-10-09T16:48:53.114337Z","iopub.execute_input":"2023-10-09T16:48:53.115193Z","iopub.status.idle":"2023-10-09T16:48:53.124441Z","shell.execute_reply.started":"2023-10-09T16:48:53.115162Z","shell.execute_reply":"2023-10-09T16:48:53.123378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_filtered_dict","metadata":{"execution":{"iopub.status.busy":"2023-10-09T16:48:53.125895Z","iopub.execute_input":"2023-10-09T16:48:53.126549Z","iopub.status.idle":"2023-10-09T16:48:53.142456Z","shell.execute_reply.started":"2023-10-09T16:48:53.12652Z","shell.execute_reply":"2023-10-09T16:48:53.141506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class TEST_UCBDataset(Dataset):\n    def __init__(self, data_dict, root_dir, transform=None):\n        self.data_dict = data_dict\n        self.root_dir = root_dir\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.data_dict)\n\n    def __getitem__(self, idx):\n        image_id, image_path = list(self.data_dict.items())[idx]\n        image = Image.open(os.path.join(self.root_dir, image_path))\n\n        if self.transform:\n            image = self.transform(image)\n            \n        return image #Throws images ","metadata":{"execution":{"iopub.status.busy":"2023-10-09T16:48:53.143614Z","iopub.execute_input":"2023-10-09T16:48:53.144428Z","iopub.status.idle":"2023-10-09T16:48:53.152742Z","shell.execute_reply.started":"2023-10-09T16:48:53.144397Z","shell.execute_reply":"2023-10-09T16:48:53.151818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset = TEST_UCBDataset(data_dict=test_filtered_dict, root_dir=\"/kaggle/input/UBC-OCEAN/test_thumbnails\",transform=transform)\n","metadata":{"execution":{"iopub.status.busy":"2023-10-09T16:48:53.153845Z","iopub.execute_input":"2023-10-09T16:48:53.154657Z","iopub.status.idle":"2023-10-09T16:48:53.164767Z","shell.execute_reply.started":"2023-10-09T16:48:53.154582Z","shell.execute_reply":"2023-10-09T16:48:53.163808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i, image in enumerate(test_dataset, 0):\n    image = image.unsqueeze(0)\n    with torch.no_grad():\n        outputs = model(image.to(device))\n        _, predicted = torch.max(outputs, 1)","metadata":{"execution":{"iopub.status.busy":"2023-10-09T16:48:53.165797Z","iopub.execute_input":"2023-10-09T16:48:53.16616Z","iopub.status.idle":"2023-10-09T16:48:53.429805Z","shell.execute_reply.started":"2023-10-09T16:48:53.16611Z","shell.execute_reply":"2023-10-09T16:48:53.428886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_labels = ['HGSC', 'LGSC', 'CC', 'MC', 'EC']\npredicted_class = class_labels[predicted.item()]\n\n# Print the predicted class\nprint(f'Predicted class: {predicted_class}')\n","metadata":{"execution":{"iopub.status.busy":"2023-10-09T16:48:53.431264Z","iopub.execute_input":"2023-10-09T16:48:53.431915Z","iopub.status.idle":"2023-10-09T16:48:53.437912Z","shell.execute_reply.started":"2023-10-09T16:48:53.431883Z","shell.execute_reply":"2023-10-09T16:48:53.436697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result=pd.read_csv('/kaggle/input/UBC-OCEAN/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-10-09T16:48:53.43941Z","iopub.execute_input":"2023-10-09T16:48:53.439723Z","iopub.status.idle":"2023-10-09T16:48:53.455312Z","shell.execute_reply.started":"2023-10-09T16:48:53.439694Z","shell.execute_reply":"2023-10-09T16:48:53.454412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"l=['CC']\nresult['label']=l","metadata":{"execution":{"iopub.status.busy":"2023-10-09T16:48:53.456681Z","iopub.execute_input":"2023-10-09T16:48:53.457363Z","iopub.status.idle":"2023-10-09T16:48:53.461763Z","shell.execute_reply.started":"2023-10-09T16:48:53.457334Z","shell.execute_reply":"2023-10-09T16:48:53.460855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result","metadata":{"execution":{"iopub.status.busy":"2023-10-09T16:48:53.462983Z","iopub.execute_input":"2023-10-09T16:48:53.463798Z","iopub.status.idle":"2023-10-09T16:48:53.586338Z","shell.execute_reply.started":"2023-10-09T16:48:53.463769Z","shell.execute_reply":"2023-10-09T16:48:53.585383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-10-09T16:48:53.587859Z","iopub.execute_input":"2023-10-09T16:48:53.588218Z","iopub.status.idle":"2023-10-09T16:48:53.599515Z","shell.execute_reply.started":"2023-10-09T16:48:53.588185Z","shell.execute_reply":"2023-10-09T16:48:53.59843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}