{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"},{"sourceId":7030514,"sourceType":"datasetVersion","datasetId":4043863},{"sourceId":89702335,"sourceType":"kernelVersion"}],"dockerImageVersionId":30587,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport random\nfrom sklearn.model_selection import train_test_split\nimport numpy as np\nfrom torch.utils.data import Dataset, DataLoader\nfrom tqdm import tqdm\nfrom sklearn.metrics import balanced_accuracy_score\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nfrom torchvision import transforms\nimport torch.optim as optim\nimport torch.nn as nn\nimport torch.nn.functional as F\n\nimport torchvision\nfrom PIL import Image\n\nimport torch\nimport torchvision.transforms as transforms\nfrom torchvision.models import resnet18\nfrom torchvision.datasets import ImageFolder\nfrom torch.utils.data import DataLoader\nimport numpy as np\nimport xgboost as xgb","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-11-22T23:52:43.443776Z","iopub.execute_input":"2023-11-22T23:52:43.444319Z","iopub.status.idle":"2023-11-22T23:52:43.454051Z","shell.execute_reply.started":"2023-11-22T23:52:43.444283Z","shell.execute_reply":"2023-11-22T23:52:43.452439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### XGBoost Model Loading\n\nLoaded an XGBoost classifier model trained previously and saved in a Kaggle dataset.","metadata":{}},{"cell_type":"code","source":"loaded_model = xgb.XGBClassifier()\nloaded_model.load_model('/kaggle/input/xgb-resnet18/xgb_best_model.json')","metadata":{"execution":{"iopub.status.busy":"2023-11-22T23:52:45.762929Z","iopub.execute_input":"2023-11-22T23:52:45.763414Z","iopub.status.idle":"2023-11-22T23:52:45.917955Z","shell.execute_reply.started":"2023-11-22T23:52:45.763333Z","shell.execute_reply":"2023-11-22T23:52:45.916101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Image Feature Extraction and Saving\n\nThis code snippet contains utility functions and a custom dataset for extracting features from images using a pre-trained model and storing them.\n\n#### `extract_features(model, dataloader)`\n\nThis function extracts features from images using a pre-trained model and a dataloader.\n\n#### `save_features(filename, features, labels)`\n\nThis function saves the extracted features and corresponding labels to a `.npz` file.\n\n#### `CustomCancerDataset`\n\nA custom dataset class designed for a cancer dataset, fetching images and applying transformations.","metadata":{}},{"cell_type":"code","source":"def extract_features(model, dataloader):\n    features = []\n    labels = []\n    model.eval()\n    with torch.no_grad():\n        for inputs in tqdm(dataloader, desc='Extracting:', leave=False):\n            inputs = inputs.to(device)\n            outputs = model(inputs)\n            features.extend(outputs.cpu().numpy())\n    return np.array(features)\n\ndef save_features(filename, features, labels):\n    np.savez(filename, features=features, labels=labels)\n\nclass CustomCancerDataset(Dataset):\n    def __init__(self, metadata_df, image_folder, transform=None):\n        self.metadata_df = metadata_df\n        self.image_folder = image_folder\n        self.transform = transform  # Use the provided transform\n\n    def __len__(self):\n        return len(self.metadata_df)\n\n    def __getitem__(self, idx):\n        image_ids = self.metadata_df.image_id[idx]  \n        image_name = os.path.join(self.image_folder, \"{}_thumbnail.png\".format(image_ids))\n        image = Image.open(image_name).convert('RGB')\n        if self.transform:\n            image = self.transform(image)\n\n        return image","metadata":{"execution":{"iopub.status.busy":"2023-11-22T23:52:48.189722Z","iopub.execute_input":"2023-11-22T23:52:48.190104Z","iopub.status.idle":"2023-11-22T23:52:48.204264Z","shell.execute_reply.started":"2023-11-22T23:52:48.190074Z","shell.execute_reply":"2023-11-22T23:52:48.202206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Test Dataset Setup and Feature Extraction\n\nThis code snippet prepares a test dataset from a CSV file and associated images, uses a ResNet18 model for feature extraction, and sets up a DataLoader for processing the test dataset.\n\n#### `test_split = pd.read_csv('/kaggle/input/UBC-OCEAN/test.csv')`\n\nLoads test dataset information from a CSV file named 'test.csv'.\n\n#### `test_transforms`\n\nDefines transformations (resize and conversion to tensor) to be applied to the test dataset images.\n\n#### `test_ucb_dataset`\n\nCreates a custom test dataset using the loaded CSV file and image folder with specified transformations.\n\n#### `train_loader = DataLoader(test_ucb_dataset, batch_size=batch_size, shuffle=True)`\n\nCreates a DataLoader for the test dataset to facilitate batch-wise processing.\n\n#### `resnet = resnet18(pretrained=False)`\n\nInitializes a ResNet18 model (pre-trained weights not used).\n\n#### `features = extract_features(resnet, train_loader)`\n\nAttempts to extract features from the test dataset using the ResNet18 model and the previously defined DataLoader.","metadata":{}},{"cell_type":"code","source":"test_split = pd.read_csv('/kaggle/input/UBC-OCEAN/test.csv')\n\ntest_transforms = transforms.Compose([\n    transforms.Resize((256, 256)),\n    transforms.ToTensor(),\n])\n\ntest_ucb_dataset = CustomCancerDataset(test_split, image_folder='/kaggle/input/UBC-OCEAN/test_thumbnails',\n                                transform=test_transforms)\n\nbatch_size = 16\ntrain_loader = DataLoader(test_ucb_dataset, batch_size=batch_size, shuffle=True)\n# val_loader = DataLoader(val_ucb_dataset, batch_size=batch_size)\n\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nresnet = resnet18(pretrained=False)\nresnet = resnet.to(device)\nresnet = torch.nn.Sequential(*(list(resnet.children())[:-1])) \n\nfeatures = extract_features(resnet, train_loader)","metadata":{"execution":{"iopub.status.busy":"2023-11-22T23:52:51.517158Z","iopub.execute_input":"2023-11-22T23:52:51.517614Z","iopub.status.idle":"2023-11-22T23:52:52.084692Z","shell.execute_reply.started":"2023-11-22T23:52:51.517579Z","shell.execute_reply":"2023-11-22T23:52:52.083389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Making Predictions using XGBoost Model\n\nThis code segment reshapes the extracted features and uses a pre-loaded XGBoost model to make predictions.\n\n#### `reshaped_features = features.reshape(features.shape[0], -1)`\n\nReshapes the extracted features into a 2D array to prepare them for prediction. The features are transformed to match the format expected by the XGBoost model.\n\n#### `preds = loaded_model.predict(reshaped_features)`\n\nUses a pre-loaded XGBoost model (`loaded_model`) to predict outcomes based on the reshaped features obtained from the image data.","metadata":{}},{"cell_type":"code","source":"reshaped_features = features.reshape(features.shape[0], -1)","metadata":{"execution":{"iopub.status.busy":"2023-11-22T23:46:01.969195Z","iopub.execute_input":"2023-11-22T23:46:01.969759Z","iopub.status.idle":"2023-11-22T23:46:01.975684Z","shell.execute_reply.started":"2023-11-22T23:46:01.969688Z","shell.execute_reply":"2023-11-22T23:46:01.974465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = loaded_model.predict(reshaped_features)","metadata":{"execution":{"iopub.status.busy":"2023-11-22T23:47:28.874811Z","iopub.execute_input":"2023-11-22T23:47:28.875227Z","iopub.status.idle":"2023-11-22T23:47:28.881319Z","shell.execute_reply.started":"2023-11-22T23:47:28.875198Z","shell.execute_reply":"2023-11-22T23:47:28.88038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Mapping Predicted Labels to Class Names\n\nThis code snippet maps the numerical labels predicted by the model to their corresponding class names and updates the DataFrame accordingly.\n\n#### `label_mapping = {0:'HGSC',1:'LGSC',2:'EC',3:'CC',4:'MC'}`\n\nDefines a mapping dictionary that associates numerical labels (0, 1, 2, 3, 4) with their respective class names ('HGSC', 'LGSC', 'EC', 'CC', 'MC').\n\n#### `sub_df['label'] = sub_df['label'].map(label_mapping)`\n\nMaps the numerical labels in the 'label' column of the DataFrame (`sub_df`) to their corresponding class names using the `label_mapping` dictionary.","metadata":{}},{"cell_type":"code","source":"sub_df = pd.DataFrame({'image_id': test_split.image_id, 'label': preds})\nlabel_mapping = {0:'HGSC',1:'LGSC',2:'EC',3:'CC',4:'MC'}\nsub_df['label'] = sub_df['label'].map(label_mapping)","metadata":{"execution":{"iopub.status.busy":"2023-11-22T23:54:02.209561Z","iopub.execute_input":"2023-11-22T23:54:02.210085Z","iopub.status.idle":"2023-11-22T23:54:02.231153Z","shell.execute_reply.started":"2023-11-22T23:54:02.21005Z","shell.execute_reply":"2023-11-22T23:54:02.229628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-11-22T23:54:04.203265Z","iopub.execute_input":"2023-11-22T23:54:04.203842Z","iopub.status.idle":"2023-11-22T23:54:04.214762Z","shell.execute_reply.started":"2023-11-22T23:54:04.203798Z","shell.execute_reply":"2023-11-22T23:54:04.213238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}