{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport os\nimport cv2\n\n# from skimage import io\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import StratifiedKFold\n\nimport torch\nimport torch.nn as nn\nfrom torch.optim.lr_scheduler import ReduceLROnPlateau\nfrom torch.optim.lr_scheduler import OneCycleLR\n\nfrom torch.utils.data import Dataset\nfrom torch.utils.data import DataLoader\n\nimport torchmetrics\nfrom torchmetrics.classification import MulticlassF1Score\n\nfrom torchvision import transforms\nfrom torchvision import models\n\nimport pytorch_lightning as pl\nfrom pytorch_lightning import Trainer\nfrom pytorch_lightning.callbacks import ModelCheckpoint\nfrom pytorch_lightning.loggers import TensorBoardLogger\nfrom pytorch_lightning.callbacks import EarlyStopping\nfrom pytorch_lightning  import Trainer, seed_everything\nseed_everything(42, workers=True)\n\nfrom PIL import Image\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-10-31T18:15:41.881241Z","iopub.execute_input":"2023-10-31T18:15:41.881969Z","iopub.status.idle":"2023-10-31T18:15:56.572871Z","shell.execute_reply.started":"2023-10-31T18:15:41.881935Z","shell.execute_reply":"2023-10-31T18:15:56.571979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_csv(\"/kaggle/input/UBC-OCEAN/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-10-31T18:15:56.574793Z","iopub.execute_input":"2023-10-31T18:15:56.575142Z","iopub.status.idle":"2023-10-31T18:15:56.587527Z","shell.execute_reply.started":"2023-10-31T18:15:56.575109Z","shell.execute_reply":"2023-10-31T18:15:56.586773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head","metadata":{"execution":{"iopub.status.busy":"2023-10-31T18:15:56.588514Z","iopub.execute_input":"2023-10-31T18:15:56.588883Z","iopub.status.idle":"2023-10-31T18:15:56.605423Z","shell.execute_reply.started":"2023-10-31T18:15:56.588853Z","shell.execute_reply":"2023-10-31T18:15:56.604513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.tail","metadata":{"execution":{"iopub.status.busy":"2023-10-31T18:15:56.607446Z","iopub.execute_input":"2023-10-31T18:15:56.607713Z","iopub.status.idle":"2023-10-31T18:15:56.614718Z","shell.execute_reply.started":"2023-10-31T18:15:56.60769Z","shell.execute_reply":"2023-10-31T18:15:56.613869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TEST_THUMBNAILS = '/kaggle/input/UBC-OCEAN/test_thumbnails'\nTEST_IMAGES = '/kaggle/input/UBC-OCEAN/test_images'\n\ndef get_file_path(image_id):\n    if os.path.exists(f\"{TEST_THUMBNAILS}/{image_id}_thumbnail.png\"):\n        return f\"{TEST_THUMBNAILS}/{image_id}_thumbnail.png\"\n    else:\n        return f\"{TEST_IMAGES}/{image_id}.png\"\n    \n    \ntest_df['file_path'] = test_df['image_id'].apply(get_file_path)\ntest_df['label'] = 0","metadata":{"execution":{"iopub.status.busy":"2023-10-31T18:15:56.615882Z","iopub.execute_input":"2023-10-31T18:15:56.6162Z","iopub.status.idle":"2023-10-31T18:15:56.629148Z","shell.execute_reply.started":"2023-10-31T18:15:56.616169Z","shell.execute_reply":"2023-10-31T18:15:56.628403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mean = [0.48828688, 0.42932517, 0.49162089]\nstd = [0.41380908, 0.37492874, 0.41795654]","metadata":{"execution":{"iopub.status.busy":"2023-10-31T18:15:56.630157Z","iopub.execute_input":"2023-10-31T18:15:56.630431Z","iopub.status.idle":"2023-10-31T18:15:56.636476Z","shell.execute_reply.started":"2023-10-31T18:15:56.630405Z","shell.execute_reply":"2023-10-31T18:15:56.635778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class UBCDataset(Dataset):\n    def __init__(self, df, transforms=None):\n        self.df = df\n        self.transforms = transforms\n        \n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        img_path = self.df.iloc[idx]['file_path']  \n        label = self.df.iloc[idx]['label']\n        \n        img = Image.open(img_path).convert(\"RGB\")\n        \n        if self.transforms:\n            img = self.transforms(img)\n\n        return img, label","metadata":{"execution":{"iopub.status.busy":"2023-10-31T18:15:56.63731Z","iopub.execute_input":"2023-10-31T18:15:56.637844Z","iopub.status.idle":"2023-10-31T18:15:56.647485Z","shell.execute_reply.started":"2023-10-31T18:15:56.63782Z","shell.execute_reply":"2023-10-31T18:15:56.646616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_transform = transforms.Compose([\n    transforms.Resize((300, 300)),\n    transforms.ToTensor(),\n    transforms.Normalize(mean, std)\n])","metadata":{"execution":{"iopub.status.busy":"2023-10-31T18:15:56.648456Z","iopub.execute_input":"2023-10-31T18:15:56.64871Z","iopub.status.idle":"2023-10-31T18:15:56.660577Z","shell.execute_reply.started":"2023-10-31T18:15:56.648681Z","shell.execute_reply":"2023-10-31T18:15:56.659712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset = UBCDataset(df=test_df, transforms=test_transform)\ntest_loader = DataLoader(test_dataset, batch_size=32, shuffle=False, num_workers=2, pin_memory=True)","metadata":{"execution":{"iopub.status.busy":"2023-10-31T18:15:56.661733Z","iopub.execute_input":"2023-10-31T18:15:56.661992Z","iopub.status.idle":"2023-10-31T18:15:56.672218Z","shell.execute_reply.started":"2023-10-31T18:15:56.661969Z","shell.execute_reply":"2023-10-31T18:15:56.671305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import timm\n\nclass UBCModel(pl.LightningModule):\n\n    def __init__(self, steps_per_epoch):\n        super(UBCModel, self).__init__()\n        self.num_classes = 5\n        self.steps_per_epoch = steps_per_epoch\n\n        self.model = timm.create_model('tf_efficientnet_b3',\n                                       checkpoint_path='/kaggle/input/tf-efficientnet/pytorch/tf-efficientnet-b3/1/tf_efficientnet_b3_aa-84b4657e.pth')\n\n        \n        self.model.classifier= torch.nn.Linear(in_features=1536, out_features=self.num_classes, bias=True)\n        self.criterion = nn.CrossEntropyLoss()\n        self.f1 = MulticlassF1Score(num_classes=self.num_classes, average='macro')\n        self.accuracy = torchmetrics.Accuracy(num_classes=self.num_classes, task='multiclass')\n        self.precision = torchmetrics.Precision(average='macro', num_classes=self.num_classes, task='multiclass')\n        self.recall = torchmetrics.Recall(average='macro', num_classes=self.num_classes, task='multiclass')\n        \n    def forward(self, x):\n        x = self.model(x)\n        return x\n\n    def training_step(self, batch, batch_idx):\n        x, y = batch\n        y_pred = self(x)\n        loss = self.criterion(y_pred, y)\n        self.log('train_loss', loss)\n        self.log('train_f1', self.f1(y_pred, y))\n        return loss\n\n    def validation_step(self, batch, batch_idx):\n        x, y = batch\n        y_pred = self(x)\n        loss = self.criterion(y_pred, y)\n        self.log('val_loss', loss)\n        self.log('val_f1', self.f1(y_pred, y))\n        self.log('val_acc', self.accuracy(y_pred, y))\n        self.log('val_precision', self.precision(y_pred, y))\n        self.log('val_recall', self.recall(y_pred, y))\n    \n    def configure_optimizers(self):\n        optimizer = torch.optim.Adam(self.parameters(), lr=1e-5, weight_decay=1e-5)\n#         scheduler = ReduceLROnPlateau(optimizer, mode='min', factor=0.1, patience=2)\n#         return {\n#             'optimizer': optimizer,\n#             'lr_scheduler': {\n#                 'scheduler': scheduler,\n#                 'interval': 'epoch',\n#                 'monitor': 'val_f1',\n#                 'frequency': 1,\n#                 'strict': True,\n#             }\n#         }\n        scheduler = OneCycleLR(optimizer, max_lr=1e-3, steps_per_epoch=self.steps_per_epoch, epochs=self.trainer.max_epochs)\n        return {\n            'optimizer': optimizer,\n            'lr_scheduler': {\n                'scheduler': scheduler,\n                'interval': 'step',\n                'frequency': 1,\n                'strict': True,\n            }\n        }\n","metadata":{"execution":{"iopub.status.busy":"2023-10-31T18:15:56.674989Z","iopub.execute_input":"2023-10-31T18:15:56.675255Z","iopub.status.idle":"2023-10-31T18:15:57.021052Z","shell.execute_reply.started":"2023-10-31T18:15:56.675232Z","shell.execute_reply":"2023-10-31T18:15:57.020274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n\ncheckpoint_path1 = '/kaggle/input/sub-7-ovarian-cancer/epoch18-step646.ckpt'\ncheckpoint_path2 = '/kaggle/input/sub-7-ovarian-cancer/epoch29-step1020 (1).ckpt'\ncheckpoint_path3 = '/kaggle/input/sub-7-ovarian-cancer/epoch29-step1020 (2).ckpt'\ncheckpoint_path4 = '/kaggle/input/sub-7-ovarian-cancer/epoch29-step1020.ckpt'\n\nmodel1 = UBCModel.load_from_checkpoint(checkpoint_path1, map_location=device, steps_per_epoch=34)\nmodel2 = UBCModel.load_from_checkpoint(checkpoint_path2, map_location=device, steps_per_epoch=34)\nmodel3 = UBCModel.load_from_checkpoint(checkpoint_path3, map_location=device, steps_per_epoch=34)\nmodel4 = UBCModel.load_from_checkpoint(checkpoint_path3, map_location=device, steps_per_epoch=34)","metadata":{"execution":{"iopub.status.busy":"2023-10-31T18:15:57.022054Z","iopub.execute_input":"2023-10-31T18:15:57.022333Z","iopub.status.idle":"2023-10-31T18:16:11.243431Z","shell.execute_reply.started":"2023-10-31T18:15:57.022308Z","shell.execute_reply":"2023-10-31T18:16:11.242602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = []\nmodels = [model1, model2, model3, model4]\nnum_classes=5\nfor model in models:\n    model.eval()\n    model.to(device)\n\nwith torch.no_grad():\n    for image, label in test_loader:\n        image = image.to(device)\n\n        ensemble_outputs = torch.zeros([image.size(0), num_classes]).to(device)\n\n        for model in models:\n            outputs = model(image)\n            ensemble_outputs += torch.softmax(outputs, dim=1)\n            \n        ensemble_outputs /= len(models)\n\n        _, preds = torch.max(ensemble_outputs, 1)\n        predictions.extend(preds.detach().cpu().numpy())\n","metadata":{"execution":{"iopub.status.busy":"2023-10-31T18:16:11.244823Z","iopub.execute_input":"2023-10-31T18:16:11.245105Z","iopub.status.idle":"2023-10-31T18:16:16.640781Z","shell.execute_reply.started":"2023-10-31T18:16:11.24508Z","shell.execute_reply":"2023-10-31T18:16:16.639718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n# checkpoint_path = '/kaggle/input/sub-6-ovarian-cancer/epoch5-step138.ckpt'\n# model = UBCModel.load_from_checkpoint(checkpoint_path, map_location=device, steps_per_epoch=12)\n# model.eval()\n# predictions = []\n\n\n# # Make predictions on the test set\n# with torch.no_grad():\n#     for image, label in test_loader:\n#         image = image.to(device)\n#         outputs = model(image)\n        \n#         _, preds = torch.max(outputs, 1)\n#         predictions.extend(preds.detach().cpu().numpy())","metadata":{"execution":{"iopub.status.busy":"2023-10-31T18:16:16.642212Z","iopub.execute_input":"2023-10-31T18:16:16.642555Z","iopub.status.idle":"2023-10-31T18:16:16.647581Z","shell.execute_reply.started":"2023-10-31T18:16:16.642523Z","shell.execute_reply":"2023-10-31T18:16:16.646622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import joblib\n\n# Map the predictions back to their original labels\n# predictions = np.concatenate(predictions).flatten()\npredictions = np.array(predictions)\nencoder = joblib.load('/kaggle/input/ovarian-cancer-label-encoder-and-trained-models/ovarian-cancer/label_encoder.pkl')\npred_labels = encoder.inverse_transform(predictions)","metadata":{"execution":{"iopub.status.busy":"2023-10-31T18:16:16.648728Z","iopub.execute_input":"2023-10-31T18:16:16.64899Z","iopub.status.idle":"2023-10-31T18:16:16.676597Z","shell.execute_reply.started":"2023-10-31T18:16:16.648967Z","shell.execute_reply":"2023-10-31T18:16:16.675873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission = pd.read_csv(\"/kaggle/input/UBC-OCEAN/sample_submission.csv\")\nsample_submission['label'] = pred_labels\nsample_submission.to_csv('submission.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2023-10-31T18:16:16.677609Z","iopub.execute_input":"2023-10-31T18:16:16.677868Z","iopub.status.idle":"2023-10-31T18:16:16.692271Z","shell.execute_reply.started":"2023-10-31T18:16:16.677845Z","shell.execute_reply":"2023-10-31T18:16:16.691589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import necessary libraries\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score\n\n# Assuming you have your features in X and labels in y\n# Split your data into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Train your model on the training data (assuming you already have a trained model)\n# model.fit(X_train, y_train)\n\n# Make predictions on the test data\n# y_pred = model.predict(X_test)\n\n# Evaluate the model's performance on the test set\n# test_accuracy = accuracy_score(y_test, y_pred)\n# print(f\"Test Accuracy: {test_accuracy}\")\n\n# Assuming 'pred_labels' contains your model's predictions on the test data\n# You can add these predictions to the sample_submission DataFrame as you did before\nsample_submission = pd.read_csv(\"/kaggle/input/UBC-OCEAN/sample_submission.csv\")\nsample_submission['label'] = pred_labels\nsample_submission.to_csv('submission.csv', index=False)\n","metadata":{"execution":{"iopub.status.busy":"2023-10-31T18:16:16.693325Z","iopub.execute_input":"2023-10-31T18:16:16.693636Z","iopub.status.idle":"2023-10-31T18:16:17.724191Z","shell.execute_reply.started":"2023-10-31T18:16:16.693612Z","shell.execute_reply":"2023-10-31T18:16:17.722893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\n# Read the sample submission CSV file\nsample_submission = pd.read_csv(\"/kaggle/input/UBC-OCEAN/sample_submission.csv\")\n\n# Assign predicted labels to the 'label' column\nsample_submission['label'] = pred_labels\n\n# Save the modified DataFrame to a new CSV file\nsample_submission.to_csv('submission.csv', index=False)\n","metadata":{"execution":{"iopub.status.busy":"2023-10-31T18:16:17.725404Z","iopub.status.idle":"2023-10-31T18:16:17.725749Z","shell.execute_reply.started":"2023-10-31T18:16:17.72558Z","shell.execute_reply":"2023-10-31T18:16:17.725595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import necessary libraries\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score\n\n# Assuming you have your training data in X_train and corresponding labels in y_train\n# You should split your data into training and validation sets to assess model performance\nX_train, X_val, y_train, y_val = train_test_split(X_train, y_train, test_size=0.2, random_state=42)\n\n# Example 1: Decision Tree Classifier\n# Create a Decision Tree model\ndecision_tree_model = DecisionTreeClassifier()\n\n# Train the model on the training data\ndecision_tree_model.fit(X_train, y_train)\n\n# Make predictions on the validation set\ndecision_tree_predictions = decision_tree_model.predict(X_val)\n\n# Evaluate the model's performance\ndecision_tree_accuracy = accuracy_score(y_val, decision_tree_predictions)\nprint(f\"Decision Tree Classifier Accuracy: {decision_tree_accuracy}\")\n\n# Example 2: Random Forest Classifier\n# Create a Random Forest model\nrandom_forest_model = RandomForestClassifier(n_estimators=100, random_state=42)\n\n# Train the model on the training data\nrandom_forest_model.fit(X_train, y_train)\n\n# Make predictions on the validation set\nrandom_forest_predictions = random_forest_model.predict(X_val)\n\n# Evaluate the model's performance\nrandom_forest_accuracy = accuracy_score(y_val, random_forest_predictions)\nprint(f\"Random Forest Classifier Accuracy: {random_forest_accuracy}\")\n","metadata":{"execution":{"iopub.status.busy":"2023-10-31T18:16:17.726557Z","iopub.status.idle":"2023-10-31T18:16:17.726906Z","shell.execute_reply.started":"2023-10-31T18:16:17.726745Z","shell.execute_reply":"2023-10-31T18:16:17.726761Z"},"trusted":true},"execution_count":null,"outputs":[]}]}