{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Preprocessing","metadata":{}},{"cell_type":"code","source":"# Download Ngrok to tunnel the tensorboard port to an external port\n!wget https://bin.equinox.io/c/4VmDzA7iaHb/ngrok-stable-linux-amd64.zip\n!unzip ngrok-stable-linux-amd64.zip\n\n# Run tensorboard as well as Ngrox (for tunneling as non-blocking processes)\nimport os\nimport multiprocessing\n\npool = multiprocessing.Pool(processes = 10)\nresults_of_processes = [pool.apply_async(os.system, args=(cmd, ), callback = None )\n                        for cmd in [\n                        f\"tensorboard --logdir /kaggle/working/logs --host 0.0.0.0 --port 6006 &\",\n                        \"./ngrok http 6006 &\"\n                        ]]","metadata":{"execution":{"iopub.status.busy":"2023-10-12T15:29:01.070632Z","iopub.execute_input":"2023-10-12T15:29:01.07118Z","iopub.status.idle":"2023-10-12T15:29:05.234118Z","shell.execute_reply.started":"2023-10-12T15:29:01.071149Z","shell.execute_reply":"2023-10-12T15:29:05.230314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! curl -s http://localhost:4040/api/tunnels | python3 -c \\\n    \"import sys, json; print(json.load(sys.stdin)['tunnels'][0]['public_url'])\"","metadata":{"execution":{"iopub.status.busy":"2023-10-12T15:29:05.239274Z","iopub.execute_input":"2023-10-12T15:29:05.246753Z","iopub.status.idle":"2023-10-12T15:29:06.803227Z","shell.execute_reply.started":"2023-10-12T15:29:05.246669Z","shell.execute_reply":"2023-10-12T15:29:06.801496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport os\nimport seaborn as sns\nimport cv2\nimport random\nimport os\nimport glob\nimport torch\nimport torch.nn as nn\nimport torchmetrics\n\n\nfrom skimage import io\nfrom PIL import Image\nfrom sklearn.model_selection import train_test_split\n\nfrom torch.utils.data import Dataset\nfrom torch.utils.data import DataLoader\n\nfrom torchvision import transforms\n\nimport pytorch_lightning as pl\nfrom pytorch_lightning import Trainer\nfrom pytorch_lightning.loggers import TensorBoardLogger\nfrom pytorch_lightning.callbacks import ModelCheckpoint\n","metadata":{"execution":{"iopub.status.busy":"2023-10-12T15:33:40.307626Z","iopub.execute_input":"2023-10-12T15:33:40.308025Z","iopub.status.idle":"2023-10-12T15:33:40.316038Z","shell.execute_reply.started":"2023-10-12T15:33:40.307997Z","shell.execute_reply":"2023-10-12T15:33:40.314521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.read_csv('/kaggle/input/UBC-OCEAN/train.csv').shape","metadata":{"execution":{"iopub.status.busy":"2023-10-12T15:29:24.563791Z","iopub.execute_input":"2023-10-12T15:29:24.565438Z","iopub.status.idle":"2023-10-12T15:29:24.590449Z","shell.execute_reply.started":"2023-10-12T15:29:24.565332Z","shell.execute_reply":"2023-10-12T15:29:24.589239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.read_csv('/kaggle/input/UBC-OCEAN/train.csv').head()","metadata":{"execution":{"iopub.status.busy":"2023-10-12T15:29:24.5928Z","iopub.execute_input":"2023-10-12T15:29:24.593704Z","iopub.status.idle":"2023-10-12T15:29:24.626518Z","shell.execute_reply.started":"2023-10-12T15:29:24.59366Z","shell.execute_reply":"2023-10-12T15:29:24.625177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/UBC-OCEAN/train.csv\")\nprint(\"Unique Labels and Counts:\")\nprint(df['label'].value_counts())","metadata":{"execution":{"iopub.status.busy":"2023-10-12T15:29:24.627901Z","iopub.execute_input":"2023-10-12T15:29:24.628221Z","iopub.status.idle":"2023-10-12T15:29:24.647494Z","shell.execute_reply.started":"2023-10-12T15:29:24.628195Z","shell.execute_reply":"2023-10-12T15:29:24.645246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plt.figure(figsize=(16,24))\n# path = \"/kaggle/input/UBC-OCEAN/train_thumbnails\"\n# j=1\n# for img, lb in zip(train_df_no_tma['image_id_path'][:24],train_df_no_tma['label'][:24]):\n#     plt.subplot(6,4,j)\n#     path = os.path.join(\"/kaggle/input/UBC-OCEAN/train_thumbnails/\",img)\n#     image = plt.imread(path)\n#     image = plt.imshow(image)\n#     plt.title(f\"Label:{lb}\")\n#     j+=1","metadata":{"execution":{"iopub.status.busy":"2023-10-12T15:29:24.649793Z","iopub.execute_input":"2023-10-12T15:29:24.650303Z","iopub.status.idle":"2023-10-12T15:29:24.656143Z","shell.execute_reply.started":"2023-10-12T15:29:24.650252Z","shell.execute_reply":"2023-10-12T15:29:24.654798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plt.figure(figsize=(16,28))\n# path = \"/kaggle/input/UBC-OCEAN/train_images\"\n# j=1\n# for img, lb in zip(train_df_tma['image_id_path'],train_df_tma['label']):\n#     plt.subplot(7,4,j)\n#     path = os.path.join(\"/kaggle/input/UBC-OCEAN/train_images\",img)\n#     image = plt.imread(path)\n#     image = plt.imshow(image)\n#     plt.title(f\"Label:{lb}\")\n#     j+=1","metadata":{"execution":{"iopub.status.busy":"2023-10-12T15:29:24.65799Z","iopub.execute_input":"2023-10-12T15:29:24.658492Z","iopub.status.idle":"2023-10-12T15:29:24.678979Z","shell.execute_reply.started":"2023-10-12T15:29:24.658446Z","shell.execute_reply":"2023-10-12T15:29:24.677477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Adding image path\ndf['image_id_path'] = np.where(df['is_tma'], \n                               df['image_id'].astype(str) + \".png\", \n                               df['image_id'].astype(str) + \"_thumbnail.png\")\n","metadata":{"execution":{"iopub.status.busy":"2023-10-12T15:29:24.681126Z","iopub.execute_input":"2023-10-12T15:29:24.682142Z","iopub.status.idle":"2023-10-12T15:29:24.700193Z","shell.execute_reply.started":"2023-10-12T15:29:24.6821Z","shell.execute_reply":"2023-10-12T15:29:24.698894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_mapping = {'CC': 0, 'EC': 1, 'HGSC': 2, 'LGSC': 3, 'MC': 4}\ndf['int_label'] = df['label'].map(label_mapping)","metadata":{"execution":{"iopub.status.busy":"2023-10-12T15:29:24.704595Z","iopub.execute_input":"2023-10-12T15:29:24.705002Z","iopub.status.idle":"2023-10-12T15:29:24.722777Z","shell.execute_reply.started":"2023-10-12T15:29:24.704964Z","shell.execute_reply":"2023-10-12T15:29:24.721165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Split dataset into training and validation\ntrain_df, val_df = train_test_split(df, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-10-12T15:29:24.724517Z","iopub.execute_input":"2023-10-12T15:29:24.725Z","iopub.status.idle":"2023-10-12T15:29:24.737799Z","shell.execute_reply.started":"2023-10-12T15:29:24.724954Z","shell.execute_reply":"2023-10-12T15:29:24.736679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Differentiate between TMA and non-TMA\ntrain_df_no_tma = train_df[train_df['is_tma']==False]\ntrain_df_tma = train_df[train_df['is_tma']==True]\n\nval_df_no_tma = val_df[val_df['is_tma']==False]\nval_df_tma = val_df[val_df['is_tma']==True]","metadata":{"execution":{"iopub.status.busy":"2023-10-12T15:29:24.739839Z","iopub.execute_input":"2023-10-12T15:29:24.740221Z","iopub.status.idle":"2023-10-12T15:29:24.755509Z","shell.execute_reply.started":"2023-10-12T15:29:24.740188Z","shell.execute_reply":"2023-10-12T15:29:24.753894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create folders for train and validation images\ntrain_folder = \"/kaggle/working/train\"\nval_folder = \"/kaggle/working/val\"\n\nos.makedirs(train_folder, exist_ok=True)\nos.makedirs(val_folder, exist_ok=True)\n","metadata":{"execution":{"iopub.status.busy":"2023-10-12T15:29:24.757109Z","iopub.execute_input":"2023-10-12T15:29:24.75763Z","iopub.status.idle":"2023-10-12T15:29:24.769442Z","shell.execute_reply.started":"2023-10-12T15:29:24.757503Z","shell.execute_reply":"2023-10-12T15:29:24.767584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize sums for mean and std calculation\nsums = np.zeros(3)\nsums_squared = np.zeros(3)\nnormalizer = 0\n\n# Paths\ntumbnail_path = \"/kaggle/input/UBC-OCEAN/train_thumbnails\"\ntrain_path = \"/kaggle/input/UBC-OCEAN/train_images\"\n\n# Function to process each dataframe\ndef process_dataframe(df, folder_name, is_tma=False):\n    global sums, sums_squared, normalizer  # Declare as global to modify\n    labels_and_paths = []\n    source_path = train_path if is_tma else tumbnail_path\n    \n    for idx, row in df.iterrows():\n        img_path = os.path.join(source_path, row['image_id_path'])\n        img = cv2.imread(img_path)\n        \n        # Convert to float and normalize\n        img = img.astype(np.float32) / 255.0\n        \n        # Resize image to fit VGG16 input size\n        img_resized = cv2.resize(img, (224, 224))\n\n        if folder_name == 'train':\n            # Update sums for mean and std calculation\n            for i in range(3):  # Assuming 3 channels: R, G, B\n                sums[i] += np.sum(img_resized[:, :, i])\n                sums_squared[i] += np.sum(np.square(img_resized[:, :, i]))\n            normalizer += img_resized[:, :, 0].size  # size of one channel\n\n        save_path = f\"/kaggle/working/{folder_name}/{row['image_id_path']}\"\n        cv2.imwrite(save_path, img_resized * 255)\n        \n        labels_and_paths.append({'path': save_path, 'label': row['int_label']})\n    \n    return pd.DataFrame(labels_and_paths)\n","metadata":{"execution":{"iopub.status.busy":"2023-10-12T15:29:24.771538Z","iopub.execute_input":"2023-10-12T15:29:24.771917Z","iopub.status.idle":"2023-10-12T15:29:24.787293Z","shell.execute_reply.started":"2023-10-12T15:29:24.771888Z","shell.execute_reply":"2023-10-12T15:29:24.785797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Process and save train and validation images\ntrain_no_tma_labels_df = process_dataframe(train_df_no_tma, 'train')\ntrain_tma_labels_df = process_dataframe(train_df_tma, 'train', is_tma=True)\nval_no_tma_labels_df = process_dataframe(val_df_no_tma, 'val')\nval_tma_labels_df = process_dataframe(val_df_tma, 'val', is_tma=True)","metadata":{"execution":{"iopub.status.busy":"2023-10-12T15:29:24.789159Z","iopub.execute_input":"2023-10-12T15:29:24.78969Z","iopub.status.idle":"2023-10-12T15:32:04.076324Z","shell.execute_reply.started":"2023-10-12T15:29:24.789642Z","shell.execute_reply":"2023-10-12T15:32:04.075074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Combine the dataframes\ntrain_labels_df = pd.concat([train_no_tma_labels_df, train_tma_labels_df])\nval_labels_df = pd.concat([val_no_tma_labels_df, val_tma_labels_df])","metadata":{"execution":{"iopub.status.busy":"2023-10-12T15:32:04.077825Z","iopub.execute_input":"2023-10-12T15:32:04.078149Z","iopub.status.idle":"2023-10-12T15:32:04.088873Z","shell.execute_reply.started":"2023-10-12T15:32:04.078122Z","shell.execute_reply":"2023-10-12T15:32:04.087223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate mean and std\nmean = sums / normalizer\nstd = np.sqrt((sums_squared / normalizer) - np.square(mean))","metadata":{"execution":{"iopub.status.busy":"2023-10-12T15:32:04.090721Z","iopub.execute_input":"2023-10-12T15:32:04.091081Z","iopub.status.idle":"2023-10-12T15:32:04.106716Z","shell.execute_reply.started":"2023-10-12T15:32:04.091051Z","shell.execute_reply":"2023-10-12T15:32:04.105595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save labels and paths\ntrain_labels_df.to_csv('/kaggle/working/train_labels_and_paths.csv', index=False)\nval_labels_df.to_csv('/kaggle/working/val_labels_and_paths.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-10-12T15:32:04.108214Z","iopub.execute_input":"2023-10-12T15:32:04.10958Z","iopub.status.idle":"2023-10-12T15:32:04.133314Z","shell.execute_reply.started":"2023-10-12T15:32:04.109331Z","shell.execute_reply":"2023-10-12T15:32:04.132293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(mean)\nprint(std)","metadata":{"execution":{"iopub.status.busy":"2023-10-12T15:32:04.135097Z","iopub.execute_input":"2023-10-12T15:32:04.135604Z","iopub.status.idle":"2023-10-12T15:32:04.143381Z","shell.execute_reply.started":"2023-10-12T15:32:04.13557Z","shell.execute_reply":"2023-10-12T15:32:04.142467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Augmenting","metadata":{}},{"cell_type":"code","source":"class OvarianCancerDataset(Dataset):\n    def __init__(self, csv_file, transform=None):\n        \"\"\"\n        Args:\n            csv_file (string): Path to the CSV file with image paths and labels.\n            transform (callable, optional): Optional transform to be applied on a sample.\n        \"\"\"\n        self.dataframe = pd.read_csv(csv_file)\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.dataframe)\n\n    def __getitem__(self, idx):\n        img_path = self.dataframe.iloc[idx, 0]  # Assuming that image path is in the first column of the dataframe\n        image = Image.open(img_path).convert(\"RGB\")\n        label = self.dataframe.iloc[idx, 1]  # Assuming that label is in the second column of the dataframe\n\n        if self.transform:\n            image = self.transform(image)\n\n        return image, label\n","metadata":{"execution":{"iopub.status.busy":"2023-10-12T15:32:04.145259Z","iopub.execute_input":"2023-10-12T15:32:04.145964Z","iopub.status.idle":"2023-10-12T15:32:04.161459Z","shell.execute_reply.started":"2023-10-12T15:32:04.145922Z","shell.execute_reply":"2023-10-12T15:32:04.16015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define transformations for training and validation datasets\ntrain_transform = transforms.Compose([\n    transforms.RandomHorizontalFlip(),\n    transforms.RandomRotation(degrees=(-10, 10)),  # Rotate between -10 and 10 degrees\n    transforms.ToTensor(),\n    transforms.Normalize([0.488, 0.429, 0.49], [0.414, 0.374, 0.418])  # Use your own calculated mean and std\n])\n\nval_transform = transforms.Compose([\n    transforms.ToTensor(),\n    transforms.Normalize([0.488, 0.429, 0.49], [0.414, 0.374, 0.418])  # Use your own calculated mean and std\n])\n","metadata":{"execution":{"iopub.status.busy":"2023-10-12T15:32:04.163203Z","iopub.execute_input":"2023-10-12T15:32:04.163601Z","iopub.status.idle":"2023-10-12T15:32:04.175681Z","shell.execute_reply.started":"2023-10-12T15:32:04.163511Z","shell.execute_reply":"2023-10-12T15:32:04.174334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize datasets with different transforms\ntrain_dataset = OvarianCancerDataset(csv_file='/kaggle/working/train_labels_and_paths.csv', transform=train_transform)\nval_dataset = OvarianCancerDataset(csv_file='/kaggle/working/val_labels_and_paths.csv', transform=val_transform)\n\ntrain_loader = DataLoader(train_dataset, batch_size=32, shuffle=True, num_workers=2)\nval_loader = DataLoader(val_dataset, batch_size=32, shuffle=False, num_workers=2)","metadata":{"execution":{"iopub.status.busy":"2023-10-12T15:32:04.177471Z","iopub.execute_input":"2023-10-12T15:32:04.177815Z","iopub.status.idle":"2023-10-12T15:32:04.195383Z","shell.execute_reply.started":"2023-10-12T15:32:04.17779Z","shell.execute_reply":"2023-10-12T15:32:04.193818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Training","metadata":{}},{"cell_type":"code","source":"class UnifiedOvarianCancerModel(pl.LightningModule):\n\n    def __init__(self):\n        super(UnifiedOvarianCancerModel, self).__init__()\n\n        # Load pre-trained ResNet18 model + higher level features\n        self.model = torch.hub.load('pytorch/vision', 'resnet18', pretrained=True)\n\n        self.learning_rate = 1e-4\n        self.num_classes = 5\n        self.model.fc = torch.nn.Linear(in_features=512, out_features=self.num_classes, bias=True)\n\n        # Loss function and accuracy metric\n        self.criterion = nn.CrossEntropyLoss()\n        \n        self.train_acc = torchmetrics.Accuracy(num_classes=self.num_classes, task='multiclass')\n        self.val_acc = torchmetrics.Accuracy(num_classes=self.num_classes, task='multiclass')\n\n    def forward(self, x):\n        return self.model(x)\n\n    def training_step(self, batch, batch_idx):\n        x, y = batch\n        y_pred = self(x)\n        loss = self.criterion(y_pred, y)\n        self.log('train_loss', loss)\n        self.log('train_acc_step', self.train_acc(y_pred, y), on_step=True, on_epoch=False)\n        return loss\n\n    def on_training_epoch_end(self):\n        self.log('train_acc_epoch', self.train_acc.compute())\n\n    def validation_step(self, batch, batch_idx):\n        x, y = batch\n        y_pred = self(x)\n        loss = self.criterion(y_pred, y)\n        self.log('val_loss', loss)\n        self.log('val_acc_step', self.val_acc(y_pred, y), on_step=True, on_epoch=False)\n\n    def on_validation_epoch_end(self):\n        self.log('val_acc_epoch', self.val_acc.compute())\n\n    def configure_optimizers(self):\n        optimizer = torch.optim.Adam(self.parameters(), lr=self.learning_rate)\n        return optimizer","metadata":{"execution":{"iopub.status.busy":"2023-10-12T15:34:24.981632Z","iopub.execute_input":"2023-10-12T15:34:24.982138Z","iopub.status.idle":"2023-10-12T15:34:24.996139Z","shell.execute_reply.started":"2023-10-12T15:34:24.982102Z","shell.execute_reply":"2023-10-12T15:34:24.994523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = UnifiedOvarianCancerModel()\n\ncheckpoint_callback = ModelCheckpoint(\n    monitor='val_loss',\n    save_top_k=10,\n    mode='max',\n)\n\n# Initialize the logger\nlogger = TensorBoardLogger(save_dir='/kaggle/working/logs')\n\n# Initialize the trainer\ntrainer = Trainer(logger=logger, log_every_n_steps=1, callbacks=checkpoint_callback, max_epochs=10)\n\n# Fit the model\ntrainer.fit(model, train_loader, val_loader)","metadata":{"execution":{"iopub.status.busy":"2023-10-12T15:40:22.220114Z","iopub.execute_input":"2023-10-12T15:40:22.221551Z","iopub.status.idle":"2023-10-12T15:53:07.993293Z","shell.execute_reply.started":"2023-10-12T15:40:22.22148Z","shell.execute_reply":"2023-10-12T15:53:07.991605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import zipfile\nimport os\n\n# Initialize zipfile\nwith zipfile.ZipFile('/kaggle/working/my_archive.zip', 'w') as zipf:\n    # Specify the files and folders to zip\n    paths_to_zip = [\n        '/kaggle/working/logs',\n        '/kaggle/working/val',\n        '/kaggle/working/train',\n        '/kaggle/working/val_labels_and_paths.csv',\n        '/kaggle/working/train_labels_and_paths.csv'\n    ]\n    \n    for path in paths_to_zip:\n        if os.path.isfile(path):\n            # If it's a file, just write it to the zip\n            zipf.write(path, os.path.basename(path))\n        else:\n            # If it's a folder, write all its contents to the zip\n            for root, _, files in os.walk(path):\n                for file in files:\n                    zipf.write(os.path.join(root, file), os.path.relpath(os.path.join(root, file), '/kaggle/working'))\n","metadata":{"execution":{"iopub.status.busy":"2023-10-12T15:54:23.084829Z","iopub.execute_input":"2023-10-12T15:54:23.085348Z","iopub.status.idle":"2023-10-12T15:54:27.9034Z","shell.execute_reply.started":"2023-10-12T15:54:23.085306Z","shell.execute_reply":"2023-10-12T15:54:27.901844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}