{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport glob\nimport openslide\nimport tifffile\nfrom openslide import OpenSlide\nimport matplotlib.pyplot as plt\nimport gc\nfrom tqdm.auto import tqdm\nimport shutil\n\nDATASET_FOLDER = \"/kaggle/input/mayo-clinic-strip-ai/\"\n\ntrain_df=pd.read_csv('../input/mayo-clinic-strip-ai/train.csv')\ntest_df=pd.read_csv('../input/mayo-clinic-strip-ai/test.csv')\n\nprint('Train Dataframe size: ',train_df.shape)\nprint('Test Dataframe size: ',test_df.shape)\n\ndisplay(test_df)\ndisplay(train_df)\n\nprint(train_df.label.value_counts())","metadata":{"execution":{"iopub.status.busy":"2022-09-16T20:39:21.592112Z","iopub.execute_input":"2022-09-16T20:39:21.592589Z","iopub.status.idle":"2022-09-16T20:39:21.965376Z","shell.execute_reply.started":"2022-09-16T20:39:21.5925Z","shell.execute_reply":"2022-09-16T20:39:21.964222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from __future__ import print_function, division\n\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.optim import lr_scheduler\nimport torch.backends.cudnn as cudnn\nfrom torch.utils.data import WeightedRandomSampler\nimport numpy as np\nimport torchvision\nfrom torchvision import datasets, models, transforms\nimport matplotlib.pyplot as plt\nimport time\nimport os\nimport copy\n\ncudnn.benchmark = True\nplt.ion()   # interactive mode","metadata":{"execution":{"iopub.status.busy":"2022-09-16T20:39:21.967301Z","iopub.execute_input":"2022-09-16T20:39:21.968705Z","iopub.status.idle":"2022-09-16T20:39:24.151691Z","shell.execute_reply.started":"2022-09-16T20:39:21.968665Z","shell.execute_reply":"2022-09-16T20:39:24.150659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! rm -rf merge_folder\n# Function to create new folder if not exists\ndef make_new_folder(folder_name, parent_folder):\n      \n    # Path\n    path = os.path.join(parent_folder, folder_name)\n      \n    # Create the folder\n    # 'new_folder' in\n    # parent_folder\n    try: \n        # mode of the folder\n        mode = 0o777\n  \n        # Create folder\n        os.mkdir(path, mode) \n    except OSError as error: \n        print(error)\n  \n# current folder path\ncurrent_folder = os.getcwd() \n  \n# list of folders to be merged\nlist_dir = ['../input/1-tiles-combine-openslide/processedimages/CE','../input/1-tiles-combine-openslide/processedimages/LAA',\n            '../input/2-tiles-combine-openslide/processedimages/CE', '../input/2-tiles-combine-openslide/processedimages/LAA',\n            '../input/3-tiles-combine-openslide/processedimages/CE','../input/3-tiles-combine-openslide/processedimages/LAA',\n            '../input/4-tiles-combine-openslide/processedimages/CE','../input/4-tiles-combine-openslide/processedimages/LAA',\n           '../input/5-tiles-combine-openslide/processedimages/CE','../input/5-tiles-combine-openslide/processedimages/LAA',\n            '../input/6-tiles-combine-openslide/processedimages/CE','../input/6-tiles-combine-openslide/processedimages/LAA',\n           '../input/7-tiles-combine-openslide/processedimages/CE','../input/7-tiles-combine-openslide/processedimages/LAA',\n            '../input/8-tiles-combine-openslide/processedimages/CE','../input/8-tiles-combine-openslide/processedimages/LAA']\n\n# enumerate on list_dir to get the \n# content of all the folders ans store \n# it in a dictionary\ncontent_list = {}\nfor index, val in enumerate(list_dir):\n    path = os.path.join(current_folder, val)\n    content_list[ list_dir[index] ] = os.listdir(path)\n  \n# folder in which all the content will\n# be merged\nmerge_folder = \"merge_folder\"\n  \n# merge_folder path - current_folder \n# + merge_folder\nmerge_folder_path = os.path.join(current_folder, merge_folder) \n  \n# create merge_folder if not exists\nmake_new_folder(merge_folder, current_folder)\nmake_new_folder(\"CE\", current_folder+\"/\"+merge_folder)\nmake_new_folder(\"LAA\", current_folder+\"/\"+merge_folder)\n  \n# loop through the list of folders\nfor sub_dir in content_list:\n  \n    # loop through the contents of the \n    # list of folders\n    for contents in content_list[sub_dir]:\n  \n        # make the path of the content to move \n        path_to_content = sub_dir + \"/\" + contents\n        #print(path_to_content)\n  \n        # make the path with the current folder\n        split =  path_to_content.split(os.sep)\n        #print(merge_folder_path + \"/\" + split[-2] + \"/\" + split[-1])\n        # move the file\n        shutil.copy(path_to_content, merge_folder_path + \"/\" + split[-2] + \"/\" + split[-1])","metadata":{"execution":{"iopub.status.busy":"2022-09-16T20:39:24.15338Z","iopub.execute_input":"2022-09-16T20:39:24.154774Z","iopub.status.idle":"2022-09-16T20:39:57.118957Z","shell.execute_reply.started":"2022-09-16T20:39:24.154735Z","shell.execute_reply":"2022-09-16T20:39:57.117688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install split-folders\nimport splitfolders\n! rm -rf data\n\n!ls  /kaggle/working/merge_folder/CE | wc -l\n!ls  /kaggle/working/merge_folder/LAA | wc -l\n\nsplitfolders.ratio( os.path.join(os.getcwd(), \"merge_folder\"), output=\"data\", seed=1337, ratio=(.8, .2)) \n\n!ls  /kaggle/working/data/train/CE | wc -l\n!ls  /kaggle/working/data/train/LAA | wc -l\n!ls  /kaggle/working/data/val/CE | wc -l\n!ls  /kaggle/working/data/val/LAA | wc -l","metadata":{"execution":{"iopub.status.busy":"2022-09-16T20:39:57.122359Z","iopub.execute_input":"2022-09-16T20:39:57.122962Z","iopub.status.idle":"2022-09-16T20:40:18.182503Z","shell.execute_reply.started":"2022-09-16T20:39:57.122918Z","shell.execute_reply":"2022-09-16T20:40:18.181339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import shutil\n\nshutil.make_archive(\"imagesip\", \"zip\", \"/kaggle/working/data/train\")","metadata":{"execution":{"iopub.status.busy":"2022-09-16T20:40:18.185349Z","iopub.execute_input":"2022-09-16T20:40:18.185786Z","iopub.status.idle":"2022-09-16T20:41:12.815621Z","shell.execute_reply.started":"2022-09-16T20:40:18.185738Z","shell.execute_reply":"2022-09-16T20:41:12.814503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Data augmentation and normalization for training\n# Just normalization for validation\ndata_transforms = {\n    'train': transforms.Compose([\n        transforms.RandomResizedCrop(512),\n        #transforms.RandomRotation(degrees=(0,180)),\n        transforms.GaussianBlur(kernel_size=(5, 9), sigma=(0.1, 5)),\n        transforms.ColorJitter(brightness=.5, hue=.3),\n        #transforms.RandomPerspective(distortion_scale=0.6, p=1.0),\n        #transforms.RandomAffine(degrees=(30, 70), translate=(0.1, 0.3), scale=(0.5, 0.75)),\n        #transforms.RandomPosterize(bits=2),\n        transforms.RandomHorizontalFlip(),\n        transforms.RandomVerticalFlip(),\n        #transforms.RandomRotation(90),\n        transforms.ToTensor(),\n        transforms.Normalize([0.485, 0.456, 0.406], [0.229, 0.224, 0.225])\n    ]),\n    'val': transforms.Compose([\n        transforms.Resize(512),\n        #transforms.CenterCrop(224),\n        transforms.ToTensor(),\n        transforms.Normalize([0.485, 0.456, 0.406], [0.229, 0.224, 0.225])\n    ]),\n}\n\ndata_dir = 'data'\n#data_dir = '../input/tiles-combine-new-64/data'\n#data_dir = '../input/catsdogs/hymenoptera_data'\nimage_datasets = {x: datasets.ImageFolder(os.path.join(data_dir, x),\n                                          data_transforms[x])\n                  for x in ['train', 'val']}\n\nprint(image_datasets[\"train\"].__len__())\n\ny_train = [image_datasets[\"train\"].targets[i] for i in range(image_datasets[\"train\"].__len__())]\n\nclass_sample_count = np.array(\n    [len(np.where(y_train == t)[0]) for t in np.unique(y_train)])\n\nprint(class_sample_count)\n\nweight = 1. / class_sample_count\nsamples_weight = np.array([weight[t] for t in y_train])\nsamples_weight = torch.from_numpy(samples_weight)\n\nsampler = WeightedRandomSampler(samples_weight.type('torch.DoubleTensor'), len(samples_weight))\n\n\n#dataloaders = {x: torch.utils.data.DataLoader(image_datasets[x], batch_size=40,\n#                                             sampler=sampler, num_workers=4)\n#              for x in ['train', 'val']}\n\ndataloaders={}\ndataloaders[\"train\"]=torch.utils.data.DataLoader(image_datasets[\"train\"], batch_size=8,\n                                             sampler=sampler, num_workers=4)\n\ndataloaders[\"val\"]=torch.utils.data.DataLoader(image_datasets[\"val\"], batch_size=8,\n                                             shuffle=True, num_workers=4)\n\n\ndataset_sizes = {x: len(image_datasets[x]) for x in ['train', 'val']}\nclass_names = image_datasets['train'].classes\nprint(class_names)\n\ndevice = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")","metadata":{"execution":{"iopub.status.busy":"2022-09-16T20:41:12.817222Z","iopub.execute_input":"2022-09-16T20:41:12.818703Z","iopub.status.idle":"2022-09-16T20:41:12.907563Z","shell.execute_reply.started":"2022-09-16T20:41:12.818665Z","shell.execute_reply":"2022-09-16T20:41:12.906586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def imshow(inp, title=None):\n    \"\"\"Imshow for Tensor.\"\"\"\n    inp = inp.numpy().transpose((1, 2, 0))\n    mean = np.array([0.485, 0.456, 0.406])\n    std = np.array([0.229, 0.224, 0.225])\n    inp = std * inp + mean\n    inp = np.clip(inp, 0, 1)\n    plt.imshow(inp)\n    if title is not None:\n        plt.title(title)\n    plt.pause(0.001)  # pause a bit so that plots are updated\n\n\n# Get a batch of training data\ninputs, classes = next(iter(dataloaders['train']))\n\n# Make a grid from batch\nout = torchvision.utils.make_grid(inputs)\n\nimshow(out, title=[class_names[x] for x in classes])","metadata":{"execution":{"iopub.status.busy":"2022-09-16T20:41:12.908956Z","iopub.execute_input":"2022-09-16T20:41:12.909823Z","iopub.status.idle":"2022-09-16T20:41:18.906579Z","shell.execute_reply.started":"2022-09-16T20:41:12.909781Z","shell.execute_reply":"2022-09-16T20:41:18.905608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_model(model, criterion, optimizer, scheduler, num_epochs=25):\n    since = time.time()\n\n    best_model_wts = copy.deepcopy(model.state_dict())\n    best_acc = 0.0\n\n    for epoch in range(num_epochs):\n        print(f'Epoch {epoch}/{num_epochs - 1}')\n        print('-' * 10)\n\n        # Each epoch has a training and validation phase\n        for phase in ['train', 'val']:\n            if phase == 'train':\n                model.train()  # Set model to training mode\n            else:\n                model.eval()   # Set model to evaluate mode\n\n            running_loss = 0.0\n            running_corrects = 0\n\n            # Iterate over data.\n            for inputs, labels in dataloaders[phase]:\n                inputs = inputs.to(device)\n                labels = labels.to(device)\n\n                # zero the parameter gradients\n                optimizer.zero_grad()\n\n                # forward\n                # track history if only in train\n                with torch.set_grad_enabled(phase == 'train'):\n                    outputs = model(inputs)\n                    _, preds = torch.max(outputs, 1)\n                    loss = criterion(outputs, labels)\n\n                    # backward + optimize only if in training phase\n                    if phase == 'train':\n                        loss.backward()\n                        optimizer.step()\n\n                # statistics\n                running_loss += loss.item() * inputs.size(0)\n                running_corrects += torch.sum(preds == labels.data)\n            #if phase == 'train':\n            #    scheduler.step()\n\n            epoch_loss = running_loss / dataset_sizes[phase]\n            epoch_acc = running_corrects.double() / dataset_sizes[phase]\n\n            print(f'{phase} Loss: {epoch_loss:.4f} Acc: {epoch_acc:.4f}')\n\n            # deep copy the model\n            if phase == 'val' and epoch_acc > best_acc:\n                best_acc = epoch_acc\n                best_model_wts = copy.deepcopy(model.state_dict())\n\n        print()\n\n    time_elapsed = time.time() - since\n    print(f'Training complete in {time_elapsed // 60:.0f}m {time_elapsed % 60:.0f}s')\n    print(f'Best val Acc: {best_acc:4f}')\n\n    # load best model weights\n    model.load_state_dict(best_model_wts)\n    return model","metadata":{"execution":{"iopub.status.busy":"2022-09-16T20:41:18.908355Z","iopub.execute_input":"2022-09-16T20:41:18.908989Z","iopub.status.idle":"2022-09-16T20:41:18.921357Z","shell.execute_reply.started":"2022-09-16T20:41:18.908944Z","shell.execute_reply":"2022-09-16T20:41:18.920265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# VGG","metadata":{}},{"cell_type":"code","source":"model_conv = models.vgg16(pretrained=True)\n# Freeze model weights\nfor param in model_conv.parameters():\n    param.requires_grad = False\n    \nmodel_conv","metadata":{"execution":{"iopub.status.busy":"2022-09-16T20:41:18.922857Z","iopub.execute_input":"2022-09-16T20:41:18.923285Z","iopub.status.idle":"2022-09-16T20:41:43.920004Z","shell.execute_reply.started":"2022-09-16T20:41:18.92325Z","shell.execute_reply":"2022-09-16T20:41:43.918861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Add on classifier\nn_classes=2\nn_inputs = model_conv.classifier[6].in_features\n\nmodel_conv.classifier[6] = nn.Linear(n_inputs, 2)\n\nmodel_conv = model_conv.to(device)\n\ncriterion = nn.CrossEntropyLoss()\n\n# Observe that only parameters of final layer are being optimized as\n# opposed to before.\noptimizer_conv = optim.SGD(model_conv.classifier[6].parameters(), lr=0.1, momentum=0.9)\n#optimizer_conv = optim.Adam(model_conv.classifier[6].parameters())\n\n# Decay LR by a factor of 0.1 every 7 epochs\nexp_lr_scheduler = lr_scheduler.StepLR(optimizer_conv, step_size=7, gamma=0.1)","metadata":{"execution":{"iopub.status.busy":"2022-09-16T20:41:43.923568Z","iopub.execute_input":"2022-09-16T20:41:43.923926Z","iopub.status.idle":"2022-09-16T20:41:47.008471Z","shell.execute_reply.started":"2022-09-16T20:41:43.92389Z","shell.execute_reply":"2022-09-16T20:41:47.007242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nnum_epocs=100\nmodel_conv = train_model(model_conv, criterion, optimizer_conv,\n                         exp_lr_scheduler, num_epochs=num_epocs)","metadata":{"execution":{"iopub.status.busy":"2022-09-16T20:41:47.010414Z","iopub.execute_input":"2022-09-16T20:41:47.011234Z","iopub.status.idle":"2022-09-16T20:44:57.777314Z","shell.execute_reply.started":"2022-09-16T20:41:47.011194Z","shell.execute_reply":"2022-09-16T20:44:57.775739Z"},"trusted":true},"execution_count":null,"outputs":[]}]}