{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport glob\nimport openslide\nimport tifffile\nfrom openslide import OpenSlide\nimport matplotlib.pyplot as plt\nimport gc\nfrom tqdm.auto import tqdm\nimport shutil\nfrom skimage.filters import threshold_otsu\nfrom PIL import Image\nimport random\nimport zipfile\n\nDATASET_FOLDER = \"/kaggle/input/mayo-clinic-strip-ai/\"\n\ntrain_df=pd.read_csv('../input/mayo-clinic-strip-ai/train.csv')\ntest_df=pd.read_csv('../input/mayo-clinic-strip-ai/test.csv')\n\nprint('Train Dataframe size: ',train_df.shape)\nprint('Test Dataframe size: ',test_df.shape)\n\ndisplay(test_df)\ndisplay(train_df)\n\nprint(train_df.label.value_counts())","metadata":{"execution":{"iopub.status.busy":"2022-10-03T19:11:39.761464Z","iopub.execute_input":"2022-10-03T19:11:39.762448Z","iopub.status.idle":"2022-10-03T19:11:39.800705Z","shell.execute_reply.started":"2022-10-03T19:11:39.762395Z","shell.execute_reply":"2022-10-03T19:11:39.799653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install efficientnet","metadata":{"execution":{"iopub.status.busy":"2022-10-03T19:11:39.802511Z","iopub.execute_input":"2022-10-03T19:11:39.80339Z","iopub.status.idle":"2022-10-03T19:11:48.706384Z","shell.execute_reply.started":"2022-10-03T19:11:39.803345Z","shell.execute_reply":"2022-10-03T19:11:48.705146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from __future__ import print_function, division\n\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.optim import lr_scheduler\nimport torch.backends.cudnn as cudnn\nfrom torch.utils.data import WeightedRandomSampler\nimport numpy as np\nimport torchvision\nfrom torchvision import datasets, models, transforms\nimport matplotlib.pyplot as plt\nimport time\nimport os\nimport copy\nimport cv2\n\ncudnn.benchmark = True\nplt.ion()   # interactive mode","metadata":{"execution":{"iopub.status.busy":"2022-10-03T19:11:48.709241Z","iopub.execute_input":"2022-10-03T19:11:48.709678Z","iopub.status.idle":"2022-10-03T19:11:48.721018Z","shell.execute_reply.started":"2022-10-03T19:11:48.709616Z","shell.execute_reply":"2022-10-03T19:11:48.720015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pickle\nimport math, re, os\nfrom os import path\nfrom tqdm import tqdm\n\nimport tensorflow as tf\nimport tensorflow.keras as keras\nfrom keras_preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras import layers\nfrom tensorflow.keras.callbacks import LearningRateScheduler, EarlyStopping, ReduceLROnPlateau, ModelCheckpoint\nimport numpy as np\nimport efficientnet.tfkeras as efn\nfrom tensorflow.keras import mixed_precision\n\nimport pandas as pd\nimport shutil\nimport pathlib\n\nEPOCHS = 50\nBATCH_SIZE = 10\nIMG_SIZE = (600, 600)","metadata":{"execution":{"iopub.status.busy":"2022-10-03T19:11:48.724832Z","iopub.execute_input":"2022-10-03T19:11:48.725178Z","iopub.status.idle":"2022-10-03T19:11:48.732876Z","shell.execute_reply.started":"2022-10-03T19:11:48.725152Z","shell.execute_reply":"2022-10-03T19:11:48.731672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def seed_everything(seed_value):\n    random.seed(seed_value)\n    np.random.seed(seed_value)\n    torch.manual_seed(seed_value)\n    os.environ['PYTHONHASHSEED'] = str(seed_value)    \n    if torch.cuda.is_available(): \n        torch.cuda.manual_seed(seed_value)\n        torch.cuda.manual_seed_all(seed_value)\n        torch.backends.cudnn.deterministic = True\n        torch.backends.cudnn.benchmark = True\n\nseed = 42\nseed_everything(seed)","metadata":{"execution":{"iopub.status.busy":"2022-10-03T19:11:48.734207Z","iopub.execute_input":"2022-10-03T19:11:48.734744Z","iopub.status.idle":"2022-10-03T19:11:48.74561Z","shell.execute_reply.started":"2022-10-03T19:11:48.734672Z","shell.execute_reply":"2022-10-03T19:11:48.744736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install split-folders\nimport splitfolders\n! rm -rf data\n\n\n!ls  ../input/1-fast-tiles-v2/processedimages/CE/ | wc -l\n!ls  ../input/1-fast-tiles-v2/processedimages/LAA | wc -l\n\nsplitfolders.ratio( os.path.join(os.getcwd(), \"../input/1-fast-tiles-v2/processedimages\"), output=\"data\", seed=1337, ratio=(.8, .2)) \n\n!ls  /kaggle/working/data/train/CE | wc -l\n!ls  /kaggle/working/data/train/LAA | wc -l\n!ls  /kaggle/working/data/val/CE | wc -l\n!ls  /kaggle/working/data/val/LAA | wc -l","metadata":{"execution":{"iopub.status.busy":"2022-10-03T19:11:48.747229Z","iopub.execute_input":"2022-10-03T19:11:48.747653Z","iopub.status.idle":"2022-10-03T19:12:30.976502Z","shell.execute_reply.started":"2022-10-03T19:11:48.747602Z","shell.execute_reply":"2022-10-03T19:12:30.975293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Data augmentation and normalization for training\n# Just normalization for validation\ndata_transforms = {\n    'train': transforms.Compose([\n        #transforms.Resize(512),\n        #transforms.RandomRotation(degrees=(0,180)),\n        #transforms.GaussianBlur(kernel_size=(5, 9), sigma=(0.1, 5)),\n        #transforms.ColorJitter(brightness=.5, hue=.3),\n        #transforms.RandomPerspective(distortion_scale=0.6, p=1.0),\n        #transforms.RandomAffine(degrees=(30, 70), translate=(0.1, 0.3), scale=(0.5, 0.75)),\n        #transforms.RandomPosterize(bits=2),\n        #transforms.RandomHorizontalFlip(),\n        #transforms.RandomVerticalFlip(),\n        #transforms.RandomRotation(90),\n        transforms.ToTensor(),\n        transforms.Normalize([0.485, 0.456, 0.406], [0.229, 0.224, 0.225])\n    ]),\n    'val': transforms.Compose([\n        #transforms.Resize(512),\n        #transforms.CenterCrop(224),\n        transforms.ToTensor(),\n        transforms.Normalize([0.485, 0.456, 0.406], [0.229, 0.224, 0.225])\n    ]),\n    \n    'test': transforms.Compose([\n        #transforms.Resize(512),\n        #transforms.CenterCrop(224),\n        transforms.ToTensor(),\n        transforms.Normalize([0.485, 0.456, 0.406], [0.229, 0.224, 0.225])\n    ]),\n}\n\ndata_dir = 'data'\n#data_dir = '../input/tiles-combine-new-64/data'\n#data_dir = '../input/catsdogs/hymenoptera_data'\nimage_datasets = {x: datasets.ImageFolder(os.path.join(data_dir, x),\n                                          data_transforms[x])\n                  for x in ['train', 'val']}\n\nprint(image_datasets[\"train\"].__len__())\n\ny_train = [image_datasets[\"train\"].targets[i] for i in range(image_datasets[\"train\"].__len__())]\n\nclass_sample_count = np.array(\n    [len(np.where(y_train == t)[0]) for t in np.unique(y_train)])\n\nprint(class_sample_count)\n\nweight = 1. / class_sample_count\nsamples_weight = np.array([weight[t] for t in y_train])\nsamples_weight = torch.from_numpy(samples_weight)\n\nsampler = WeightedRandomSampler(samples_weight.type('torch.DoubleTensor'), len(samples_weight))\n\n\n#dataloaders = {x: torch.utils.data.DataLoader(image_datasets[x], batch_size=40,\n#                                             sampler=sampler, num_workers=4)\n#              for x in ['train', 'val']}\n\ndataloaders={}\ndataloaders[\"train\"]=torch.utils.data.DataLoader(image_datasets[\"train\"], batch_size=8,\n                                             sampler=sampler, num_workers=4)\n\ndataloaders[\"val\"]=torch.utils.data.DataLoader(image_datasets[\"val\"], batch_size=8,\n                                             shuffle=True, num_workers=4)\n\n\ndataset_sizes = {x: len(image_datasets[x]) for x in ['train', 'val']}\nclass_names = image_datasets['train'].classes\nprint(class_names)\n\ndevice = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")","metadata":{"execution":{"iopub.status.busy":"2022-10-03T19:12:30.979506Z","iopub.execute_input":"2022-10-03T19:12:30.980021Z","iopub.status.idle":"2022-10-03T19:12:31.031756Z","shell.execute_reply.started":"2022-10-03T19:12:30.979975Z","shell.execute_reply":"2022-10-03T19:12:31.03071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def imshow(inp, title=None):\n    \"\"\"Imshow for Tensor.\"\"\"\n    inp = inp.numpy().transpose((1, 2, 0))\n    mean = np.array([0.485, 0.456, 0.406])\n    std = np.array([0.229, 0.224, 0.225])\n    inp = std * inp + mean\n    inp = np.clip(inp, 0, 1)\n    plt.imshow(inp)\n    if title is not None:\n        plt.title(title)\n    plt.pause(0.001)  # pause a bit so that plots are updated\n\n\n# Get a batch of training data\ninputs, classes = next(iter(dataloaders['train']))\n\n# Make a grid from batch\nout = torchvision.utils.make_grid(inputs)\n\nimshow(out, title=[class_names[x] for x in classes])","metadata":{"execution":{"iopub.status.busy":"2022-10-03T19:12:31.033286Z","iopub.execute_input":"2022-10-03T19:12:31.034928Z","iopub.status.idle":"2022-10-03T19:12:33.306683Z","shell.execute_reply.started":"2022-10-03T19:12:31.034887Z","shell.execute_reply":"2022-10-03T19:12:33.305683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nmixed_precision.set_global_policy('mixed_float16')\n\ndef create_cnn_model():\n\n    model = keras.models.Sequential()\n    pre_trained_model = efn.EfficientNetB4(input_shape=(*IMG_SIZE, 3),\n                                    include_top=False, \n                                    weights='noisy-student')\n\n    # freeze the batch normalisation layers\n    for layer in reversed(pre_trained_model.layers):\n        if isinstance(layer, tf.keras.layers.BatchNormalization):\n            layer.trainable = False\n        else:\n            layer.trainable = True\n\n    model.add(pre_trained_model)\n    model.add(layers.Dropout(0.4))\n    model.add(layers.GlobalAveragePooling2D())\n    model.add(layers.Dropout(0.4))\n    model.add(layers.Dense(2, activation='softmax'))\n\n    # add metrics\n    metrics = [\n        tf.keras.metrics.CategoricalAccuracy(name='accuracy'),\n    ]\n\n    optimizer = tf.keras.optimizers.Adam()\n    loss = tf.keras.losses.CategoricalCrossentropy()\n\n    model.compile(optimizer=optimizer, loss=loss, metrics=metrics)\n    print(model.summary())\n    return model\n","metadata":{"execution":{"iopub.status.busy":"2022-10-03T19:12:33.308741Z","iopub.execute_input":"2022-10-03T19:12:33.309147Z","iopub.status.idle":"2022-10-03T19:12:33.319474Z","shell.execute_reply.started":"2022-10-03T19:12:33.309102Z","shell.execute_reply":"2022-10-03T19:12:33.318344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LEARNING_RATE = 3e-5\nLR_START = 1e-8\nLR_MIN = 1e-8\nLR_MAX = LEARNING_RATE\nLR_RAMPUP_EPOCHS = 3\nLR_SUSTAIN_EPOCHS = 0\nN_CYCLES = .5","metadata":{"execution":{"iopub.status.busy":"2022-10-03T19:12:33.323341Z","iopub.execute_input":"2022-10-03T19:12:33.324154Z","iopub.status.idle":"2022-10-03T19:12:33.330667Z","shell.execute_reply.started":"2022-10-03T19:12:33.324125Z","shell.execute_reply":"2022-10-03T19:12:33.329617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def lrfn(epoch):\n    if epoch < LR_RAMPUP_EPOCHS:\n        lr = (LR_MAX - LR_START) / LR_RAMPUP_EPOCHS * epoch + LR_START\n    elif epoch < LR_RAMPUP_EPOCHS + LR_SUSTAIN_EPOCHS:\n        lr = LR_MAX\n    else:\n        progress = (epoch - LR_RAMPUP_EPOCHS - LR_SUSTAIN_EPOCHS) / (EPOCHS - LR_RAMPUP_EPOCHS - LR_SUSTAIN_EPOCHS)\n        lr = LR_MAX * (0.5 * (1.0 + tf.math.cos(math.pi * N_CYCLES * 2.0 * progress)))\n        if LR_MIN is not None:\n            lr = tf.math.maximum(LR_MIN, lr)\n            \n    return lr","metadata":{"execution":{"iopub.status.busy":"2022-10-03T19:12:33.331952Z","iopub.execute_input":"2022-10-03T19:12:33.332319Z","iopub.status.idle":"2022-10-03T19:12:33.341944Z","shell.execute_reply.started":"2022-10-03T19:12:33.332282Z","shell.execute_reply":"2022-10-03T19:12:33.341038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_callbacks():\n    early_stopping = EarlyStopping(patience=3, monitor='val_loss', verbose=1)\n\n    lr_schedule = LearningRateScheduler(lrfn, verbose=1)\n\n    model_checkpoint = ModelCheckpoint(monitor='val_loss',\n                                       filepath='./best-model-efn.h5',\n                                       save_best_only=True,\n                                       verbose=1)\n\n    callbacks = [\n        early_stopping,\n        lr_schedule,\n        model_checkpoint,\n    ]\n\n    return callbacks","metadata":{"execution":{"iopub.status.busy":"2022-10-03T19:12:33.343498Z","iopub.execute_input":"2022-10-03T19:12:33.344166Z","iopub.status.idle":"2022-10-03T19:12:33.355551Z","shell.execute_reply.started":"2022-10-03T19:12:33.34411Z","shell.execute_reply":"2022-10-03T19:12:33.35461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_model_naive_split():\n\n    inp_train_gen = ImageDataGenerator(\n        rescale=1. / 255,\n        rotation_range=260,\n        width_shift_range=0.4,\n        height_shift_range=0.4,\n        shear_range=0.2,\n        zoom_range=0.4,\n        horizontal_flip=True,\n        vertical_flip=True,\n        fill_mode='nearest'\n    )\n\n    # Create training and validation generator.\n    # train_iterator = lm.MixupImageDataGenerator(generator=inp_train_gen,\n    #                                           directory='./train/train',\n    #                                           batch_size=BATCH_SIZE,\n    #                                           img_height=IMG_SIZE[0],\n    #                                           img_width=IMG_SIZE[1],\n    #                                           subset='training')\n\n    train_iterator = inp_train_gen.flow_from_directory('/kaggle/working/data/train',\n                                                   target_size=IMG_SIZE,\n                                                   batch_size=BATCH_SIZE,\n                                                   class_mode='categorical')\n\n    validation_gen = ImageDataGenerator(rescale=1. / 255.0)\n    validation_iterator = validation_gen.flow_from_directory('/kaggle/working/data/val',\n                                                             target_size=IMG_SIZE,\n                                                             batch_size=BATCH_SIZE,\n                                                             class_mode='categorical')\n\n    model = create_cnn_model()\n\n    history = model.fit(train_iterator,\n                        #steps_per_epoch=train_iterator.get_steps_per_epoch(),\n                        validation_data=validation_iterator,\n                        epochs=EPOCHS,\n                        callbacks=create_callbacks())\n\n    return history\n\n","metadata":{"execution":{"iopub.status.busy":"2022-10-03T19:12:33.357059Z","iopub.execute_input":"2022-10-03T19:12:33.357656Z","iopub.status.idle":"2022-10-03T19:12:33.367632Z","shell.execute_reply.started":"2022-10-03T19:12:33.357601Z","shell.execute_reply":"2022-10-03T19:12:33.36663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_and_predict(model):\n\n    test_generator = ImageDataGenerator(rescale=1. / 255,\n        rotation_range=360,\n        width_shift_range=0.2,\n        height_shift_range=0.2,\n        shear_range=0.2,\n        zoom_range=0.2,\n        horizontal_flip=True,\n        vertical_flip=True,\n        fill_mode='nearest')\n\n    ids = []\n    tta_predictions = []\n\n    for i in tqdm(range(10)):\n        test_iterator = test_generator.flow_from_directory(\n            './test/',\n            target_size=IMG_SIZE,\n            shuffle=False,\n            class_mode='categorical',\n            batch_size=1)\n        \n        if i == 1: \n            for filename in test_iterator.filenames:\n                print(filename)\n                ids.append(filename.split('/')[1])\n        \n        predict_result = model.predict(test_iterator, steps=len(test_iterator.filenames))\n        tta_predictions.append(predict_result)\n    \n    result = []\n    predictions = np.mean(tta_predictions, axis=0)\n    for index, prediction in enumerate(predictions):\n        classes = np.argmax(prediction)\n        result.append([ids[index], classes])\n    result.sort()\n\n    return result\n","metadata":{"execution":{"iopub.status.busy":"2022-10-03T19:12:33.369207Z","iopub.execute_input":"2022-10-03T19:12:33.369882Z","iopub.status.idle":"2022-10-03T19:12:33.382034Z","shell.execute_reply.started":"2022-10-03T19:12:33.369848Z","shell.execute_reply":"2022-10-03T19:12:33.381094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def store_prediction():\n    model = keras.models.load_model('./best-model.h5', compile = True)\n\n    pathlib.Path(f'./test/1/').mkdir(parents=True, exist_ok=True)\n\n    test_images = os.listdir(TEST_IMAGES_INPUT)\n    ld.copy_test_images(test_images, TEST_IMAGES_INPUT)\n\n    predictions = load_and_predict(model)\n\n    # clean temp files\n    if os.path.exists(\"./train\"):\n        shutil.rmtree('./train')\n\n    if os.path.exists(\"./test\"):\n        shutil.rmtree('./test')\n\n    df = pd.DataFrame(data=predictions, columns=['image_id', 'label'])\n    df = df.set_index(['image_id'])\n\n    if os.path.exists(SUBMISSION_FILE):\n        os.remove(SUBMISSION_FILE)\n\n    print(df.head())\n    print('Writing submission')\n    df.to_csv(SUBMISSION_FILE)\n","metadata":{"execution":{"iopub.status.busy":"2022-10-03T19:12:33.383341Z","iopub.execute_input":"2022-10-03T19:12:33.383935Z","iopub.status.idle":"2022-10-03T19:12:33.395685Z","shell.execute_reply.started":"2022-10-03T19:12:33.383897Z","shell.execute_reply":"2022-10-03T19:12:33.39486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = train_model_naive_split()\nlp.plot_result(history)\nstore_prediction()","metadata":{"execution":{"iopub.status.busy":"2022-10-03T19:12:33.397251Z","iopub.execute_input":"2022-10-03T19:12:33.397885Z","iopub.status.idle":"2022-10-03T19:13:35.223659Z","shell.execute_reply.started":"2022-10-03T19:12:33.39785Z","shell.execute_reply":"2022-10-03T19:13:35.221133Z"},"trusted":true},"execution_count":null,"outputs":[]}]}