{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# [RSNA Screening Mammography Breast Cancer Detection](https://www.kaggle.com/c/petfinder-pawpularity-score)\n> Find breast cancers in screening mammograms\n\n![](https://storage.googleapis.com/kaggle-competitions/kaggle/39272/logos/header.png?t=2022-11-28-17-29-35)","metadata":{}},{"cell_type":"markdown","source":"# Install Libraries","metadata":{}},{"cell_type":"markdown","source":"## Install for `TPU 1VM v3-8`\nUsually these libraries come pre-installed for other accelerators but fo `tpu-1vm` we need to install them manually. If you want to use rather `remote-TPU` or `GPU` or `CPU`, then comment out the following cell.","metadata":{}},{"cell_type":"code","source":"# !pip install -q /lib/wheels/tensorflow-2.9.1-cp38-cp38-linux_x86_64.whl\n!pip install -q tensorflow-addons==0.18.0\n!pip install -q tensorflow-probability==0.17.0\n!pip install -q opencv-python-headless\n!pip install -q seaborn","metadata":{"_kg_hide-input":false,"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Install Custom Libraries","metadata":{}},{"cell_type":"code","source":"!pip install -qU wandb\n!pip install -qU scikit-learn","metadata":{"_kg_hide-output":true,"_kg_hide-input":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Import Libraries","metadata":{}},{"cell_type":"code","source":"import os\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '3'  # to avoid too many logging messages\nimport pandas as pd, numpy as np, random, shutil\nimport tensorflow as tf, re, math\nimport tensorflow.keras.backend as K\nimport sklearn\nimport matplotlib.pyplot as plt\nimport tensorflow_addons as tfa\nimport tensorflow_probability as tfp\nimport wandb\nimport yaml\n\nfrom IPython import display as ipd\nfrom glob import glob\nfrom tqdm import tqdm\nfrom sklearn.model_selection import KFold, StratifiedKFold, GroupKFold, StratifiedGroupKFold\nfrom sklearn.metrics import roc_auc_score","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Version Check","metadata":{}},{"cell_type":"code","source":"print('np:', np.__version__)\nprint('pd:', pd.__version__)\nprint('sklearn:', sklearn.__version__)\nprint('tf:',tf.__version__)\nprint('tfp:', tfp.__version__)\nprint('tfa:', tfa.__version__)\nprint('w&b:', wandb.__version__)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Configuration","metadata":{}},{"cell_type":"code","source":"class CFG:\n    wandb         = True\n    competition   = 'rsna-bcd' \n    _wandb_kernel = 'awsaf49'\n    debug         = False\n    comment       = 'GCViTTiny-224x224-check1'\n    exp_name      = 'roi-v2-fix' # name of the experiment, folds will be grouped using 'exp_name'\n    \n    # use verbose=0 for silent, vebose=1 for interactive,\n    verbose      = 1 if debug else 0\n    display_plot = True\n\n    # device\n    device = \"TPU-1VM\" #or \"GPU\"\n\n    model_name = 'GCViTTiny'\n\n    # seed for data-split, layer init, augs\n    seed = 42\n\n    # number of folds for data-split\n    folds = 5\n    \n    # which folds to train\n    selected_folds = [0, 1, 2]\n\n    # size of the image\n    img_size = [224, 224]\n\n    # batch_size and epochs\n    batch_size = 4\n    \n    epochs = 10\n    \n    # upsample\n    upsample = 10\n\n    # loss and optimizer\n    loss      = 'Focal'  # BCE, Focal\n    optimizer = 'Adam'\n\n    # augmentation\n    augment   = True\n\n    # scale-shift-rotate-shear\n    transform = True\n    fill_mode = 'constant'\n    rot    = 2.0\n    shr    = 2.0\n    hzoom  = 50.0\n    wzoom  = 50.0\n    hshift = 10.0\n    wshift = 10.0\n\n    # flip\n    hflip = True\n    vflip = True\n\n    # clip\n    clip = False\n\n    # lr-scheduler\n    scheduler   = 'exp' # cosine\n\n    # dropout\n    drop_prob   = 0.6\n    drop_cnt    = 10\n    drop_size   = 0.08\n    \n    # cut-mix-up\n    mixup_prob = 0.0\n    mixup_alpha = 0.5\n    \n    cutmix_prob = 0.0\n    cutmix_alpha = 2.5\n\n    # pixel-augment\n    pixel_aug = True\n    sat  = [0.7, 1.3]\n    cont = [0.8, 1.2]\n    bri  = 0.15\n    hue  = 0.05\n\n    # test-time augs\n    tta = 1\n    \n    # target column\n    target_col  = ['cancer']","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Reproducibility","metadata":{}},{"cell_type":"code","source":"def seeding(SEED):\n    np.random.seed(SEED)\n    random.seed(SEED)\n    os.environ['PYTHONHASHSEED'] = str(SEED)\n#     os.environ['TF_CUDNN_DETERMINISTIC'] = str(SEED)\n    tf.random.set_seed(SEED)\n    print('seeding done!!!')\nseeding(CFG.seed)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Device Configs\nThis notebook is compatible for **remote-tpu**, **local-tpu**, **multi-gpu** and **single-gpu**. Simple change to `device=\"TPU\"` for **remote-tpu** and `device=\"TPU-1VM\"` for **local-tpu** and finally, `device=\"GPU\"` for single or multi-gpu.","metadata":{}},{"cell_type":"code","source":"if \"TPU\" in CFG.device:\n    tpu = 'local' if CFG.device=='TPU-1VM' else None\n    print(\"connecting to TPU...\")\n    try:\n        tpu = tf.distribute.cluster_resolver.TPUClusterResolver.connect(tpu=tpu)\n        strategy = tf.distribute.TPUStrategy(tpu)\n    except:\n        CFG.device = \"GPU\"\n        \nif CFG.device == \"GPU\"  or CFG.device==\"CPU\":\n    ngpu = len(tf.config.experimental.list_physical_devices('GPU'))\n    if ngpu>1:\n        print(\"Using multi GPU\")\n        strategy = tf.distribute.MirroredStrategy()\n    elif ngpu==1:\n        print(\"Using single GPU\")\n        strategy = tf.distribute.get_strategy()\n    else:\n        print(\"Using CPU\")\n        strategy = tf.distribute.get_strategy()\n        CFG.device = \"CPU\"\n\nif CFG.device == \"GPU\":\n    print(\"Num GPUs Available: \", ngpu)\n    \n\nAUTO     = tf.data.experimental.AUTOTUNE\nREPLICAS = strategy.num_replicas_in_sync\nprint(f'REPLICAS: {REPLICAS}')","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# GCS Path for TPU\n* Remote-TPU requires **GCS** path. Luckily Kaggle Provides that for us :)\n* Local-TPU aka TPU-1VM doesn't need GCS path =)","metadata":{}},{"cell_type":"code","source":"BASE_PATH = '/kaggle/input/rsna-bcd-roi-1024x512-png-v2-dataset'\n\nif CFG.device==\"TPU\":\n    from kaggle_datasets import KaggleDatasets\n    GCS_PATH = KaggleDatasets().get_gcs_path(BASE_PATH.split('/')[-1])","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Meta Data","metadata":{}},{"cell_type":"code","source":"# use gcs_path for remote-tpu\nif CFG.device==\"TPU\":\n    BASE_PATH = GCS_PATH\n\n# train\ndf = pd.read_csv(f'{BASE_PATH}/train.csv')\ndf['image_path'] = f'{BASE_PATH}/train_images'\\\n                    + '/' + df.patient_id.astype(str)\\\n                    + '/' + df.image_id.astype(str)\\\n                    + '.png'\nprint('Train:')\ndisplay(df.head(2))\n\n# test\ntest_df = pd.read_csv(f'{BASE_PATH}/test.csv')\ntest_df['image_path'] = f'{BASE_PATH}/test_images'\\\n                    + '/' + test_df.patient_id.astype(str)\\\n                    + '/' + test_df.image_id.astype(str)\\\n                    + '.png'\nprint('\\nTest:')\ndisplay(test_df.head(2))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Check If Data Exist?","metadata":{}},{"cell_type":"code","source":"tf.io.gfile.exists(df.image_path.iloc[0]), tf.io.gfile.exists(test_df.image_path.iloc[0])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train-Test Ditribution","metadata":{}},{"cell_type":"code","source":"print('train_files:',df.shape[0])\nprint('test_files:',test_df.shape[0])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Split\n* Data is splited while stratifying, 'laterality', 'view', 'age','cancer', 'biopsy', 'invasive', 'BIRADS', 'implant', 'density','machine_id', 'difficult_negative_case','cancer'.\n* To avoid leakage, data is also split keeping images from same patient in either train or valid not in both.\n* `StratifiedGroupKFold` does the both job **stratifying** and **group** split.\n\n","metadata":{}},{"cell_type":"code","source":"num_bins = 5\ndf[\"age_bin\"] = pd.cut(df['age'].values.reshape(-1), bins=num_bins, labels=False)\n\nstrat_cols = [\n    'laterality', 'view', 'biopsy','invasive', 'BIRADS', 'age_bin',\n    'implant', 'density','machine_id', 'difficult_negative_case',\n    'cancer',\n]\n\ndf['stratify'] = ''\nfor col in strat_cols:\n    df['stratify'] += df[col].astype(str)\n\nskf = StratifiedGroupKFold(n_splits=CFG.folds, shuffle=True, random_state=CFG.seed)\nfor fold, (train_idx, val_idx) in enumerate(skf.split(df, df['stratify'], df[\"patient_id\"])):\n    df.loc[val_idx, 'fold'] = fold\ndisplay(df.groupby(['fold', \"cancer\"]).size())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Pipeline\n* Reads the raw file and then decodes it to tf.Tensor\n* Resizes the image in desired size\n* Chages the datatype to **float32**\n* Caches the Data for boosting up the speed.\n* Uses Augmentations to reduce overfitting and make model more robust.\n* Finally, splits the data into batches.\n","metadata":{}},{"cell_type":"code","source":"def build_decoder(with_labels=True, target_size=CFG.img_size, ext='png'):\n    def decode(path):\n        file_bytes = tf.io.read_file(path)\n        if ext == 'png':\n            img = tf.image.decode_png(file_bytes, channels=3)\n        elif ext in ['jpg', 'jpeg']:\n            img = tf.image.decode_jpeg(file_bytes, channels=3)\n        else:\n            raise ValueError(\"Image extension not supported\")\n\n        img = tf.image.resize(img, target_size, method='bilinear')\n        img = tf.cast(img, tf.float32) / 255.0\n        img = tf.reshape(img, [*target_size, 3])\n\n        return img\n    \n    def decode_with_labels(path, label):\n        return decode(path), tf.cast(label, tf.float32)\n    \n    return decode_with_labels if with_labels else decode\n\n\ndef build_augmenter(with_labels=True, dim=CFG.img_size):\n    def augment(img, dim=dim):\n        img = tf.image.random_flip_left_right(img) if CFG.hflip else img\n        img = tf.image.random_flip_up_down(img) if CFG.vflip else img\n        if CFG.pixel_aug:\n            img = tf.image.random_hue(img, CFG.hue)\n            img = tf.image.random_saturation(img, CFG.sat[0], CFG.sat[1])\n            img = tf.image.random_contrast(img, CFG.cont[0], CFG.cont[1])\n            img = tf.image.random_brightness(img, CFG.bri)\n        img = tf.clip_by_value(img, 0, 1)  if CFG.clip else img         \n        img = tf.reshape(img, [*dim, 3])\n        return img\n    \n    def augment_with_labels(img, label):    \n        return augment(img), label\n    \n    return augment_with_labels if with_labels else augment\n\n\ndef build_dataset(paths, labels=None, batch_size=32, cache=True,\n                  decode_fn=None, augment_fn=None,\n                  augment=True, repeat=True, shuffle=1024, \n                  cache_dir=\"\", drop_remainder=False):\n    if cache_dir != \"\" and cache is True:\n        os.makedirs(cache_dir, exist_ok=True)\n    \n    if decode_fn is None:\n        decode_fn = build_decoder(labels is not None)\n    \n    if augment_fn is None:\n        augment_fn = build_augmenter(labels is not None)\n    \n    AUTO = tf.data.experimental.AUTOTUNE\n    slices = paths if labels is None else (paths, labels)\n    \n    ds = tf.data.Dataset.from_tensor_slices(slices)\n    ds = ds.map(decode_fn, num_parallel_calls=AUTO)\n    ds = ds.cache(cache_dir) if cache else ds\n    ds = ds.repeat() if repeat else ds\n    if shuffle: \n        ds = ds.shuffle(shuffle, seed=CFG.seed)\n        opt = tf.data.Options()\n        opt.experimental_deterministic = False\n        ds = ds.with_options(opt)\n    ds = ds.map(augment_fn, num_parallel_calls=AUTO) if augment else ds\n    ds = ds.batch(batch_size, drop_remainder=drop_remainder)\n    ds = ds.prefetch(AUTO)\n    return ds","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Visualization\n* Check if augmentation is working properly or not.","metadata":{}},{"cell_type":"code","source":"def display_batch(batch, size=2):\n    if isinstance(batch, tuple):\n        imgs, tars = batch\n    else:\n        imgs = batch\n        tars = None\n    tars = tars.numpy().squeeze()\n    plt.figure(figsize=(size*2, 10))\n    for img_idx in range(size):\n        plt.subplot(1, size, img_idx+1)\n        if tars is not None:\n            plt.title(f'{CFG.target_col[0]}: {tars[img_idx]:0.3f}', fontsize=10)\n        plt.imshow(imgs[img_idx,:, :, :])\n        plt.xticks([]); plt.yticks([])\n    plt.tight_layout()\n    plt.show() ","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Generate Image","metadata":{}},{"cell_type":"code","source":"fold = 0\nfold_df = df.groupby('cancer').head(16)\npaths  = fold_df.image_path.tolist()\nlabels = fold_df[CFG.target_col].values\nds = build_dataset(paths, labels, cache=False, batch_size=32,\n                   repeat=True, shuffle=True, augment=False)\nds = ds.unbatch().batch(20)\nbatch = next(iter(ds))","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## No-Augmentation","metadata":{}},{"cell_type":"code","source":"display_batch(batch, 5);","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Metric\nMetric for this competition is **probabilistic F1 score (pF1)**. This extension of the traditional F score accepts probabilities instead of binary classifications,\n\n$$\npF1=2\\frac{pPrecision⋅pRecall}{pPrecision+pRecall}\n$$\n\nWhere,\n$$\npPrecision=\\frac{pTP}{pTP+pFP}\n$$\n$$\npRecall=\\frac{pTP}{TP+FN}\n$$","metadata":{}},{"cell_type":"code","source":"# tensorflow\ndef pFScore(labels, preds, beta=1):\n    eps = 1e-5\n    preds = tf.clip_by_value(preds, 0, 1)\n    y_true_count = tf.reduce_sum(labels)\n    ctp = tf.reduce_sum(preds[labels==1])\n    cfp = tf.reduce_sum(preds[labels==0])\n    beta_squared = beta * beta\n    c_precision = ctp / (ctp + cfp + eps)\n    c_recall = ctp / (y_true_count + eps)\n    if (c_precision > 0 and c_recall > 0):\n        result = (1 + beta_squared) * (c_precision * c_recall) / (beta_squared * c_precision + c_recall + eps)\n        return result\n    else:\n        return 0.0\npFScore.__name__='pF1'\n\n# numpy\ndef pfbeta(labels, preds, beta=1):\n    eps = 1e-5\n    preds = preds.clip(0, 1)\n    y_true_count = labels.sum()\n    ctp = preds[labels==1].sum()\n    cfp = preds[labels==0].sum()\n    beta_squared = beta * beta\n    c_precision = ctp / (ctp + cfp + eps)\n    c_recall = ctp / (y_true_count + eps)\n    if (c_precision > 0 and c_recall > 0):\n        result = (1 + beta_squared) * (c_precision * c_recall) / (beta_squared * c_precision + c_recall + eps)\n        return result\n    else:\n        return 0.0","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Build Model","metadata":{}},{"cell_type":"code","source":"!pip install gcvit --no-deps","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gcvit\n\ndef build_model(model_name=CFG.model_name,\n                loss_name=CFG.loss,\n                dim=CFG.img_size,\n                compile_model=True,\n                include_top=False):\n    model = getattr(gcvit, model_name)(input_shape=[*dim, 3],\n                                       pretrain=True)  # when pretrain=True, num_classes must be 1000\n    model.reset_classifier(num_classes=1,\n                           head_act='sigmoid')\n    if compile_model:\n        # optimizer\n        opt = tf.keras.optimizers.Adam(learning_rate=0.0001)\n        # loss\n        if loss_name == 'BCE':\n            loss = tf.keras.losses.BinaryCrossentropy(label_smoothing=0.05)\n        elif loss_name == 'Focal':\n            loss = tfa.losses.SigmoidFocalCrossEntropy(alpha=0.80, gamma=2.0)\n        # metric\n        auc = tf.keras.metrics.AUC(name='auc')\n        pf1 = pFScore\n        metrics = [pf1, auc]\n        # compile\n        model.compile(optimizer=opt,\n                      loss=loss,\n                      metrics=metrics)\n    return model","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model Check","metadata":{}},{"cell_type":"code","source":"tmp = build_model(CFG.model_name, dim=CFG.img_size, compile_model=True)\ntmp.summary()","metadata":{"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Learning-Rate Scheduler","metadata":{}},{"cell_type":"code","source":"def get_lr_callback(batch_size=8, plot=False):\n    lr_start   = 0.000005\n    lr_max     = 0.00000105 * REPLICAS * batch_size\n    lr_min     = 0.000001\n    lr_ramp_ep = 4\n    lr_sus_ep  = 0\n    lr_decay   = 0.8\n   \n    def lrfn(epoch):\n        if epoch < lr_ramp_ep:\n            lr = (lr_max - lr_start) / lr_ramp_ep * epoch + lr_start\n            \n        elif epoch < lr_ramp_ep + lr_sus_ep:\n            lr = lr_max\n            \n        elif CFG.scheduler=='exp':\n            lr = (lr_max - lr_min) * lr_decay**(epoch - lr_ramp_ep - lr_sus_ep) + lr_min\n            \n        elif CFG.scheduler=='cosine':\n            decay_total_epochs = CFG.epochs - lr_ramp_ep - lr_sus_ep + 3\n            decay_epoch_index = epoch - lr_ramp_ep - lr_sus_ep\n            phase = math.pi * decay_epoch_index / decay_total_epochs\n            cosine_decay = 0.4 * (1 + math.cos(phase))\n            lr = (lr_max - lr_min) * cosine_decay + lr_min\n        return lr\n    if plot:\n        plt.figure(figsize=(10,5))\n        plt.plot(np.arange(CFG.epochs), [lrfn(epoch) for epoch in np.arange(CFG.epochs)], marker='o')\n        plt.xlabel('epoch'); plt.ylabel('learnig rate')\n        plt.title('Learning Rate Scheduler')\n        plt.show()\n\n    lr_callback = tf.keras.callbacks.LearningRateScheduler(lrfn, verbose=False)\n    return lr_callback\n\n_=get_lr_callback(CFG.batch_size, plot=True )","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Wandb** Logger\nLog:\n* Best Score\n* Attention MAP","metadata":{}},{"cell_type":"markdown","source":"# Train Model\n* Cross-Validation: 5 fold\n* **WandB** dashboard is shown end of the each fold. So we don't need to plot anything. We can select best model from here.","metadata":{}},{"cell_type":"code","source":"CFG.debug = True\nCFG.wandb = False\nCFG.epochs = 5\nCFG.verbose = 1\nCFG.selected_folds = [0, 1]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof_pred = []; oof_tar = []; oof_val = []; oof_ids = []; oof_folds = []\npreds = np.zeros((test_df.shape[0],1))\n\nfor fold in np.arange(CFG.folds):\n    \n    # ignore not selected folds\n    if fold not in CFG.selected_folds:\n        continue\n            \n    # train and valid dataframe\n    train_df = df.query(\"fold!=@fold\")\n    valid_df = df.query(\"fold==@fold\")\n    \n    # upsample cancer data\n    pos_df = train_df.query(\"cancer==1\").sample(frac=CFG.upsample, replace=True)\n    neg_df = train_df.query(\"cancer==0\")\n    train_df = pd.concat([pos_df, neg_df], axis=0, ignore_index=True)\n    \n    # get image_paths and labels\n    train_paths = train_df.image_path.values; train_labels = train_df[CFG.target_col].values.astype(np.float32)\n    valid_paths = valid_df.image_path.values; valid_labels = valid_df[CFG.target_col].values.astype(np.float32)\n    test_paths  = test_df.image_path.values\n    \n    # shuffle train data\n    index = np.arange(len(train_df))\n    np.random.shuffle(index)\n    train_paths  = train_paths[index]\n    train_labels = train_labels[index]\n    \n    # min samples in debug mode\n    min_samples = CFG.batch_size*REPLICAS*2\n    \n    # for debug model run on small portion\n    if CFG.debug:\n        train_paths = train_paths[:min_samples]; train_labels = train_labels[:min_samples]\n        valid_paths = valid_paths[:min_samples]; valid_labels = valid_labels[:min_samples]\n    \n    # show message\n    print('#'*40); print('#### FOLD: ',fold)\n    print('#### IMAGE_SIZE: (%i, %i) | MODEL_NAME: %s | BATCH_SIZE: %i'%\n          (CFG.img_size[0],CFG.img_size[1],CFG.model_name,CFG.batch_size*REPLICAS))\n    num_train = len(train_paths)\n    num_valid = len(valid_paths)\n    print('#### NUM_TRAIN: {:,} | NUM_VALID: {:,}'.format(num_train, num_valid))\n    \n    # build model\n    K.clear_session()\n    with strategy.scope():\n        model = build_model(CFG.model_name, dim=CFG.img_size, compile_model=True)\n\n    # for gcvit resize layer\n    tf.config.set_soft_device_placement(True)\n    \n    # build dataset\n    cache = 'TPU' not in CFG.device\n    train_ds = build_dataset(train_paths, train_labels, cache=cache, batch_size=CFG.batch_size*REPLICAS,\n                   repeat=True, shuffle=True, augment=CFG.augment)\n    val_ds = build_dataset(valid_paths, valid_labels, cache=cache, batch_size=CFG.batch_size*REPLICAS,\n                   repeat=False, shuffle=False, augment=False)\n    print('#'*40)   \n    \n    # callbacks\n    callbacks = []\n    ## save best model after each fold\n    sv = tf.keras.callbacks.ModelCheckpoint(\n        'fold-%i.h5'%fold, monitor='val_pF1', verbose=CFG.verbose, save_best_only=True,\n        save_weights_only=True, mode='max', save_freq='epoch')\n    callbacks +=[sv]\n    ## lr-scheduler\n    callbacks += [get_lr_callback(CFG.batch_size)]\n        \n    # train\n    print('Training...')\n    history = model.fit(\n        train_ds, \n        epochs=CFG.epochs if not CFG.debug else 2, \n        callbacks = callbacks, \n        steps_per_epoch=len(train_paths)/CFG.batch_size//REPLICAS,\n        validation_data=val_ds, \n        #class_weight = {0:1,1:2},\n        verbose=CFG.verbose\n    )\n    \n    # load best model for inference\n    print('Loading best model...')\n    model.load_weights('fold-%i.h5'%fold)  \n    \n    # predict on valid data\n    print('Predicting OOF with TTA...')\n    ds_valid = build_dataset(valid_paths, labels=None, cache=False, batch_size=CFG.batch_size*REPLICAS*2,\n                   repeat=True, shuffle=False, augment=CFG.tta>1)\n    ct_valid = len(valid_paths); STEPS = CFG.tta * ct_valid/CFG.batch_size/2/REPLICAS\n    pred = model.predict(ds_valid,steps=STEPS,verbose=CFG.verbose)[:CFG.tta*ct_valid,] \n    oof_pred.append(np.mean(pred.reshape((CFG.tta, ct_valid,-1)),axis=0))                 \n    \n    # get id and target for valid data\n    oof_tar.append(valid_df[CFG.target_col].values[:(min_samples if CFG.debug else len(valid_df))])\n    oof_folds.append(np.ones_like(oof_tar[-1],dtype='int8')*fold)\n    oof_ids.append(valid_df.image_path.tolist()[:(min_samples if CFG.debug else len(valid_df))])\n    \n    # predict on test data\n    print('Predicting Test...')\n    ds_test = build_dataset(test_paths, labels=None, cache=False, \n                    batch_size=(CFG.batch_size*2 if len(test_df)>4 else 1)*REPLICAS,\n                   repeat=True, shuffle=False, augment=CFG.tta>1)\n    ct_test = len(test_paths); STEPS = 1 if len(test_df)<=4 else (CFG.tta * ct_test/CFG.batch_size/2/REPLICAS)\n    pred = model.predict(ds_test,steps=STEPS,verbose=CFG.verbose)[:CFG.tta*ct_test,] \n    preds[:ct_test, :] += np.mean(pred.reshape((CFG.tta, ct_test,-1)),axis=0) / CFG.folds # not meaningful for DIBUG = True\n    \n    # store best results\n    y_true = oof_tar[-1]; y_pred = oof_pred[-1]\n    pF1 = pfbeta(y_true.astype(np.float32), y_pred)\n    oof_val.append(np.max(history.history['val_pF1'] ))\n    print('>>>> FOLD %i OOF pF1_batch = %.3f, pF1 = %.3f\\n'%(fold,oof_val[-1], pF1))  # pF1_batch => pF1 batchwise canculated then aggregated","metadata":{"_kg_hide-input":true,"_kg_hide-output":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Calculate OOF Score","metadata":{}},{"cell_type":"code","source":"# overall oof pF1\noof = np.concatenate(oof_pred); true = np.concatenate(oof_tar);\nids = np.concatenate(oof_ids); folds = np.concatenate(oof_folds)\npF1 = pfbeta(true.astype(np.float32),oof)\nprint('Overall OOF pF1 = %.3f'%pF1)","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# save oof\ncolumns = ['image_path', 'true', 'pred']\ndf_oof = pd.DataFrame(np.concatenate([ids[:,None], true, oof], axis=1), columns=columns)\ndf_oof = df_oof.merge(df, on=['image_path'], how='left') # merge with train data\ndf_oof.to_csv('oof.csv',index=False)\ndf_oof.head(2)","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]}]}