{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install tensorflow --upgrade --quiet\n!yes | apt install --allow-change-held-packages libcudnn8=8.1.0.77-1+cuda11.2\n!pip install -q /lib/wheels/tensorflow-2.9.1-cp38-cp38-linux_x86_64.whl\n!pip install -q tensorflow-addons==0.18.0\n!pip install -q tensorflow-probability==0.17.0\n!pip install -q opencv-python-headless\n!pip install -q seaborn\n!pip install -q imgaug\n!pip install -q ipyplot","metadata":{"execution":{"iopub.status.busy":"2022-12-16T22:31:41.85208Z","iopub.execute_input":"2022-12-16T22:31:41.852739Z","iopub.status.idle":"2022-12-16T22:35:42.479708Z","shell.execute_reply.started":"2022-12-16T22:31:41.852623Z","shell.execute_reply":"2022-12-16T22:35:42.478287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -q efficientnet >> /dev/null\n!pip install -qU wandb\n!pip install -qU scikit-learn","metadata":{"execution":{"iopub.status.busy":"2022-12-16T22:35:42.48637Z","iopub.execute_input":"2022-12-16T22:35:42.489408Z","iopub.status.idle":"2022-12-16T22:36:18.039771Z","shell.execute_reply.started":"2022-12-16T22:35:42.48936Z","shell.execute_reply":"2022-12-16T22:36:18.038542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nfrom glob import glob\n\nimport tensorflow as tf\n\nfrom joblib import Parallel, delayed\n\nimport pandas as pd\nimport numpy as np\n\n# import imageio\n# import imgaug as ia\n# import imgaug.augmenters as iaa\n\n%matplotlib inline\n\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nfrom sklearn.model_selection import KFold, StratifiedKFold, GroupKFold, StratifiedGroupKFold\n","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-12-16T22:36:25.682536Z","iopub.execute_input":"2022-12-16T22:36:25.682929Z","iopub.status.idle":"2022-12-16T22:36:25.699007Z","shell.execute_reply.started":"2022-12-16T22:36:25.682896Z","shell.execute_reply":"2022-12-16T22:36:25.697826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model Configuration","metadata":{}},{"cell_type":"code","source":"class CFG:\n    competition = 'rsna-mammography'\n    wandb = True\n    \n    # device\n    device = \"TPU-1VM\"\n    seed = 123\n    \n    # paths \n    base_path = \"/kaggle/input/\"\n    work_path = \"/kaggle/working/\"\n    images_base_path = os.path.join(base_path, 'rsna-mammography-images-as-pngs')\n    # Specify the training directory and the testing directory\n    base_dir = os.path.join(base_path, 'rsna-breast-cancer-detection')\n    train_dir = os.path.join(base_path, 'rsna-breast-cancer-detection/train.csv')\n    test_dir = os.path.join(base_path, 'rsna-breast-cancer-detection/test.csv')\n    \n    train_images = os.path.join(base_path, 'rsna-mammography-images-as-pngs/images_as_pngs_512/train_images_processed_512')\n    test_images = os.path.join(base_path, 'rsna-breast-cancer-detection/test_images')\n\n    folds = 4\n    \n    selected_folds = [0, 1, 2, 3, 4]\n    \n    img_size = [200, 100]\n    \n    #batch size & epochs \n    batch_size = 20\n    epochs = 10\n    \n    # augmentation\n    augment = False\n    hflip = True\n    vflip = True\n    rot = True\n    crop = False\n    noise = True\n    contrast = False\n    distortions = False\n    clip =True\n    \n    upsample = 10\n    \n    # loss & optimizer\n    loss = 'Focal'\n    optimizer = 'Adam'\n    \n    # target_column\n    target_col = ['cancer']\n    \n    #verbose \n    verbose = 1\n    tta = 1\n\n    ","metadata":{"execution":{"iopub.status.busy":"2022-12-16T22:36:26.378731Z","iopub.execute_input":"2022-12-16T22:36:26.379498Z","iopub.status.idle":"2022-12-16T22:36:26.388406Z","shell.execute_reply.started":"2022-12-16T22:36:26.379461Z","shell.execute_reply":"2022-12-16T22:36:26.387187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Device Configuration","metadata":{}},{"cell_type":"code","source":"if \"TPU\" in CFG.device:\n    tpu = 'local' if CFG.device=='TPU-1VM' else None\n    print(\"connecting to TPU...\")\n    try:\n        tpu = tf.distribute.cluster_resolver.TPUClusterResolver.connect()\n        print('Device:', tpu.master())\n        tf.config.experimental_connect_to_cluster(tpu)\n        tf.tpu.experimental.initialize_tpu_system(tpu)\n        strategy = tf.distribute.TPUStrategy(tpu)\n    except:\n        CFG.device = \"GPU\"\n        \nif CFG.device == \"GPU\"  or CFG.device==\"CPU\":\n    ngpu = len(tf.config.experimental.list_physical_devices('GPU'))\n    if ngpu>1:\n        print(\"Using multi GPU\")\n        strategy = tf.distribute.MirroredStrategy()\n    elif ngpu==1:\n        print(\"Using single GPU\")\n        strategy = tf.distribute.get_strategy()\n    else:\n        print(\"Using CPU\")\n        strategy = tf.distribute.get_strategy()\n        CFG.device = \"CPU\"\n\nif CFG.device == \"GPU\":\n    print(\"Num GPUs Available: \", ngpu)\n    \n\nAUTO     = tf.data.experimental.AUTOTUNE\nREPLICAS = strategy.num_replicas_in_sync\nprint(f'REPLICAS: {REPLICAS}')","metadata":{"execution":{"iopub.status.busy":"2022-12-16T22:36:26.933224Z","iopub.execute_input":"2022-12-16T22:36:26.933619Z","iopub.status.idle":"2022-12-16T22:36:30.368237Z","shell.execute_reply.started":"2022-12-16T22:36:26.933586Z","shell.execute_reply":"2022-12-16T22:36:30.367227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# tpu = tf.distribute.cluster_resolver.TPUClusterResolver.connect('local')\n# print('Device:', tpu.master())\n# tf.config.experimental_connect_to_cluster(tpu)\n# tf.tpu.experimental.initialize_tpu_system(tpu)\n# strategy = tf.distribute.TPUStrategy(tpu)\n# AUTO     = tf.data.experimental.AUTOTUNE\n# REPLICAS = strategy.num_replicas_in_sync\n# print(f'REPLICAS: {REPLICAS}')","metadata":{"execution":{"iopub.status.busy":"2022-12-16T22:36:30.370268Z","iopub.execute_input":"2022-12-16T22:36:30.371016Z","iopub.status.idle":"2022-12-16T22:36:30.377561Z","shell.execute_reply.started":"2022-12-16T22:36:30.370971Z","shell.execute_reply":"2022-12-16T22:36:30.376493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Path to the 512px pngs\nIMG_PATH = CFG.images_base_path\nCSV_PATH = CFG.base_dir\n\ndef get_remote_gcs_path(PATH):\n    from kaggle_datasets import KaggleDatasets\n    return KaggleDatasets().get_gcs_path(PATH.split('/')[-1])\n\n\nif CFG.device==\"TPU\":\n    from kaggle_datasets import KaggleDatasets\n    GCS_IMG_PATH = get_remote_gcs_path(IMG_PATH)\n    GCS_CSV_PATH = get_remote_gcs_path(CSV_PATH)","metadata":{"execution":{"iopub.status.busy":"2022-12-16T22:36:30.379625Z","iopub.execute_input":"2022-12-16T22:36:30.380257Z","iopub.status.idle":"2022-12-16T22:36:30.388982Z","shell.execute_reply.started":"2022-12-16T22:36:30.380219Z","shell.execute_reply":"2022-12-16T22:36:30.388013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# use gcs_path for remote-tpu\nif CFG.device==\"TPU\":\n    BASE_CSV_PATH = GCS_CSV_PATH\n    BASE_IMG_PATH = GCS_IMG_PATH\nelse:\n    BASE_CSV_PATH = CSV_PATH\n    BASE_IMG_PATH = IMG_PATH\n    \n    \n# Train Images and test images paths\n# train_images = glob(f\"{CFG.train_images}/*/*.png\")\n# test_images = glob(f\"{CFG.test_images}/*/*.dcm\")\n\ntrain_df = pd.read_csv(f'{BASE_CSV_PATH}/train.csv')\ntrain_df['image_path'] = f'{BASE_IMG_PATH}/images_as_pngs_512/train_images_processed_512'\\\n                    + '/' + train_df.patient_id.astype(str)\\\n                    + '/' + train_df.image_id.astype(str)\\\n                    + '.png'\nprint('Train:')\ndisplay(train_df.head(2))\n\n# test\ntest_df = pd.read_csv(f'{BASE_CSV_PATH}/test.csv')\ntest_df['image_path'] = f'{BASE_CSV_PATH}/test_images'\\\n                    + '/' + test_df.patient_id.astype(str)\\\n                    + '/' + test_df.image_id.astype(str)\\\n                    + '.dcm'\nprint('\\nTest:')\ndisplay(test_df.head(2))","metadata":{"execution":{"iopub.status.busy":"2022-12-16T22:36:30.392147Z","iopub.execute_input":"2022-12-16T22:36:30.392675Z","iopub.status.idle":"2022-12-16T22:36:30.672864Z","shell.execute_reply.started":"2022-12-16T22:36:30.392634Z","shell.execute_reply":"2022-12-16T22:36:30.671873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train_df.iloc[:(len(train_df)//2)]","metadata":{"execution":{"iopub.status.busy":"2022-12-16T22:36:30.674584Z","iopub.execute_input":"2022-12-16T22:36:30.675012Z","iopub.status.idle":"2022-12-16T22:36:30.682635Z","shell.execute_reply.started":"2022-12-16T22:36:30.674971Z","shell.execute_reply":"2022-12-16T22:36:30.681481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## WANDB","metadata":{}},{"cell_type":"code","source":"import wandb\n\ntry:\n    from kaggle_secrets import UserSecretsClient\n    user_secrets = UserSecretsClient()\n    api_key = user_secrets.get_secret(\"WANDB\")\n\n    wandb.login(key=api_key)\n    anonymous = None\nexcept:\n    anonymous = \"must\"\n    print('To use your W&B account,\\nGo to Add-ons -> Secrets and provide your W&B access token. Use the Label name as WANDB. \\nGet your W&B access token from here: https://wandb.ai/authorize')","metadata":{"execution":{"iopub.status.busy":"2022-12-16T22:36:30.68421Z","iopub.execute_input":"2022-12-16T22:36:30.684891Z","iopub.status.idle":"2022-12-16T22:36:34.681942Z","shell.execute_reply.started":"2022-12-16T22:36:30.684854Z","shell.execute_reply":"2022-12-16T22:36:34.680646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[\"age_bin\"] = pd.cut(train_df.age.values.reshape(-1), bins=5, labels=False)\nbin_range = [(train_df[train_df[\"age_bin\"] == 1.0][\"age\"].min(), train_df[train_df[\"age_bin\"] == 1.0][\"age\"].max()),\n             (train_df[train_df[\"age_bin\"] == 2.0][\"age\"].min(), train_df[train_df[\"age_bin\"] == 2.0][\"age\"].max())]\nprint(f'Bin 1: {bin_range[0][0]} years to {bin_range[0][1]} years old', \"\\n\", f'Bin 2: {bin_range[1][0]} years to {bin_range[1][1]} years old')\n             ","metadata":{"execution":{"iopub.status.busy":"2022-12-16T22:36:34.68747Z","iopub.execute_input":"2022-12-16T22:36:34.689762Z","iopub.status.idle":"2022-12-16T22:36:34.740505Z","shell.execute_reply.started":"2022-12-16T22:36:34.689721Z","shell.execute_reply":"2022-12-16T22:36:34.739342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"strat_cols = ['laterality', 'view', 'age', 'cancer', 'biopsy', 'invasive', 'BIRADS', 'implant', 'density', 'machine_id', 'difficult_negative_case', 'age_bin', 'stratify']\n\n\ndef stratify_cols(df, strat_cols):\n    df['stratify'] = ''\n\n    for col in strat_cols:\n        df['stratify'] += df[col].astype(str)\n    return df\n\ntrain_df = stratify_cols(train_df, strat_cols)\n\nskf = StratifiedKFold(n_splits=CFG.folds, shuffle=True, random_state=CFG.seed)\nfor fold, (train_idx, val_idx) in enumerate(skf.split(train_df, train_df['stratify'], train_df['patient_id'])):\n    train_df.loc[val_idx, 'fold'] = fold\n\ndisplay(train_df.groupby(['fold', 'cancer']).size())","metadata":{"execution":{"iopub.status.busy":"2022-12-16T22:36:34.745409Z","iopub.execute_input":"2022-12-16T22:36:34.747891Z","iopub.status.idle":"2022-12-16T22:36:35.108479Z","shell.execute_reply.started":"2022-12-16T22:36:34.747846Z","shell.execute_reply":"2022-12-16T22:36:35.107372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Machine Learning Model","metadata":{}},{"cell_type":"markdown","source":"### Augmenter","metadata":{}},{"cell_type":"code","source":"# Could find way to rework or use one below for now\n# def build_augmenter(with_labels=True, dim=CFG.img_size):\n#     def augment_image(image: tf.Tensor, dim=dim, hflip=CFG.hflip, vflip=CFG.vflip, rot=CFG.rot, crop=CFG.crop, noise=CFG.noise, \n#                    contrast=CFG.contrast, distortion=CFG.distortions):\n#         import random\n#         print(type(image))\n#         #Horizontal Flip\n#         if hflip:\n#             hflip = iaa.Fliplr(p=0.5)\n#             hflipimage = hflip.augment_image(image)\n#             image = hflipimage\n\n#         #Vertical Flip\n#         if vflip:\n#             vflip = iaa.Flipud(p=0.5) \n#             vflipimage = vflip.augment_image(image)\n#             image = vflipimage\n\n#         # Rotation\n#         if rot:\n#             random_rotation = (random.randrange(-20, 20), random.randrange(-20, 20))\n#             rot = iaa.Affine(rotate=random_rotation)\n#             rotimage = rot.augment_image(image)\n#             image = rotimage\n\n#         # Cropping\n#         if crop:\n#             crop = iaa.Crop(percent=(random.uniform(0, 0.1), random.uniform(0, 0.1)))\n#             cropimage = crop.augment_image(image)\n#             image = cropimage\n\n#         # Noise\n#         if noise:\n#             random_noise1, random_noise2 = random.uniform(-.01, .01), random.uniform(.03, .045)\n#             noise = iaa.AdditiveGaussianNoise(random_noise1, random_noise2)\n#             noiseimage = noise.augment_image(image)\n#             image = noiseimage\n\n#         # Contrast- only apply one\n#         if contrast:\n#             random_contrast = random.randint(0, 2)\n#             if random_contrast == 0:\n#                 contrast_gam = iaa.GammaContrast((random.uniform(0.1, 1.0), random.uniform(1.0, 2.0)))\n#                 conimage = contrast_gam.augment_image(image)\n#                 image = conimage\n#             elif random_contrast == 1:\n#                 contrast_sig = iaa.SigmoidContrast(gain=(random.uniform(0.1, 5), random.uniform(5.1, 10)), cutoff=(random.uniform(0.1, 0.4), random.uniform(0.6, 1.0)))\n#                 conimage = contrast_sig.augment_image(image)\n#                 image = conimage\n#             elif random_contrast == 2:\n#                 contrast_lin = iaa.LinearContrast((random.uniform(0.1, 1.0), random.uniform(0.1, 1.0)))\n#                 conimage = contrast_lin.augment_image(image)\n#                 conimage = image\n\n#         # Distortions- only apply one  \n#         if distortion:\n#             random_distortion = random.randint(0, 2)\n#             if random_distortion == 0:\n#                 elastic = iaa.ElasticTransformation(alpha=random.uniform(0, 1.0), sigma=random.uniform(0.5, 1.0))\n#                 distimage = elastic.augment_image(image)\n#                 image = distimage\n#             elif random_distortion == 1:\n#                 polar = iaa.WithPolarWarping(iaa.CropAndPad(percent=(random.uniform(-.3, -0.01), random.uniform(0.0, 0.3))))\n#                 distimage = polar.augment_image(image)\n#                 image = distimage\n#             elif random_distortion ==  2:\n#                 jigsaw = iaa.Jigsaw(nb_rows=random.randrange(1, 5), nb_cols=random.randrange(1, 5), max_steps=(random.randrange(1,5),random.randrange(5,10)))\n#                 distimage = jigsaw.augment_image(image)\n#                 image = distimage\n#         image = tf.clip_by_value(image, 0, 1)  if CFG.clip else img         \n#         image = tf.reshape(image, [*dim, 3])\n#         return image\n    \n#     def augment_with_labels(image, label):    \n#         return augment_image(image), label\n    \n#     return augment_with_labels if with_labels else augment_image","metadata":{"execution":{"iopub.status.busy":"2022-12-16T22:36:35.112571Z","iopub.execute_input":"2022-12-16T22:36:35.112903Z","iopub.status.idle":"2022-12-16T22:36:35.12323Z","shell.execute_reply.started":"2022-12-16T22:36:35.112872Z","shell.execute_reply":"2022-12-16T22:36:35.122042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_augmenter(with_labels=True, dim=CFG.img_size):\n    def augment_image(image: tf.Tensor, dim=dim, hflip=CFG.hflip, vflip=CFG.vflip, rot=CFG.rot, crop=CFG.crop, noise=CFG.noise, \n                   color=CFG.contrast):\n        import random\n        print(type(image))\n        #Horizontal Flip\n        if hflip:\n            image = tf.image.random_flip_left_right(image)\n            \n        #Vertical Flip\n        if vflip:\n            image = tf.image.random_flip_up_down(image)\n\n        # Rotation\n        if rot:\n            image = tf.image.rot90(image, tf.random.uniform(shape=[], minval=0, maxval=4, dtype=tf.int32))\n\n\n        # Cropping\n        if crop:\n            scales = list(np.arange(0.8, 1.0, 0.01))\n            boxes = np.zeros((len(scales), 4))\n            image = tf.image.crop_and_resize([image], boxes=boxes, box_ind=np.zeros(len(scales)), crop_size=(32, 32))\n\n        # Color\n        if color:\n            image = tf.image.random_hue(image, random.uniform(0.05, 0.1))\n            image = tf.image.random_saturation(image, random.uniform(0.5, 0.7), random.uniform(1.5,1.75))\n            image = tf.image.random_brightness(image, random.uniform(0.03, 0.08))\n            image = tf.image.random_contrast(image, random.uniform(0.5, 0.8), random.uniform(1.0, 1.5))\n        image = tf.clip_by_value(image, 0, 1)  if CFG.clip else img         \n        image = tf.reshape(image, [*dim, 3])\n        return image\n    \n    def augment_with_labels(image, label):    \n        return augment_image(image), label\n    \n    return augment_with_labels if with_labels else augment_image","metadata":{"execution":{"iopub.status.busy":"2022-12-16T22:36:35.125153Z","iopub.execute_input":"2022-12-16T22:36:35.125918Z","iopub.status.idle":"2022-12-16T22:36:35.157338Z","shell.execute_reply.started":"2022-12-16T22:36:35.125867Z","shell.execute_reply":"2022-12-16T22:36:35.156073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_decoder(with_labels=True, target_size=CFG.img_size, ext='png'):\n    def decode(path):\n        file_bytes = tf.io.read_file(path)\n        if ext == 'png':\n            img = tf.image.decode_png(file_bytes, channels=3)\n        elif ext == 'png':\n            img = tf.image.decode_jpeg(file_bytes, channels=3)\n        else:\n            raise ValueError(\"Image Extension Unsupported\")\n            \n        img = tf.image.resize(img, target_size)\n        img = tf.cast(img, tf.float32) / 255.0\n        img = tf.reshape(img, [*target_size, 3])\n        \n        return img\n    \n    def decode_with_labels(path, label):\n        return decode(path), tf.cast(label, tf.float32)\n    return decode_with_labels if with_labels else decode\n\ndef build_dataset(paths, labels=None, batch_size=CFG.batch_size, cache=True,\n                 decode_fn=None, augment_fn=None, augment=False, repeat=True, shuffle=1024,\n                 cache_dir=\"\", drop_remainder=False):\n    if cache_dir != \"\" and cache is True:\n        os.makedirs(cache_dir, exist_ok=True)\n    \n    if decode_fn is None:\n        decode_fn = build_decoder(labels is not None)\n        \n    if augment_fn is None and augment is True:\n        augment_fn = build_augmenter(labels is not None)\n    \n    AUTO = tf.data.experimental.AUTOTUNE\n    slices = paths if labels is None else (paths, labels)\n    \n    ds = tf.data.Dataset.from_tensor_slices(slices)\n    ds = ds.map(decode_fn, num_parallel_calls=AUTO)\n    ds = ds.cache(cache_dir) if cache else ds\n    ds = ds.repeat() if repeat else ds\n    \n    if shuffle:\n        ds = ds.shuffle(shuffle, seed=CFG.seed)\n        opt = tf.data.Options()\n        opt.experimental_deterministic = False\n        ds = ds.with_options(opt)\n    \n    ds = ds.map(augment_fn, num_parallel_calls=AUTO) if augment else ds\n    \n    ds = ds.batch(batch_size, drop_remainder=drop_remainder)\n    ds = ds.prefetch(AUTO)\n    return ds","metadata":{"execution":{"iopub.status.busy":"2022-12-16T22:36:35.162665Z","iopub.execute_input":"2022-12-16T22:36:35.165147Z","iopub.status.idle":"2022-12-16T22:36:35.182512Z","shell.execute_reply.started":"2022-12-16T22:36:35.165109Z","shell.execute_reply":"2022-12-16T22:36:35.181866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Display","metadata":{}},{"cell_type":"code","source":"def display_batch(batch, size=2):\n    if isinstance(batch, tuple):\n        imgs, tars = batch\n    else:\n        imgs = batch\n        tars = None\n    \n    tars = tars.numpy().squeeze()\n    plt.figure(figsize=(size*2, 10))\n    for img_idx in range(size):\n        plt.subplot(1, size, img_idx+1)\n        if tars is not None:\n            plt.title(f'{CFG.target_col[0]}: {tars[img_idx]:0.3f}', fontsize=10)\n        plt.imshow(imgs[img_idx, :, :, :])\n        plt.xticks([]); plt.yticks([])\n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-16T22:36:35.183715Z","iopub.execute_input":"2022-12-16T22:36:35.184398Z","iopub.status.idle":"2022-12-16T22:36:35.204158Z","shell.execute_reply.started":"2022-12-16T22:36:35.184364Z","shell.execute_reply":"2022-12-16T22:36:35.203007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Generate Image","metadata":{}},{"cell_type":"code","source":"# fold = 0 \n# fold_df = train_df.groupby('cancer').head(16)\n# paths = fold_df.image_path.to_list()\n# labels = fold_df[CFG.target_col].values\n# ds = build_dataset(paths, labels, cache=False, repeat=True, shuffle=True, augment=True)\n# batch = next(iter(ds))","metadata":{"execution":{"iopub.status.busy":"2022-12-16T22:36:35.205538Z","iopub.execute_input":"2022-12-16T22:36:35.206404Z","iopub.status.idle":"2022-12-16T22:36:35.216743Z","shell.execute_reply.started":"2022-12-16T22:36:35.206369Z","shell.execute_reply":"2022-12-16T22:36:35.215868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# display_batch(batch, 5)","metadata":{"execution":{"iopub.status.busy":"2022-12-16T22:36:35.218033Z","iopub.execute_input":"2022-12-16T22:36:35.218476Z","iopub.status.idle":"2022-12-16T22:36:35.227359Z","shell.execute_reply.started":"2022-12-16T22:36:35.218442Z","shell.execute_reply":"2022-12-16T22:36:35.226464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras import Model, layers\nfrom tensorflow.keras.layers import Input, Convolution2D, MaxPool2D, BatchNormalization, Flatten, Dropout, Dense, GaussianNoise\nfrom tensorflow.keras.regularizers import l2\nfrom tensorflow.keras.activations import relu, sigmoid\nfrom tensorflow.keras.initializers import GlorotNormal\n# this configuration uses backend.set_image_data_format('channels_first')\n\n\"\"\"\nThis design creates the same network than before but using the layer by layer configuration.\nNotice the Input layer and each layer description.\nFunction returns a model which require inputs and outputs (could be multiple of each one)\n\"\"\"\ndef get_model_design(filters: list, input_shape: tuple) -> Model:\n    input_layer = Input(shape=input_shape)\n    \n    conv1_layer = Convolution2D(filters[0], (5, 5), padding='same', kernel_regularizer=l2(0.001), activation=relu)(input_layer)\n    conv2_layer = Convolution2D(filters[1], (3, 3), padding='same', kernel_regularizer=l2(0.001), activation=relu)(conv1_layer)\n    conv3_layer = Convolution2D(filters[2], (3, 3), padding='same', kernel_regularizer=l2(0.001), activation=relu)(conv2_layer)\n\n    maxpool1_layer = MaxPool2D(pool_size=(2, 2))(conv3_layer)\n    norm1_layer = BatchNormalization()(maxpool1_layer)\n\n    flat1_layer = Flatten()(norm1_layer)\n    drop1_layer = Dropout(0.5)(flat1_layer)\n    pred_layer = Dense(1, kernel_initializer=GlorotNormal(), activation=sigmoid)(drop1_layer)\n\n    model = Model(inputs=input_layer, outputs=pred_layer)\n    return model\n\n# for this example, we used 128 and 64 filters for the two first conv layers\n# note the input size of 3 channel for an image size of 64x64 pixels\nmodel = get_model_design([128, 64, 32], (200, 100, 3))\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-12-16T22:36:35.228718Z","iopub.execute_input":"2022-12-16T22:36:35.229526Z","iopub.status.idle":"2022-12-16T22:36:35.37864Z","shell.execute_reply.started":"2022-12-16T22:36:35.229488Z","shell.execute_reply":"2022-12-16T22:36:35.377871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.losses import BinaryCrossentropy\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.metrics import TruePositives, FalsePositives, TrueNegatives, FalseNegatives, BinaryAccuracy, Precision, Recall, AUC\nfrom tensorflow.keras.metrics import SpecificityAtSensitivity\n\n\"\"\"\nDefinition of metrics commonly used on medical imaging classification, segmentation, and localization problems.\nThe metrics will appear on each iteration of the training process to monitor the progress of our design.\n\"\"\"\nMETRICS = [\n      TruePositives(name='tp'),\n      FalsePositives(name='fp'),\n      TrueNegatives(name='tn'),\n      FalseNegatives(name='fn'), \n      BinaryAccuracy(name='accuracy'),\n      Precision(name='precision'),\n      Recall(name='recall'),\n      AUC(name='auc'),\n      SpecificityAtSensitivity(sensitivity=0.8, name='sensitivity'),\n]\n\n\"\"\"\nFor example, the loss function is to determine is an image contains or not a lesion/disease using the binary cross-entropy loss.\nThe optimizer is a first-order gradient-based optimization\n\"\"\"\nmodel.compile(loss=BinaryCrossentropy(),\n              optimizer=Adam(learning_rate=1e-3, beta_1=0.92, beta_2=0.999),\n              metrics=METRICS)","metadata":{"execution":{"iopub.status.busy":"2022-12-16T22:36:35.379688Z","iopub.execute_input":"2022-12-16T22:36:35.380046Z","iopub.status.idle":"2022-12-16T22:36:35.441817Z","shell.execute_reply.started":"2022-12-16T22:36:35.38001Z","shell.execute_reply":"2022-12-16T22:36:35.440718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.callbacks import EarlyStopping\n\n\"\"\"\nThis callback will stop the training when there is no improvement in the validation accuracy across epochs\n\"\"\"\nearly_callback = EarlyStopping(monitor='val_auc', \n                               verbose=1,\n                               patience=10,\n                               mode='max',\n                               restore_best_weights=True)","metadata":{"execution":{"iopub.status.busy":"2022-12-16T22:36:35.445906Z","iopub.execute_input":"2022-12-16T22:36:35.446303Z","iopub.status.idle":"2022-12-16T22:36:35.453714Z","shell.execute_reply.started":"2022-12-16T22:36:35.446272Z","shell.execute_reply":"2022-12-16T22:36:35.452771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_augmenter(with_labels=True, dim=CFG.img_size):\n    def augment_image(image, hflip=CFG.hflip, vflip=CFG.vflip, rot=CFG.rot, crop=CFG.crop, noise=CFG.noise, \n                   contrast=CFG.contrast, distortion=CFG.distortions):\n        import random\n        #Horizontal Flip\n        if hflip:\n            hflip = iaa.Fliplr(p=0.5)\n            hflipimage = hflip.augment_image(image)\n            image = hflipimage\n\n        #Vertical Flip\n        if vflip:\n            vflip = iaa.Flipud(p=0.5) \n            vflipimage = vflip.augment_image(image)\n            image = vflipimage\n\n        # Rotation\n        if rot:\n            random_rotation = (random.randrange(-20, 20), random.randrange(-20, 20))\n            rot = iaa.Affine(rotate=random_rotation)\n            rotimage = rot.augment_image(image)\n            image = rotimage\n\n        # Cropping\n        if crop:\n            crop = iaa.Crop(percent=(random.uniform(0, 0.1), random.uniform(0, 0.1)))\n            cropimage = crop.augment_image(image)\n            image = cropimage\n\n        # Noise\n        if noise:\n            random_noise1, random_noise2 = random.uniform(-.01, .01), random.uniform(.03, .045)\n            noise = iaa.AdditiveGaussianNoise(random_noise1, random_noise2)\n            noiseimage = noise.augment_image(image)\n            image = noiseimage\n\n        # Contrast- only apply one\n        if contrast:\n            random_contrast = random.randint(0, 2)\n            if random_contrast == 0:\n                contrast_gam = iaa.GammaContrast((random.uniform(0.1, 1.0), random.uniform(1.0, 2.0)))\n                conimage = contrast_gam.augment_image(image)\n                image = conimage\n            elif random_contrast == 1:\n                contrast_sig = iaa.SigmoidContrast(gain=(random.uniform(0.1, 5), random.uniform(5.1, 10)), cutoff=(random.uniform(0.1, 0.4), random.uniform(0.6, 1.0)))\n                conimage = contrast_sig.augment_image(image)\n                image = conimage\n            elif random_contrast == 2:\n                contrast_lin = iaa.LinearContrast((random.uniform(0.1, 1.0), random.uniform(0.1, 1.0)))\n                conimage = contrast_lin.augment_image(image)\n                conimage = image\n\n        # Distortions- only apply one  \n        if distortion:\n            random_distortion = random.randint(0, 2)\n            if random_distortion == 0:\n                elastic = iaa.ElasticTransformation(alpha=random.uniform(0, 1.0), sigma=random.uniform(0.5, 1.0))\n                distimage = elastic.augment_image(image)\n                image = distimage\n            elif random_distortion == 1:\n                polar = iaa.WithPolarWarping(iaa.CropAndPad(percent=(random.uniform(-.3, -0.01), random.uniform(0.0, 0.3))))\n                distimage = polar.augment_image(image)\n                image = distimage\n            elif random_distortion ==  2:\n                jigsaw = iaa.Jigsaw(nb_rows=random.randrange(1, 5), nb_cols=random.randrange(1, 5), max_steps=(random.randrange(1,5),random.randrange(5,10)))\n                distimage = jigsaw.augment_image(image)\n                image = distimage\n        return image\n    \n    def augment_with_labels(img, label):    \n        return augment(img), label\n    \n    return augment_with_labels if with_labels else augment","metadata":{"execution":{"iopub.status.busy":"2022-12-16T22:36:35.725557Z","iopub.execute_input":"2022-12-16T22:36:35.725964Z","iopub.status.idle":"2022-12-16T22:36:35.744396Z","shell.execute_reply.started":"2022-12-16T22:36:35.725931Z","shell.execute_reply":"2022-12-16T22:36:35.743483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size = 20\nAUTOTUNE = tf.data.AUTOTUNE\n\ndata_augmentation = tf.keras.Sequential([\n  layers.RandomFlip(\"horizontal_and_vertical\"),\n  layers.RandomRotation(0.2),\n])\n\ndef prepare(ds, shuffle=False, augment=False):\n  if shuffle:\n    ds = ds.shuffle(1000)\n\n  # Batch all datasets.\n    ds = ds.batch(batch_size)\n\n  # Use data augmentation only on the training set.\n  if augment:\n    ds = ds.map(lambda x, y: (data_augmentation(x, training=True), y), \n                num_parallel_calls=AUTOTUNE)\n\n  # Use buffered prefetching on all datasets.\n  return ds.prefetch(buffer_size=AUTOTUNE)","metadata":{"execution":{"iopub.status.busy":"2022-12-16T22:36:36.25659Z","iopub.execute_input":"2022-12-16T22:36:36.257315Z","iopub.status.idle":"2022-12-16T22:36:36.273116Z","shell.execute_reply.started":"2022-12-16T22:36:36.257278Z","shell.execute_reply":"2022-12-16T22:36:36.272098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# training and validation dataframes\ntrain_df = train_df.query(\"fold!=@fold\")\nvalid_df = train_df.query(\"fold==@fold\")\n\n# upsample the cancer data\npos_df = train_df.query(\"cancer==1\").sample(frac=CFG.upsample, replace=True)\nneg_df = train_df.query(\"cancer==0\")\ntrain_df = pd.concat([pos_df, neg_df], axis=0, ignore_index=True)\n\n# get image_paths and labels\ntrain_paths = train_df.image_path.values; train_labels = train_df[CFG.target_col].values.astype(np.float32)\nvalid_paths = valid_df.image_path.values; valid_labels = valid_df[CFG.target_col].values.astype(np.float32)\n\n# shuffle training data\nindex = np.arange(len(train_df))\nnp.random.shuffle(index)\ntrain_paths = train_paths[index]\ntrain_labels = train_labels[index]\n\n# build dataset\nprint(\"Building datasets....\")\ntrain_ds = build_dataset(train_paths, train_labels, batch_size=CFG.batch_size,\n                        repeat=True, shuffle=True, augment=False)\nval_ds = build_dataset(valid_paths, valid_labels, batch_size=CFG.batch_size,\n                      repeat=False, shuffle=False, augment=False)\n\nprint(\"#\"*40)\n\n# callbacks\ncallbacks = []\n## save best model after each fold\nsv = tf.keras.callbacks.ModelCheckpoint(\n    'fold-%i.h5'%fold, monitor='val_pF1', verbose=CFG.verbose, save_best_only=True,\n    save_weights_only=False, mode='max', save_freq='epoch')\ncallbacks +=[sv]\n\ntrain_ds = prepare(train_ds)\nval_ds = prepare(val_ds)\n\n# train\nprint('Training...')\nhistory = model.fit(\n    train_ds, \n    epochs=4, \n    callbacks = callbacks, \n    steps_per_epoch=len(train_paths)/CFG.batch_size//REPLICAS,\n    validation_data=val_ds, \n    verbose=CFG.verbose\n)\n\n\nmodel.save('model_1.h5')\n    ","metadata":{"execution":{"iopub.status.busy":"2022-12-16T22:36:36.89416Z","iopub.execute_input":"2022-12-16T22:36:36.89485Z","iopub.status.idle":"2022-12-16T22:41:00.798257Z","shell.execute_reply.started":"2022-12-16T22:36:36.894787Z","shell.execute_reply":"2022-12-16T22:41:00.797137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# oof_pred, oof_tar, oof_val, oof_ids, oof_folds = [], [], [], [], []\n\n# preds = np.zeros((test_df.shape[0], 1))\n\n# for fold in np.arange(CFG.folds):\n#     if fold not in CFG.selected_folds:\n#         continue\n    \n#     # training and validation dataframes\n#     train_df = train_df.query(\"fold!=@fold\")\n#     valid_df = train_df.query(\"fold==@fold\")\n    \n#     # upsample the cancer data\n#     pos_df = train_df.query(\"cancer==1\").sample(frac=CFG.upsample, replace=True)\n#     neg_df = train_df.query(\"cancer==0\")\n#     train_df = pd.concat([pos_df, neg_df], axis=0, ignore_index=True)\n    \n#     # get image_paths and labels\n#     train_paths = train_df.image_path.values; train_labels = train_df[CFG.target_col].values.astype(np.float32)\n#     valid_paths = valid_df.image_path.values; valid_labels = valid_df[CFG.target_col].values.astype(np.float32)\n    \n#     # shuffle training data\n#     index = np.arange(len(train_df))\n#     np.random.shuffle(index)\n#     train_paths = train_paths[index]\n#     train_labels = train_labels[index]\n    \n#     # build dataset\n#     print(\"Building datasets....\")\n#     train_ds = build_dataset(train_paths, train_labels, batch_size=CFG.batch_size,\n#                             repeat=True, shuffle=True, augment=False)\n#     val_ds = build_dataset(valid_paths, valid_labels, batch_size=CFG.batch_size,\n#                           repeat=False, shuffle=False, augment=False)\n\n#     print(\"#\"*40)\n    \n#     # callbacks\n#     callbacks = []\n#     ## save best model after each fold\n#     sv = tf.keras.callbacks.ModelCheckpoint(\n#         'fold-%i.h5'%fold, monitor='val_pF1', verbose=CFG.verbose, save_best_only=True,\n#         save_weights_only=False, mode='max', save_freq='epoch')\n#     callbacks +=[sv]\n    \n#     train_ds = prepare(train_ds)\n#     val_ds = prepare(val_ds)\n    \n#     # train\n#     print('Training...')\n#     history = model.fit(\n#         train_ds, \n#         epochs=4, \n#         callbacks = callbacks, \n#         steps_per_epoch=len(train_paths)/CFG.batch_size//REPLICAS,\n#         validation_data=val_ds, \n#         verbose=CFG.verbose\n#     )\n    \n\n#     model.save('model_1.h5')\n    ","metadata":{"execution":{"iopub.status.busy":"2022-12-16T22:24:29.375811Z","iopub.execute_input":"2022-12-16T22:24:29.376206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#     # Load best model \n#     print(\"Loading best model....\")\n#     model.load_weights('fold-%i-h5'%fold)\n    \n#     # predict on valid data\n#     print('Predicting OOF with TTA...')\n#     ds_valid = build_dataset(valid_paths, labels=None, cache=False, batch_size=CFG.batch_size,\n#                    repeat=True, shuffle=False, augment=CFG.tta>1)\n#     ct_valid = len(valid_paths); STEPS = CFG.tta * ct_valid/CFG.batch_size/2/REPLICAS\n#     pred = model.predict(ds_valid,steps=STEPS,verbose=CFG.verbose)[:CFG.tta*ct_valid,] \n#     oof_pred.append(np.mean(pred.reshape((CFG.tta, ct_valid,-1)),axis=0))                 \n    \n#     # get id and target for valid data\n#     oof_tar.append(valid_df[CFG.target_col].values[:(min_samples if CFG.debug else len(valid_df))])\n#     oof_folds.append(np.ones_like(oof_tar[-1],dtype='int8')*fold)\n#     oof_ids.append(valid_df.image_path.tolist()[:(min_samples if CFG.debug else len(valid_df))])\n    \n#     # predict on test data\n#     print('Predicting Test...')\n#     ds_test = build_dataset(test_paths, labels=None, cache=False, \n#                     batch_size=(CFG.batch_size*2 if len(test_df)>4 else 1),\n#                    repeat=True, shuffle=False, augment=CFG.tta>1)\n#     ct_test = len(test_paths); STEPS = 1 if len(test_df)<=4 else (CFG.tta * ct_test/CFG.batch_size/2)\n#     pred = model.predict(ds_test,steps=STEPS,verbose=CFG.verbose)[:CFG.tta*ct_test,] \n#     preds[:ct_test, :] += np.mean(pred.reshape((CFG.tta, ct_test,-1)),axis=0) / CFG.folds # not meaningful for DIBUG = True\n    \n#     # store best results\n#     y_true = oof_tar[-1]; y_pred = oof_pred[-1]\n#     pF1 = pfbeta(y_true.astype(np.float32), y_pred)\n#     oof_val.append(np.max(history.history['val_pF1'] ))\n#     print('#### FOLD %i OOF pF1_batch = %.3f, pF1 = %.3f'%(fold,oof_val[-1], pF1))  # pF1_batch => pF1 batchwise canculated then aggregated\n    ","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Predicting on the Test Data","metadata":{}},{"cell_type":"code","source":"!pip install dicomsdl\n\n# Thanks!!!\n# https://www.kaggle.com/code/radek1/how-to-process-dicom-images-to-pngs?scriptVersionId=113227375\n\nimport pydicom\nimport numpy as np\nimport cv2\nimport os\nfrom joblib import Parallel, delayed\nfrom tqdm.notebook import tqdm\nfrom pathlib import Path\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nimport dicomsdl\n\nRESIZE_TO = (200, 100)\n\n!rm -rf train_images_processed_cv2_dicomsdl_{RESIZE_TO[0]}\n!mkdir train_images_processed_cv2_dicomsdl_{RESIZE_TO[0]}\n\n# https://www.kaggle.com/code/tanlikesmath/brain-tumor-radiogenomic-classification-eda/notebook\ndef dicom_file_to_ary(path):\n    dcm_file = dicomsdl.open(str(path))\n    data = dcm_file.pixelData()\n\n    data = (data - data.min()) / (data.max() - data.min())\n\n    if dcm_file.getPixelDataInfo()['PhotometricInterpretation'] == \"MONOCHROME1\":\n        data = 1 - data\n\n    data = cv2.resize(data, RESIZE_TO)\n    data = (data * 255).astype(np.uint8)\n    return data\n\ndirectories = list(Path('/kaggle/input/rsna-breast-cancer-detection/test_images').iterdir())\n\ndef process_directory(directory_path):\n    parent_directory = str(directory_path).split('/')[-1]\n    !mkdir -p train_images_processed_cv2_dicomsdl_{RESIZE_TO[0]}/{parent_directory}\n    for image_path in directory_path.iterdir():\n        processed_ary = dicom_file_to_ary(image_path)\n        \n        cv2.imwrite(\n            f'train_images_processed_cv2_dicomsdl_{RESIZE_TO[0]}/{parent_directory}/{image_path.stem}.png',\n            processed_ary\n        )\n    print(\"Done!!!\")\n        \nimport multiprocessing as mp\n\nwith mp.Pool(64) as p:\n    p.map(process_directory, directories)\n    ","metadata":{"execution":{"iopub.status.busy":"2022-12-16T23:05:28.115075Z","iopub.execute_input":"2022-12-16T23:05:28.115511Z","iopub.status.idle":"2022-12-16T23:05:58.302131Z","shell.execute_reply.started":"2022-12-16T23:05:28.115479Z","shell.execute_reply":"2022-12-16T23:05:58.300264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm -rf /kaggle/working/test_images\n!mkdir /kaggle/working/test_images\nfrom PIL import Image\nfor image_path in test_df.image_path:\n    image = dicom_file_to_ary(image_path)\n    im = Image.fromarray(image)\n    im.save(f\"/kaggle/working/test_images{image_path.split('/')[-1]}.png\")   ","metadata":{"execution":{"iopub.status.busy":"2022-12-16T23:09:31.773072Z","iopub.execute_input":"2022-12-16T23:09:31.773471Z","iopub.status.idle":"2022-12-16T23:09:37.179355Z","shell.execute_reply.started":"2022-12-16T23:09:31.773437Z","shell.execute_reply":"2022-12-16T23:09:37.177854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset = tf.data.Dataset.from_tensor_slices(test_images)","metadata":{"execution":{"iopub.status.busy":"2022-12-16T22:58:00.49486Z","iopub.execute_input":"2022-12-16T22:58:00.495895Z","iopub.status.idle":"2022-12-16T22:58:00.525229Z","shell.execute_reply.started":"2022-12-16T22:58:00.495845Z","shell.execute_reply":"2022-12-16T22:58:00.524255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"checkpoint_path = \"/kaggle/working/model_1.h5\"\n# Create basic instance of model\nmodel = get_model_design([128, 64, 32], (200, 100, 3))\n# Loads the weights\nmodel.load_weights(checkpoint_path)","metadata":{"execution":{"iopub.status.busy":"2022-12-16T22:58:03.854528Z","iopub.execute_input":"2022-12-16T22:58:03.85505Z","iopub.status.idle":"2022-12-16T22:58:03.938978Z","shell.execute_reply.started":"2022-12-16T22:58:03.855013Z","shell.execute_reply":"2022-12-16T22:58:03.937872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_images[0][0]","metadata":{"execution":{"iopub.status.busy":"2022-12-16T23:01:05.406215Z","iopub.execute_input":"2022-12-16T23:01:05.406583Z","iopub.status.idle":"2022-12-16T23:01:05.41545Z","shell.execute_reply.started":"2022-12-16T23:01:05.406553Z","shell.execute_reply":"2022-12-16T23:01:05.414346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = model.predict(test_dataset)\npreds","metadata":{"execution":{"iopub.status.busy":"2022-12-16T22:58:14.27835Z","iopub.execute_input":"2022-12-16T22:58:14.278902Z","iopub.status.idle":"2022-12-16T22:58:14.405599Z","shell.execute_reply.started":"2022-12-16T22:58:14.278864Z","shell.execute_reply":"2022-12-16T22:58:14.403093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_predsdf = pd.DataFrame(preds, columns=[\"pneg\", \"ppos\"])\nfinal_test_df = pd.concat([test_df, test_predsdf], axis = 1)\nfinal_test_df.drop(['test', 'prediction_id'], inplace=True, axis=1)\nfinal_test_df","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame(data={'prediction_id': test_df.prediction_id, 'cancer': final_test_df.ppos}).reset_index(drop=True)\nsubmission = submission.sort_values('cancer', ascending=False).drop_duplicates(['prediction_id']).reset_index(drop=True)\nsubmission.head()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Plot Metrics","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom matplotlib import rcParams\n\nrcParams['figure.figsize'] = (12, 10)\ncolors = plt.rcParams['axes.prop_cycle'].by_key()['color']\n\ndef plot_log_loss(history: history, title_label: str, n: int) -> ():\n    # Use a log scale to show the wide range of values.\n    plt.semilogy(history.epoch,  history.history['loss'],\n               color=colors[n], label='Train '+title_label)\n    plt.semilogy(history.epoch,  history.history['val_loss'],\n          color=colors[n], label='Val '+title_label,\n          linestyle=\"--\")\n    plt.xlabel('Epoch')\n    plt.ylabel('Loss')\n\n    plt.legend()\n\nplot_log_loss(history, \"Model Base\", 1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_metrics(history: history) -> ():\n    metrics =  ['loss', 'precision', 'recall', 'auc', 'tp', 'sensitivity']\n    for n, metric in enumerate(metrics):\n        name = metric.replace(\"_\",\" \").capitalize()\n        plt.subplot(3, 2, n+1)  # adjust according to metrics\n        plt.plot(history.epoch,  history.history[metric], color=colors[0], label='Train')\n        plt.plot(history.epoch, history.history['val_'+metric],\n                 color=colors[0], linestyle=\"--\", label='Val')\n        plt.xlabel('Epoch')\n        plt.ylabel(name)\n        # selecting the metric, the value of plt.ylim could be changed\n    plt.legend()\n\nplot_metrics(history)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluate the model on the test data using `evaluate`\nprint(\"Evaluate on test data\")\nscore_test = model.evaluate(test_ds.batch(batch_size))\nfor name, value in zip(model.metrics_names, score_test):\n    print(name, ': ', value)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\nimport seaborn as sns\n\n# notice the threshold\ndef plot_cm(labels: numpy.ndarray, predictions: numpy.ndarray, p: float=0.5) -> ():\n    cm = confusion_matrix(labels, predictions > p)\n    # you can normalize the confusion matrix\n\n    plt.figure(figsize=(5,5))\n    sns.heatmap(cm, annot=True, fmt=\"d\")\n    plt.title('Confusion matrix @{:.2f}'.format(p))\n    plt.ylabel('Actual label')\n    plt.xlabel('Predicted label')\n\n    print('Lesions Detected (True Negatives): ', cm[0][0])\n    print('Lesions Incorrectly Detected (False Positives): ', cm[0][1])\n    print('No-Lesions Missed (False Negatives): ', cm[1][0])\n    print('No-Lesions Detected (True Positives): ', cm[1][1])\n    print('Total Lesions: ', np.sum(cm[1]))\n\nplot_cm(y_test, y_test_pred)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}