{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Baseline notebook for breast cancer detection, written in keras and Tensorflow.\n","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport seaborn as sns\nimport numpy as np\nimport tensorflow as tf\nimport os\nimport pandas as pd, numpy as np, random, shutil\nimport tensorflow as tf, re, math\nimport tensorflow.keras.backend as K\nimport sklearn\nimport matplotlib.pyplot as plt\nimport tensorflow_addons as tfa\nimport tensorflow_probability as tfp\nimport wandb\nimport yaml\n\nfrom IPython import display as ipd\nfrom glob import glob\nfrom tqdm import tqdm\nfrom sklearn.model_selection import KFold, StratifiedKFold, GroupKFold, StratifiedGroupKFold\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.utils.class_weight import compute_class_weight\n\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\nimport os\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom tqdm import tqdm\nfrom sklearn.utils import shuffle\nfrom sklearn.utils import class_weight\nfrom sklearn.preprocessing import minmax_scale\nimport random\nimport cv2\nfrom imgaug import augmenters as iaa\nimport warnings\nwarnings.filterwarnings('ignore')\n\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential, Model\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.layers import Dense, Dropout, Activation, Input, BatchNormalization, GlobalAveragePooling2D\nfrom tensorflow.keras import layers\nfrom tensorflow.keras.callbacks import ModelCheckpoint, ReduceLROnPlateau, EarlyStopping\nfrom tensorflow.keras.experimental import CosineDecay\nfrom tensorflow.keras.utils import to_categorical\nfrom tensorflow.keras.applications import EfficientNetB3\nfrom tensorflow.keras.layers.experimental.preprocessing import RandomCrop,CenterCrop, RandomRotation","metadata":{"execution":{"iopub.status.busy":"2022-12-26T16:13:52.505054Z","iopub.execute_input":"2022-12-26T16:13:52.505553Z","iopub.status.idle":"2022-12-26T16:13:52.51904Z","shell.execute_reply.started":"2022-12-26T16:13:52.505505Z","shell.execute_reply":"2022-12-26T16:13:52.517654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Dataset\nUsing 256 processed dicom images to png","metadata":{}},{"cell_type":"code","source":"np.random.seed(10)\n\ntrain_df = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\n\ntest_df = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')\n\n\nbase_path='/kaggle/input/rsna-mammography-images-as-pngs/images_as_pngs_cv2_256/'\n\n# saving image path into train dataframe\ntrain_df['img_path']= f'{base_path}/train_images_processed_cv2_256'\\\n                    + '/' + train_df.patient_id.astype(str)\\\n                    + '/' + train_df.image_id.astype(str)\\\n                    + '.png'\n\n\n\ndisplay(train_df.head(3))","metadata":{"execution":{"iopub.status.busy":"2022-12-26T16:13:52.525236Z","iopub.execute_input":"2022-12-26T16:13:52.525711Z","iopub.status.idle":"2022-12-26T16:13:52.734539Z","shell.execute_reply.started":"2022-12-26T16:13:52.52567Z","shell.execute_reply":"2022-12-26T16:13:52.733492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get the counts for each class\ncases_count = train_df['cancer'].value_counts()\nprint(cases_count)\n\n# Plot the results \nplt.figure(figsize=(10,8))\nsns.barplot(x=cases_count.index, y= cases_count.values)\nplt.title('Number of cases', fontsize=14)\nplt.xlabel('Case type', fontsize=12)\nplt.ylabel('Count', fontsize=12)\nplt.xticks(range(len(cases_count.index)), ['Normal(0)', 'Cancer(1)'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-26T16:13:52.736871Z","iopub.execute_input":"2022-12-26T16:13:52.73723Z","iopub.status.idle":"2022-12-26T16:13:52.939807Z","shell.execute_reply.started":"2022-12-26T16:13:52.737199Z","shell.execute_reply":"2022-12-26T16:13:52.938505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As you can see the data is highly imbalance where only 1158 of cancer cases and 53548 normal sample images.\n\n","metadata":{}},{"cell_type":"markdown","source":"# show some samples of cancer and normal cases\n# ","metadata":{}},{"cell_type":"code","source":"Cancer_samples = (train_df[train_df['cancer']==1]['img_path'].iloc[0:5]).tolist()\nNormal_samples = (train_df[train_df['cancer']==0]['img_path'].iloc[0:5]).tolist()\n# Concat the data in a single list and del the above two list\nsamples = Cancer_samples + Normal_samples\n# source = \"../input/melanoma-merged-external-data-512x512-jpeg/512x512-dataset-melanoma/512x512-dataset-melanoma/\"\n# Plot the data \nf, ax = plt.subplots(2,5, figsize=(30,10))\nfor i in range(10):\n    img = tf.keras.preprocessing.image.load_img(samples[i])\n    ax[i//5, i%5].imshow(img, cmap='gray')\n    if i<5:\n        ax[i//5, i%5].set_title(\"Cancer\")\n    else:\n        ax[i//5, i%5].set_title(\"Normal\")\n    ax[i//5, i%5].axis('off')\n    ax[i//5, i%5].set_aspect('auto')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-26T16:15:30.164401Z","iopub.execute_input":"2022-12-26T16:15:30.164874Z","iopub.status.idle":"2022-12-26T16:15:31.304875Z","shell.execute_reply.started":"2022-12-26T16:15:30.164839Z","shell.execute_reply":"2022-12-26T16:15:31.303487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Augmentation Functions\n# ","metadata":{}},{"cell_type":"code","source":"data_augmentation_layers = tf.keras.Sequential(\n    [\n        layers.experimental.preprocessing.RandomCrop(height=256, width=256),\n        layers.experimental.preprocessing.RandomFlip(\"horizontal_and_vertical\"),\n        layers.experimental.preprocessing.RandomRotation(0.25),\n        layers.experimental.preprocessing.RandomZoom((-0.2, 0)),\n        layers.experimental.preprocessing.RandomContrast((0.2,0.2)),\n])\n","metadata":{"execution":{"iopub.status.busy":"2022-12-26T16:16:04.513812Z","iopub.execute_input":"2022-12-26T16:16:04.514233Z","iopub.status.idle":"2022-12-26T16:16:04.573031Z","shell.execute_reply.started":"2022-12-26T16:16:04.5142Z","shell.execute_reply":"2022-12-26T16:16:04.57177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Show Augmentations\n","metadata":{}},{"cell_type":"code","source":"image = tf.keras.preprocessing.image.load_img(train_df['img_path'][90])\n\n\nimage = tf.expand_dims(np.array(image), 0)\n\nprint(image.shape)\n\nplt.figure(figsize=(10, 10))\nfor i in range(6):\n  augmented_image = data_augmentation_layers(image)\n  print('augmented_image ',augmented_image.shape)\n  ax = plt.subplot(3, 3, i + 1)\n  plt.imshow(augmented_image[0])\n  plt.axis(\"off\")","metadata":{"execution":{"iopub.status.busy":"2022-12-26T16:16:06.896761Z","iopub.execute_input":"2022-12-26T16:16:06.897197Z","iopub.status.idle":"2022-12-26T16:16:07.903823Z","shell.execute_reply.started":"2022-12-26T16:16:06.897161Z","shell.execute_reply":"2022-12-26T16:16:07.902562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# DataGenerator Class to generate batches of images and labels","metadata":{}},{"cell_type":"code","source":"from imgaug import augmenters as iaa\n\n\n\nclass DataGenerator(tf.keras.utils.Sequence):\n    def __init__(self, df, path, batch_size=32, shuffle=True,aug=True,labels=True):\n        self.df = df.copy()\n        if 'prediction_id' not in df:\n            self.df['prediction_id'] = df[\"patient_id\"].astype(str) + '_' + df[\"laterality\"].astype(str)\n\n        self.prediction_ids = self.df['prediction_id'].unique()\n        self.labels = labels\n        if self.labels ==True:\n            self.labels = self.df.groupby('prediction_id')['cancer'].max()\n        self.path = path\n        self.batch_size = batch_size\n        self.aug=aug\n        self.shuffle = shuffle\n        self.on_epoch_end()\n\n    def __len__(self):\n        \"\"\"Denotes the number of batches per epoch\"\"\"\n        return int(len(self.prediction_ids) / self.batch_size)\n\n    def __getitem__(self, index):\n        \"\"\"Generate one batch of data\"\"\"\n        batch_indexes = self.prediction_ids[index * self.batch_size:(index + 1) * self.batch_size]\n        X, y = self.__data_generation(batch_indexes)\n        return X, y\n\n    def __get_input(self, path):\n        \n#         print('path   ',path)\n        image = tf.keras.preprocessing.image.load_img(path)\n        image_arr = tf.keras.preprocessing.image.img_to_array(image)\n        \n        if self.aug:\n            \n             image_arr=self.augmentor(image_arr)\n\n        \n        return image_arr\n\n    \n    def augmentor(self, images):\n        'Apply data augmentation'\n        images=data_augmentation_layers(images)\n\n        return images\n    \n    \n    def on_epoch_end(self):\n        \"\"\"Updates indexes after each epoch\"\"\"\n        if self.shuffle:\n            self.df = self.df.sample(frac=1).reset_index(drop=True)\n\n    def __data_generation(self, batch_indexes):\n        paths = self.get_paths_images(batch_indexes)\n        X = np.asarray([self.__get_input(path) for path in paths])\n        y = np.array([self.labels[batch_indexes]])\n        return X, y\n\n    def get_paths_images(self, batch_indexes):\n        batch = self.df[self.df['prediction_id'].isin(batch_indexes)]\n        rows_batch = self.get_rows(batch)\n        return self.path + rows_batch[\"patient_id\"].astype(str) + \"/\" + rows_batch[\"image_id\"].astype(\n            str) + \".png\"\n\n    def get_rows(self, batch):\n        \"\"\"Select only 1 MLO view picture per breast\"\"\"\n        only_MLO_view_images = batch[batch['view'] == 'MLO']\n        only_one_per_prediction_id = only_MLO_view_images.groupby('prediction_id')[['patient_id', 'image_id']].max()\n        return only_one_per_prediction_id","metadata":{"execution":{"iopub.status.busy":"2022-12-26T16:16:13.249244Z","iopub.execute_input":"2022-12-26T16:16:13.249714Z","iopub.status.idle":"2022-12-26T16:16:13.270002Z","shell.execute_reply.started":"2022-12-26T16:16:13.249677Z","shell.execute_reply":"2022-12-26T16:16:13.26877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Define Metrics\n","metadata":{}},{"cell_type":"code","source":"def weighted_binary_loss(weight, from_logits=True, reduction=\"mean\"):\n    def inverse_sigmoid(sigmoidal):\n        return - tf.math.log(1. / sigmoidal - 1.)\n\n    def weighted_loss(labels, predictions):\n        predictions = tf.convert_to_tensor(predictions)\n        labels = tf.cast(labels, predictions.dtype)\n        num_samples = tf.cast(tf.shape(labels)[-1], dtype=labels.dtype)\n\n        logits = tf.cond(\n            tf.cast(from_logits, dtype=tf.bool),\n            lambda: predictions,\n            lambda: inverse_sigmoid(sigmoidal=predictions),\n        )\n        loss = tf.nn.weighted_cross_entropy_with_logits(\n            tf.cast(labels, dtype=tf.float32), logits, pos_weight=weight\n        )\n        \n        if reduction.lower() == \"mean\":\n            return tf.reduce_mean(loss)\n        elif reduction.lower() == \"sum\":\n            return tf.reduce_sum(loss) / num_samples\n        elif reduction.lower() == \"none\":\n            return loss\n        else:\n            raise ValueError(\n                'Reduction type is should be `mean` or `sum` or `none`. ',\n                f'But, received {reduction}'\n            )\n    return weighted_loss\n\ndef binary_focal_loss(\n    alpha=0.25, \n    gamma=2.0, \n    label_smoothing=0, \n    from_logits=False,\n    apply_class_balancing=False,\n    apply_positive_weight=1,\n    reduction=\"mean\"\n):\n    '''\n    alpha: A weight balancing factor for class 1, default is 0.25. \n        The weight for class 0 is 1.0 - alpha.\n    \n    gamma: A focusing parameter used to compute the focal factor, default is 2.0\n    \n    apply_class_balancing: A bool, whether to apply weight balancing on the binary \n        classes 0 and 1.\n    '''\n    \n    def smooth_labels(labels):\n        return labels * (1.0 - label_smoothing) + 0.5 * label_smoothing\n    \n    def compute_loss(labels, logits):\n        logits = tf.convert_to_tensor(logits)\n        labels = tf.cast(labels, logits.dtype)\n        labels = tf.cond(\n            tf.cast(label_smoothing, dtype=tf.bool),\n            lambda: smooth_labels(labels),\n            lambda: labels,\n        )\n        num_samples = tf.cast(tf.shape(labels)[-1], dtype=labels.dtype)\n        cross_entropy = weighted_binary_loss(\n            apply_positive_weight, from_logits, reduction='none'\n        )(labels, logits)\n        \n        sigmoidal = tf.cond(\n            tf.cast(from_logits, dtype=tf.bool),\n            lambda: tf.nn.sigmoid(logits),\n            lambda: logits,\n        )\n        pt = labels * sigmoidal + (1.0 - labels) * (1.0 - sigmoidal)\n        focal_factor = tf.pow(1.0 - pt, gamma)\n        focal_bce =  focal_factor * cross_entropy\n        \n        if apply_class_balancing:\n            weight = labels * alpha + (1 - labels) * (1 - alpha)\n            focal_bce = weight * focal_bce\n\n        if reduction == 'mean':\n            return tf.reduce_mean(focal_bce)\n        elif reduction == 'sum':\n            return tf.reduce_sum(focal_bce) / num_samples\n        else:\n            raise ValueError(\n                'Reduction type should be `mean` or `sum` ',\n                f'But, received {reduction}'\n            )\n    return compute_loss\n\ndef tf_pfbeta(from_logits=True, beta=1.0, epsilon=1e-07):\n    \n    def pfbeta(y_true, y_pred):\n        y_pred = tf.cond(\n            tf.cast(from_logits, dtype=tf.bool),\n            lambda: tf.nn.sigmoid(y_pred),\n            lambda: y_pred,\n        )\n        y_true = tf.reshape(y_true, [-1])\n        y_pred = tf.reshape(y_pred, [-1])\n\n        ctp = tf.reduce_sum(y_true * y_pred, axis=-1)\n        cfp = tf.reduce_sum(y_pred, axis=-1) - ctp\n\n        c_precision = ctp / (ctp + cfp)\n        c_recall = ctp / tf.reduce_sum(y_true)\n        \n        def compute_fractions():\n            numerator = c_precision * c_recall\n            denominator = beta**2 * c_precision + c_recall\n            return (1 + beta**2) * tf.math.divide_no_nan(numerator, denominator)\n        \n        return tf.cond(\n            tf.logical_and(\n                tf.greater(c_precision, 0.), tf.greater(c_recall, 0.)\n            ),\n            compute_fractions,\n            lambda: tf.constant(0, dtype=tf.float32)\n        )\n    \n    return pfbeta\n\ndef tf_auc(from_logits=True):\n    auc_fn = metrics.AUC()\n    \n    def auc(y_true, y_pred):\n        y_pred = tf.cond(\n            tf.cast(from_logits, dtype=tf.bool),\n            lambda: tf.nn.sigmoid(y_pred),\n            lambda: y_pred,\n        )\n        return auc_fn(y_true, y_pred)\n    \n    return auc","metadata":{"execution":{"iopub.status.busy":"2022-12-26T16:16:15.88855Z","iopub.execute_input":"2022-12-26T16:16:15.888944Z","iopub.status.idle":"2022-12-26T16:16:15.914882Z","shell.execute_reply.started":"2022-12-26T16:16:15.888912Z","shell.execute_reply":"2022-12-26T16:16:15.913368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_lr_callback(batch_size=8,scheduler='cosine'):\n    lr_start   = 0.000005\n    lr_max     = 0.00000105  * batch_size\n    lr_min     = 0.000001\n    lr_ramp_ep = 4\n    lr_sus_ep  = 0\n    lr_decay   = 0.8\n   \n    def lrfn(epoch):\n        if epoch < lr_ramp_ep:\n            lr = (lr_max - lr_start) / lr_ramp_ep * epoch + lr_start\n            \n        elif epoch < lr_ramp_ep + lr_sus_ep:\n            lr = lr_max\n            \n        elif scheduler=='exp':\n            lr = (lr_max - lr_min) * lr_decay**(epoch - lr_ramp_ep - lr_sus_ep) + lr_min\n            \n        elif scheduler=='cosine':\n            decay_total_epochs = CFG.epochs - lr_ramp_ep - lr_sus_ep + 3\n            decay_epoch_index = epoch - lr_ramp_ep - lr_sus_ep\n            phase = math.pi * decay_epoch_index / decay_total_epochs\n            cosine_decay = 0.4 * (1 + math.cos(phase))\n            lr = (lr_max - lr_min) * cosine_decay + lr_min\n        return lr\n  \n    lr_callback = tf.keras.callbacks.LearningRateScheduler(lrfn, verbose=False)\n    return lr_callback\n\n","metadata":{"execution":{"iopub.status.busy":"2022-12-26T16:19:56.354303Z","iopub.execute_input":"2022-12-26T16:19:56.354786Z","iopub.status.idle":"2022-12-26T16:19:56.365346Z","shell.execute_reply.started":"2022-12-26T16:19:56.354751Z","shell.execute_reply":"2022-12-26T16:19:56.36424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Baseline Model\n","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\n\n\ndef create_model(inp_dim=(256, 256, 3)):\n    inputs = tf.keras.layers.Input(inp_dim)\n    x = tf.keras.layers.Conv2D(32, (3, 3), activation='relu', strides=4)(inputs)\n    x = tf.keras.layers.Flatten()(x)\n    outputs = tf.keras.layers.Dense(1, activation='sigmoid')(x)\n    model = tf.keras.models.Model(inputs=inputs, outputs=outputs)\n    \n    opti =  tf.keras.optimizers.Adam(lr=0.003,clipvalue=0.7)\n\n    \n    model.compile(\n        optimizer = tfa.optimizers.RectifiedAdam(\n            learning_rate=0.003, amsgrad=False\n        ),\n        loss = binary_focal_loss(\n            apply_class_balancing=True, \n            apply_positive_weight=5,  \n            alpha=0.65, \n            gamma=5.0,\n            label_smoothing=0.01,\n            from_logits=True,\n            reduction='mean'\n        ), \n        metrics = [\n            tf_pfbeta(beta=1.0, from_logits=True),\n            tf_auc(from_logits=True),\n        ]\n    )\n\n    \n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2022-12-26T16:16:19.412247Z","iopub.execute_input":"2022-12-26T16:16:19.41269Z","iopub.status.idle":"2022-12-26T16:16:19.422743Z","shell.execute_reply.started":"2022-12-26T16:16:19.412651Z","shell.execute_reply":"2022-12-26T16:16:19.421541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Function to generate train and validation batches\n# ","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nnp.random.seed(0)\n\nbatch_size=8\nepochs=2\nimg_path='/kaggle/input/rsna-mammography-images-as-pngs/images_as_pngs_cv2_256/train_images_processed_cv2_256/'\n\ndef get_train_val_generator(train_size=0.8, batch_size=batch_size, filename=train_df, image_dir=img_path):\n\n    patient_ids = train_df[\"patient_id\"].unique()\n    np.random.shuffle(patient_ids)\n    train_size = int(len(patient_ids) * train_size)\n    train_ids = patient_ids[:train_size]\n    val_ids = patient_ids[train_size:]\n\n    df_train = train_df[train_df['patient_id'].isin(train_ids)]\n    df_val = train_df[train_df['patient_id'].isin(val_ids)]\n\n    train_gen = DataGenerator(df_train, batch_size=batch_size, path=image_dir,aug=True)\n    val_gen = DataGenerator(df_val, batch_size=batch_size, path=image_dir,aug=False)\n    \n    return train_gen, val_gen","metadata":{"execution":{"iopub.status.busy":"2022-12-26T16:25:33.253309Z","iopub.execute_input":"2022-12-26T16:25:33.254816Z","iopub.status.idle":"2022-12-26T16:25:33.266657Z","shell.execute_reply.started":"2022-12-26T16:25:33.254762Z","shell.execute_reply":"2022-12-26T16:25:33.265444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras import metrics\n\nmodel = create_model()\n\nimage_dir = f'{base_path}/train_images_processed_cv2_256/'\ndataset_path = train_df\ntrain_gen, val_gen = get_train_val_generator(filename=dataset_path, image_dir=image_dir)\n\n\n \nsave_path='../'\nval_pfbeta_weights=tf.keras.callbacks.ModelCheckpoint(save_path, monitor='val_pfbeta', mode='max', verbose=2, save_best_only=True, save_weights_only=True, period=1)\n\ncallbacks=[val_pfbeta_weights,get_lr_callback(batch_size=batch_size)]\n\ntrain_gen, val_gen = get_train_val_generator(filename=dataset_path, image_dir=image_dir)      \nhistory = model.fit(train_gen, validation_data=val_gen, epochs=epochs,callbacks=callbacks)\n","metadata":{"execution":{"iopub.status.busy":"2022-12-26T16:25:35.184102Z","iopub.execute_input":"2022-12-26T16:25:35.184551Z","iopub.status.idle":"2022-12-26T16:49:00.866069Z","shell.execute_reply.started":"2022-12-26T16:25:35.184513Z","shell.execute_reply":"2022-12-26T16:49:00.864694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Test Images\n\nDATASET_PATH='/kaggle/input/rsna-breast-cancer-detection/'\ntest_df = pd.read_csv(os.path.join(DATASET_PATH, \"test.csv\"))\ndisplay(test_df.head())\nprint(f'cases: {len(test_df)}')\n\n# Show sample submission example\n\ndf_sub = pd.read_csv(os.path.join(DATASET_PATH, \"sample_submission.csv\"))\ndisplay(df_sub.head())\nprint(f'cases: {len(df_sub)}')","metadata":{"execution":{"iopub.status.busy":"2022-12-26T17:05:16.363974Z","iopub.execute_input":"2022-12-26T17:05:16.364496Z","iopub.status.idle":"2022-12-26T17:05:16.39837Z","shell.execute_reply.started":"2022-12-26T17:05:16.36444Z","shell.execute_reply":"2022-12-26T17:05:16.397159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_image(image_path):\n    img = tf.io.read_file(image_path)\n    img = tf.image.decode_jpeg(img, channels = 3)\n    img = tf.image.resize(img, [256, 256])\n    img = tf.cast(img, dtype = tf.float32)\n    img = img/255.0\n    return img\n\n# test_paths=[]\n# test_dir='/kaggle/input/rsnatest/test_images_256/10008/'\n# img_path= os.listdir (test_dir)\n\n\n\nDF_PATH = '/kaggle/input/rsna-breast-cancer-detection'\ndf = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/test.csv\")\n\ntest_path='/kaggle/input/rsnatest/test_images_256/'\n\ntest_dir = f'{test_path}'\n\n\n\n\ntest_df['img_path']= f'{test_dir}/'\\\n                    + '/' + test_df.patient_id.astype(str)\\\n                    + '/' + test_df.image_id.astype(str)\\\n                    + '.png'\n\n\ntest_df\n# image = tf.keras.preprocessing.image.load_img(test_df['img_path'][1])\n        \n# plt.imshow(image)\n# plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-26T17:05:19.300033Z","iopub.execute_input":"2022-12-26T17:05:19.300499Z","iopub.status.idle":"2022-12-26T17:05:19.324165Z","shell.execute_reply.started":"2022-12-26T17:05:19.300443Z","shell.execute_reply":"2022-12-26T17:05:19.323209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds=[]\nfor i in range (len(test_df['img_path'])):\n    \n    image = tf.keras.preprocessing.image.load_img(test_df['img_path'][i])\n\n    image=tf.expand_dims(np.array(image), 0)\n    pred=model.predict(np.asarray(image))\n    \n    preds.append(pred)","metadata":{"execution":{"iopub.status.busy":"2022-12-26T17:05:22.660095Z","iopub.execute_input":"2022-12-26T17:05:22.660511Z","iopub.status.idle":"2022-12-26T17:05:23.023467Z","shell.execute_reply.started":"2022-12-26T17:05:22.660451Z","shell.execute_reply":"2022-12-26T17:05:23.022179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_df = pd.DataFrame({'prediction_id':test_df.prediction_id,\n                        'cancer':preds})\n\npred_df['cancer']=(pred_df.cancer > 0.5).astype(int)\n\n\n","metadata":{"execution":{"iopub.status.busy":"2022-12-26T17:05:34.506643Z","iopub.execute_input":"2022-12-26T17:05:34.507175Z","iopub.status.idle":"2022-12-26T17:05:34.51561Z","shell.execute_reply.started":"2022-12-26T17:05:34.507135Z","shell.execute_reply":"2022-12-26T17:05:34.514055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_df.to_csv('sample_submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-12-26T17:06:00.631181Z","iopub.execute_input":"2022-12-26T17:06:00.631623Z","iopub.status.idle":"2022-12-26T17:06:00.638637Z","shell.execute_reply.started":"2022-12-26T17:06:00.631583Z","shell.execute_reply":"2022-12-26T17:06:00.637633Z"},"trusted":true},"execution_count":null,"outputs":[]}]}