{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Idea:\n* In this notebook we will be doing basic EDA\n* Than we will be writing custom class for image generator, as we have images in dcm format\n* we will be writing efficient net B0 model in tensorflow from scratch\n* Training the Vanilla viT model on 1000 images\n* making the predictions and submission","metadata":{}},{"cell_type":"markdown","source":"**Please vote and comment if you find this useful :)**","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport pandas_profiling\nimport numpy as np\nimport warnings\nimport os\n\n\n# visualization\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n%matplotlib inline\n\n# install gdcm to open the DICOM file\n!pip install -qU python-gdcm pydicom pylibjpeg\n\nimport pydicom\nimport pylibjpeg\n\nimport cv2\nimport glob\n\nfrom path import Path\nfrom tqdm import tqdm\nimport pydicom as dicom\n\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm, trange\nfrom sklearn.model_selection import train_test_split\n\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom keras_preprocessing.image.dataframe_iterator import DataFrameIterator\n\nDEVICE = 'GPU' # 'GPU', 'TPU'\nwarnings.simplefilter(action=\"ignore\")\nos.environ[\"TF_CPP_MIN_LOG_LEVEL\"] = \"3\"","metadata":{"execution":{"iopub.status.busy":"2023-01-06T18:19:32.168302Z","iopub.execute_input":"2023-01-06T18:19:32.168777Z","iopub.status.idle":"2023-01-06T18:19:59.056974Z","shell.execute_reply.started":"2023-01-06T18:19:32.168683Z","shell.execute_reply":"2023-01-06T18:19:59.055875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Load train csv file and path for training images**","metadata":{}},{"cell_type":"code","source":"train_dir = \"/kaggle/input/rsna-breast-cancer-detection/train_images/\"\ntrain = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-06T18:20:05.362402Z","iopub.execute_input":"2023-01-06T18:20:05.36321Z","iopub.status.idle":"2023-01-06T18:20:05.510364Z","shell.execute_reply.started":"2023-01-06T18:20:05.363168Z","shell.execute_reply":"2023-01-06T18:20:05.509045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Perform basic EDA on provided meta data in csv","metadata":{}},{"cell_type":"code","source":"# 'cancer' and 'age'\nfig, ax1 = plt.subplots()\n\ncolor = 'tab:blue'\nax1.set_xlabel('age')\nax1.set_ylabel('cancer=0 count', color=color)\nax1.hist(train.loc[train['cancer'] == 0, 'age'].dropna(), bins=30, alpha=0.5, label='0', color=color)\nax1.tick_params(axis='y', labelcolor=color)\n\nax2 = ax1.twinx()  # instantiate a second axes that shares the same x-axis\n\ncolor = 'tab:green'\nax2.set_ylabel('cancer=1 count', color=color)  # we already handled the x-label with ax1\nax2.hist(train.loc[train['cancer'] == 1, 'age'].dropna(), bins=30, alpha=0.8, label='1', color=color)\nax2.tick_params(axis='y', labelcolor=color)\n\nfig.tight_layout()  # otherwise the right y-label is slightly clipped\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-06T18:20:14.221422Z","iopub.execute_input":"2023-01-06T18:20:14.22184Z","iopub.status.idle":"2023-01-06T18:20:14.768586Z","shell.execute_reply.started":"2023-01-06T18:20:14.221803Z","shell.execute_reply":"2023-01-06T18:20:14.767553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Statistical data for 'cancer' and 'age'\ntrain.loc[train['cancer'] == 1, 'age'].describe()","metadata":{"execution":{"iopub.status.busy":"2023-01-06T18:20:21.586172Z","iopub.execute_input":"2023-01-06T18:20:21.586571Z","iopub.status.idle":"2023-01-06T18:20:21.601319Z","shell.execute_reply.started":"2023-01-06T18:20:21.58654Z","shell.execute_reply":"2023-01-06T18:20:21.59981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 'cancer' and 'laterality'\n# laterality - Whether the image is of the left or right breast.\n\nsns.countplot(x='laterality', hue='cancer', data=train)\nplt.legend(loc='upper right', title='cancer')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-06T18:20:24.683634Z","iopub.execute_input":"2023-01-06T18:20:24.684104Z","iopub.status.idle":"2023-01-06T18:20:24.943671Z","shell.execute_reply.started":"2023-01-06T18:20:24.684056Z","shell.execute_reply":"2023-01-06T18:20:24.942316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 'cancer' and 'view'\n# view - The orientation of the image.\n\nsns.countplot(x='view', hue='cancer', data=train)\nplt.legend(loc='upper right', title='cancer')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-06T18:20:28.38632Z","iopub.execute_input":"2023-01-06T18:20:28.386855Z","iopub.status.idle":"2023-01-06T18:20:28.664157Z","shell.execute_reply.started":"2023-01-06T18:20:28.386808Z","shell.execute_reply":"2023-01-06T18:20:28.663016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 'cancer' and 'density' *ONLY available in TRAIN data\n# density - A rating for how dense the breast tissue is, with A being the least dense and D being the most dense. \n# Extremely dense tissue can make diagnosis more difficult. Only provided for train.\n\nsns.countplot(x='density', hue='cancer', data=train)\nplt.legend(loc='upper right', title='cancer')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-06T18:20:32.549521Z","iopub.execute_input":"2023-01-06T18:20:32.549969Z","iopub.status.idle":"2023-01-06T18:20:33.007166Z","shell.execute_reply.started":"2023-01-06T18:20:32.549914Z","shell.execute_reply":"2023-01-06T18:20:33.006241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 'cancer' and 'BIRADS' *ONLY available in TRAIN data\n# BIRADS - 0 if the breast required follow-up, 1 if the breast was rated as negative for cancer, and \n# 2 if the breast was rated as normal. Only provided for train.\n\nsns.countplot(x='BIRADS', hue='cancer', data=train)\nplt.legend(loc='upper right', title='cancer')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-06T18:20:36.37507Z","iopub.execute_input":"2023-01-06T18:20:36.375493Z","iopub.status.idle":"2023-01-06T18:20:36.604915Z","shell.execute_reply.started":"2023-01-06T18:20:36.375459Z","shell.execute_reply":"2023-01-06T18:20:36.603658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head(1)","metadata":{"execution":{"iopub.status.busy":"2023-01-06T18:20:51.449537Z","iopub.execute_input":"2023-01-06T18:20:51.450013Z","iopub.status.idle":"2023-01-06T18:20:51.466963Z","shell.execute_reply.started":"2023-01-06T18:20:51.449967Z","shell.execute_reply":"2023-01-06T18:20:51.465669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate the correlation matrix\ncorr = train[['cancer', 'age', 'view', 'laterality', 'biopsy', 'BIRADS', 'implant', 'invasive', 'density', 'difficult_negative_case']].corr()\n\n# Heatmap of the correlation matrix\nsns.heatmap(corr)","metadata":{"execution":{"iopub.status.busy":"2023-01-06T18:23:34.511009Z","iopub.execute_input":"2023-01-06T18:23:34.511856Z","iopub.status.idle":"2023-01-06T18:23:34.872614Z","shell.execute_reply.started":"2023-01-06T18:23:34.511811Z","shell.execute_reply":"2023-01-06T18:23:34.871455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prepare dataframe for datapipeline ","metadata":{}},{"cell_type":"code","source":"# prepare a training data frame for image data generator\ntrain_df = train[['patient_id','image_id','cancer']].copy()","metadata":{"execution":{"iopub.status.busy":"2022-12-26T12:03:31.881081Z","iopub.execute_input":"2022-12-26T12:03:31.882096Z","iopub.status.idle":"2022-12-26T12:03:31.889166Z","shell.execute_reply.started":"2022-12-26T12:03:31.882041Z","shell.execute_reply":"2022-12-26T12:03:31.887652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# add column for image path\ntrain_df['image_path'] = train_df['patient_id'].astype(str) + '/' + train_df['image_id'].astype(str) + '.dcm'","metadata":{"execution":{"iopub.status.busy":"2022-12-26T12:03:31.891175Z","iopub.execute_input":"2022-12-26T12:03:31.891661Z","iopub.status.idle":"2022-12-26T12:03:31.9927Z","shell.execute_reply.started":"2022-12-26T12:03:31.891619Z","shell.execute_reply":"2022-12-26T12:03:31.991446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.loc[train_df[\"cancer\"] == 0, \"cancer\"] = \"negative\"\ntrain_df.loc[train_df[\"cancer\"] == 1, \"cancer\"] = \"positive\"","metadata":{"execution":{"iopub.status.busy":"2022-12-26T12:03:31.994379Z","iopub.execute_input":"2022-12-26T12:03:31.994753Z","iopub.status.idle":"2022-12-26T12:03:32.014158Z","shell.execute_reply.started":"2022-12-26T12:03:31.994719Z","shell.execute_reply":"2022-12-26T12:03:32.012827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# selecting only 1000 images\ntraining_df = train_df.head(1000)","metadata":{"execution":{"iopub.status.busy":"2022-12-26T12:03:32.016153Z","iopub.execute_input":"2022-12-26T12:03:32.016541Z","iopub.status.idle":"2022-12-26T12:03:32.022752Z","shell.execute_reply.started":"2022-12-26T12:03:32.016508Z","shell.execute_reply":"2022-12-26T12:03:32.020898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Display and have look on a sample image from training data**","metadata":{}},{"cell_type":"code","source":"import pydicom as dicom\nds = dicom.dcmread(train_dir+train_df['image_path'][0])\nplt.imshow(ds.pixel_array, cmap='gray')\n# MetaData\nprint(ds)","metadata":{"execution":{"iopub.status.busy":"2022-12-26T12:03:32.025476Z","iopub.execute_input":"2022-12-26T12:03:32.026095Z","iopub.status.idle":"2022-12-26T12:03:35.573639Z","shell.execute_reply.started":"2022-12-26T12:03:32.026023Z","shell.execute_reply":"2022-12-26T12:03:35.572731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Write custom class inheriting from DataFrameIterator as we have dcm format in training images","metadata":{}},{"cell_type":"code","source":"class DCMDataFrameIterator(DataFrameIterator):\n    def __init__(self, *arg, **kwargs):\n        self.white_list_formats = ('dcm')\n        super(DCMDataFrameIterator, self).__init__(*arg, **kwargs)\n        self.dataframe = kwargs['dataframe']\n        self.x = self.dataframe[kwargs['x_col']]\n        self.y = self.dataframe[kwargs['y_col']]\n        self.color_mode = kwargs['color_mode']\n        self.target_size = kwargs['target_size']\n        self.directory = kwargs['directory']\n\n    def _get_batches_of_transformed_samples(self, indices_array):\n        # get batch of images\n        batch_x = np.array([self.read_dcm_as_array(self.directory + dcm_path, self.target_size, color_mode=self.color_mode)\n                            for dcm_path in self.x.iloc[indices_array]])\n\n        for i in self.y.iloc[indices_array]:\n            if i == \"negative\":\n                batch_y = np.array([np.array(0)])\n            else:\n                batch_y = np.array([np.array(1)])\n        batch_y = batch_y.reshape((-1,1))\n        # transform images\n        if self.image_data_generator is not None:\n            for i, (x, y) in enumerate(zip(batch_x, batch_y)):\n                transform_params = self.image_data_generator.get_random_transform(x.shape)\n                batch_x[i] = self.image_data_generator.apply_transform(x, transform_params)\n                # you can change y here as well, eg: in semantic segmentation you want to transform masks as well \n                # using the same image_data_generator transformations.\n\n        return batch_x, batch_y\n\n    @staticmethod\n    def read_dcm_as_array(dcm_path, target_size=(256, 256), color_mode='rgb'):\n        image_array = pydicom.dcmread(dcm_path).pixel_array\n        image_array = cv2.resize(image_array, target_size, interpolation=cv2.INTER_NEAREST)  #this returns a 2d array\n        image_array = np.expand_dims(image_array, -1)\n        if color_mode == 'rgb':\n            image_array = cv2.cvtColor(image_array, cv2.COLOR_GRAY2RGB)\n        return image_array\n\n      \n\n# augmentation parameters\n# you can use preprocessing_function instead of rescale in all generators\n# if you are using a pretrained network\ntrain_augmentation_parameters = dict(\n    rescale=1.0/255.0,\n    rotation_range=10,\n    zoom_range=0.2,\n    horizontal_flip=True,\n    fill_mode='nearest',\n    brightness_range = [0.8, 1.2],\n    validation_split = 0.2\n)\n\nvalid_augmentation_parameters = dict(\n    rescale=1.0/255.0,\n    validation_split = 0.2\n)\n\ntest_augmentation_parameters = dict(\n    rescale=1.0/255.0\n)\n\n# training parameters\nBATCH_SIZE = 32\nCLASS_MODE = 'binary'\nCOLOR_MODE = 'grayscale'\nTARGET_SIZE = (48, 48)\nEPOCHS = 10\nSEED = 1337\n\ntrain_consts = {\n    'seed': SEED,\n    'batch_size': BATCH_SIZE,\n    'class_mode': CLASS_MODE,\n    'color_mode': COLOR_MODE,\n    'target_size': TARGET_SIZE,  \n    'subset': 'training'\n}\n\nvalid_consts = {\n    'seed': SEED,\n    'batch_size': BATCH_SIZE,\n    'class_mode': CLASS_MODE,\n    'color_mode': COLOR_MODE,\n    'target_size': TARGET_SIZE, \n    'subset': 'validation'\n}\n\ntest_consts = {\n    'batch_size': 1,  # should be 1 in testing\n    'class_mode': CLASS_MODE,\n    'color_mode': COLOR_MODE,\n    'target_size': TARGET_SIZE,  # resize input images\n    'shuffle': False\n}\n\n# Using the training phase generators \ntrain_augmenter = ImageDataGenerator(**train_augmentation_parameters)\nvalid_augmenter = ImageDataGenerator(**valid_augmentation_parameters)\n\n\ntrain_generator = DCMDataFrameIterator(dataframe=training_df,\n                             x_col='image_path',\n                             y_col='cancer',\n                             directory=train_dir,\n                             image_data_generator=train_augmenter,\n                             **train_consts)\n\nvalid_generator = DCMDataFrameIterator(dataframe=training_df,\n                             x_col='image_path',\n                             y_col='cancer',\n                             directory=train_dir,\n                             image_data_generator=valid_augmenter,\n                             **valid_consts)","metadata":{"execution":{"iopub.status.busy":"2022-12-26T12:03:35.577612Z","iopub.execute_input":"2022-12-26T12:03:35.578667Z","iopub.status.idle":"2022-12-26T12:03:36.955713Z","shell.execute_reply.started":"2022-12-26T12:03:35.578623Z","shell.execute_reply":"2022-12-26T12:03:36.954612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Vision transformer model from scratch in tensorflow","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\n\n# parameters for model\nbatch_size = 256\nimage_size = 224\n\npatch_size = 16\n\nprojection_dim = 64\ntransformer_units = [projection_dim * 2, projection_dim, ]\ntransformer_layers = 8\nmlp_head_units = [2048, 1024]\n\n\ndef mlp(x, hidden_units, dropout_rate):\n    for units in hidden_units:\n        x = layers.Dense(units, activation=tf.nn.gelu)(x)\n        x = layers.Dropout(dropout_rate)(x)\n    return x\n\n\nclass Patches(layers.Layer):\n    def __init__(self, patch_size):\n        super(Patches, self).__init__()\n        self.patch_size = patch_size\n\n    def call(self, images):\n        batch_size = tf.shape(images)[0]\n        patches = tf.image.extract_patches(\n            images=images,\n            sizes=[1, self.patch_size, self.patch_size, 1],\n            strides=[1, self.patch_size, self.patch_size, 1],\n            rates=[1, 1, 1, 1],\n            padding=\"VALID\",\n        )\n        patch_dims = patches.shape[-1]\n        patches = tf.reshape(patches, [batch_size, -1, patch_dims])\n        return patches\n\n\nclass PatchEncoder(layers.Layer):\n    def __init__(self, num_patches, projection_dim):\n        super(PatchEncoder, self).__init__()\n        self.num_patches = num_patches\n        self.projection = layers.Dense(units=projection_dim)\n        self.position_embedding = layers.Embedding(\n            input_dim=num_patches, output_dim=projection_dim\n        )\n\n    def call(self, patch):\n        positions = tf.range(start=0, limit=self.num_patches, delta=1)\n        encoded = self.projection(patch) + self.position_embedding(positions)\n        return encoded","metadata":{"execution":{"iopub.status.busy":"2022-12-26T12:03:36.957024Z","iopub.execute_input":"2022-12-26T12:03:36.9581Z","iopub.status.idle":"2022-12-26T12:03:36.973367Z","shell.execute_reply.started":"2022-12-26T12:03:36.95801Z","shell.execute_reply":"2022-12-26T12:03:36.971878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def model(input_shape=(224, 224, 3), classes=3):\n    inputs = layers.Input(shape=input_shape)\n    patches = Patches(patch_size)(inputs)\n    num_patches = (input_shape[0] // patch_size) ** 2\n    encoded_patches = PatchEncoder(num_patches, projection_dim)(patches)\n\n    for _ in range(transformer_layers):\n        x1 = layers.LayerNormalization(epsilon=1e-6)(encoded_patches)\n        attention_output = layers.MultiHeadAttention(\n            num_heads=4, key_dim=projection_dim, dropout=0.1\n        )(x1, x1)\n        x2 = layers.Add()([attention_output, encoded_patches])\n        x3 = layers.LayerNormalization(epsilon=1e-6)(x2)\n        x3 = mlp(x3, hidden_units=transformer_units, dropout_rate=0.1)\n        encoded_patches = layers.Add()([x3, x2])\n\n    representation = layers.LayerNormalization(epsilon=1e-6)(encoded_patches)\n    representation = layers.Flatten()(representation)\n    representation = layers.Dropout(0.5)(representation)\n    features = mlp(representation, hidden_units=mlp_head_units, dropout_rate=0.5)\n    logits = layers.Dense(classes, activation=\"sigmoid\")(features)\n    model = keras.Model(inputs=inputs, outputs=logits)\n    return model","metadata":{"execution":{"iopub.status.busy":"2022-12-26T12:03:36.975466Z","iopub.execute_input":"2022-12-26T12:03:36.975943Z","iopub.status.idle":"2022-12-26T12:03:36.993518Z","shell.execute_reply.started":"2022-12-26T12:03:36.975904Z","shell.execute_reply":"2022-12-26T12:03:36.992138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = model((48,48,1),1)","metadata":{"execution":{"iopub.status.busy":"2022-12-26T12:03:36.995292Z","iopub.execute_input":"2022-12-26T12:03:36.995748Z","iopub.status.idle":"2022-12-26T12:03:38.332028Z","shell.execute_reply.started":"2022-12-26T12:03:36.995707Z","shell.execute_reply":"2022-12-26T12:03:38.330774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2022-12-26T12:03:38.333374Z","iopub.execute_input":"2022-12-26T12:03:38.333741Z","iopub.status.idle":"2022-12-26T12:03:38.358415Z","shell.execute_reply.started":"2022-12-26T12:03:38.333708Z","shell.execute_reply":"2022-12-26T12:03:38.357134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"EPOCHS = 10\nmodel.compile('adam','binary_crossentropy',['accuracy'])\nearly_stop_callback = tf.keras.callbacks.EarlyStopping(patience=2,\n                                                       restore_best_weights=True)\ncallbacks = [early_stop_callback]\nmodel.fit(train_generator,\n          epochs=EPOCHS,\n          validation_data=valid_generator,\n          callbacks=callbacks,\n)","metadata":{"execution":{"iopub.status.busy":"2022-12-26T12:03:38.360382Z","iopub.execute_input":"2022-12-26T12:03:38.361501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Test data","metadata":{}},{"cell_type":"code","source":"AUTOTUNE = tf.data.experimental.AUTOTUNE\nfrom joblib import Parallel, delayed\ntest_images = glob.glob(\"/kaggle/input/rsna-breast-cancer-detection/test_images/*/*.dcm\")\n\nimage_dir = '/kaggle/tmp/test_images'\nos.makedirs(image_dir, exist_ok=True)\n\nimage_size = 48\ndcm_dir  = '/kaggle/input/rsna-breast-cancer-detection/test_images'\ntest_df = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')\n\ndef dicom_to_png(dcm_file, image_size=image_size, image_dir=''):\n    patient_id = dcm_file.split('/')[-2]\n    image_id   = dcm_file.split('/')[-1][:-4]\n\n    dicom = pydicom.dcmread(dcm_file)\n    img = dicom.pixel_array\n    img = (img - img.min()) / (img.max() - img.min())\n    if dicom.PhotometricInterpretation == 'MONOCHROME1':\n        img = 1 - img\n\n    img = cv2.resize(img, (image_size, image_size), interpolation=cv2.INTER_LINEAR)\n    img = (img * 255).astype(np.uint8)\n    cv2.imwrite(image_dir +'/'+ f'{patient_id}_{image_id}.png', img)\n\n\ndcm_file = dcm_dir + '/' + test_df.patient_id.astype(str) + '/'  + test_df.image_id.astype(str) + '.dcm'\nParallel(n_jobs=2)(\n        delayed(dicom_to_png)(f, image_size=image_size, image_dir=image_dir)\n        for f in tqdm(dcm_file))\n\ndef load_image(image_path):\n    img = tf.io.read_file(image_path)\n    img = tf.image.decode_jpeg(img, channels = 1)\n    img = tf.image.resize(img, [image_size, image_size])\n    img = tf.cast(img, dtype = tf.float32)\n    img = img/255.0\n    return img\n\ntest_image_paths = []\ntest_dir = os.listdir('/kaggle/tmp/test_images')\nfor i in range(len(test_dir)):\n    img_path = '/kaggle/tmp/test_images' + '/' + test_dir[i]\n    test_image_paths.append(img_path)\ntest_img_path_ds = tf.data.Dataset.from_tensor_slices(test_image_paths)\ntest_ds = test_img_path_ds.map(load_image, num_parallel_calls = AUTOTUNE)\ntest_ds = test_ds.batch(BATCH_SIZE)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Predictions**","metadata":{}},{"cell_type":"code","source":"preds = model.predict(test_ds)\npreds","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/test.csv\")\ndf['cancer'] = 0\n\nTHRESHOLD = 0.02\n\n#preds = np.mean([prediction], 0)\npreds = (preds > THRESHOLD).astype(int)\ndf[\"cancer\"] = preds","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"df['prediction_id'] = df['patient_id'].astype(str) + \"_\" + df['laterality']\n\nsub = df[['prediction_id', 'cancer']].groupby(\"prediction_id\").mean().reset_index()\n\nsub.to_csv('/kaggle/working/submission.csv', index=False)\n\nsub.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}