{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"%%capture\n\n!pip install /kaggle/input/rsnamodules/dicomsdl-0.109.1-cp37-cp37m-manylinux_2_12_x86_64.manylinux2010_x86_64.whl \n\ntry:\n    import pylibjpeg\nexcept:\n   !pip install /kaggle/input/rsna-2022-whl/{pylibjpeg-1.4.0-py3-none-any.whl,python_gdcm-3.0.15-cp37-cp37m-manylinux_2_17_x86_64.manylinux2014_x86_64.whl}\n!pip install -q /kaggle/input/rsna-bcd-whl-ds/python_gdcm-3.0.20-cp37-cp37m-manylinux_2_17_x86_64.manylinux2014_x86_64.whl\n!pip install -q /kaggle/input/rsna-bcd-whl-ds/pylibjpeg-1.4.0-py3-none-any.whl\n!pip install -q /kaggle/input/rsna-bcd-whl-ds/dicomsdl-0.109.1-cp37-cp37m-manylinux_2_12_x86_64.manylinux2010_x86_64.whl\n!cp -r /kaggle/input/efficientnet-keras-dataset/efficientnet_kaggle /tmp/ && pip install -q /tmp/efficientnet_kaggle\n!pip install --no-deps  /kaggle/input/kerasefficientnetv2/keras_efficientnet_v2-1.2.2-py3-none-any.whl","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-17T20:41:22.711554Z","iopub.execute_input":"2023-01-17T20:41:22.711917Z","iopub.status.idle":"2023-01-17T20:43:52.789867Z","shell.execute_reply.started":"2023-01-17T20:41:22.711887Z","shell.execute_reply":"2023-01-17T20:43:52.788396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install --no-deps  /kaggle/input/kerasefficientnetv2/keras_efficientnet_v2-1.2.2-py3-none-any.whl","metadata":{"execution":{"iopub.status.busy":"2023-01-17T20:43:52.792568Z","iopub.execute_input":"2023-01-17T20:43:52.793002Z","iopub.status.idle":"2023-01-17T20:44:14.847877Z","shell.execute_reply.started":"2023-01-17T20:43:52.792958Z","shell.execute_reply":"2023-01-17T20:44:14.846519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '3' \nimport warnings\nwarnings.filterwarnings(\"ignore\", category=DeprecationWarning)\nwarnings.filterwarnings(\"ignore\", category=UserWarning)\nwarnings.filterwarnings(\"ignore\", category=FutureWarning)\n\n\nimport dicomsdl as dicoml\nimport cv2\nimport pydicom\n\nfrom joblib import Parallel, delayed\nimport glob\nimport time\nimport numpy as np\nimport os\nfrom matplotlib import pyplot as plt\n\n# import pydicom\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n\nfrom pathlib import Path\nimport glob\nimport pandas as pd\n# import pylibjpeg\n\nimport numpy as np\nfrom tensorflow import keras\nimport tensorflow as tf\nfrom tensorflow.keras import layers\nimport random\nfrom scipy.ndimage import gaussian_filter\nfrom scipy import ndimage\nimport cv2\nimport matplotlib.pyplot as plt\nfrom skimage.transform import rescale, resize, downscale_local_mean\nfrom tqdm import tqdm\nimport tensorflow.keras.backend as K\n# import efficientnet.tfkeras as efn\nimport sklearn\nimport keras_efficientnet_v2\n\nprint('np:', np.__version__)\nprint('pd:', pd.__version__)\nprint('sklearn:', sklearn.__version__)\nprint('tf:',tf.__version__)\n# print('tfa:', tfa.__version__)\n","metadata":{"execution":{"iopub.status.busy":"2023-01-17T20:44:14.85168Z","iopub.execute_input":"2023-01-17T20:44:14.852043Z","iopub.status.idle":"2023-01-17T20:44:21.887844Z","shell.execute_reply.started":"2023-01-17T20:44:14.852011Z","shell.execute_reply":"2023-01-17T20:44:21.886699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IS_INTERACTIVE = os.environ['KAGGLE_KERNEL_RUN_TYPE'] == 'Interactive'\n\nTARGET_HEIGHT = 256\nTARGET_WIDTH = 256\nN_CHANNELS = 3\nINPUT_SHAPE = (TARGET_HEIGHT, TARGET_WIDTH, N_CHANNELS)\nTARGET_HEIGHT_WIDTH_RATIO = TARGET_HEIGHT / TARGET_WIDTH\nTHRESHOLD_BEST = 0.50\n\nCLAHE = cv2.createCLAHE(clipLimit=2.0, tileGridSize=(32, 32))\n\ndef get_model():\n    # Inputs, note the names are equal to the dictionary keys in the dataset\n    image = tf.keras.layers.Input((256,256,3), name='image', dtype=tf.uint8)\n\n    # Normalize Input\n#     image_norm = normalize(image)\n\n    # CNN Prediction\n    outputs = keras_efficientnet_v2.EfficientNetV2T(\n        input_shape=[TARGET_HEIGHT, TARGET_WIDTH, 3],\n        pretrained=None,\n        num_classes=1,\n        classifier_activation='sigmoid',\n        dropout=0.15,\n    )(image)\n\n    model = tf.keras.models.Model(inputs=image, outputs=outputs)\n\n#     model.load_weights('/kaggle/input/rsna-efficientnetv2-training-tensorflow-tpu-ds/model.h5')\n\n#     model.trainable = False\n\n    model.compile()\n\n    return model\n\n","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-01-17T20:44:21.890739Z","iopub.execute_input":"2023-01-17T20:44:21.891484Z","iopub.status.idle":"2023-01-17T20:44:21.907005Z","shell.execute_reply.started":"2023-01-17T20:44:21.891441Z","shell.execute_reply":"2023-01-17T20:44:21.906079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# source: https://www.kaggle.com/code/allunia/rsna-csf-cervical-spine-fracture-eda/notebook\ndef rescale_img_to_hu(dcm_ds):\n    \"\"\"Rescales the image to Hounsfield unit.\"\"\"\n    data = dcm_ds.pixel_array\n    if dcm_ds.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n    return data * dcm_ds.RescaleSlope + dcm_ds.RescaleIntercept\n\ndef show_images_for_patient(patient_id):\n    patient_dir = os.path.join('../input/rsna-breast-cancer-detection/train_images', str(patient_id))\n    num_images = len(glob.glob(f\"{patient_dir}/*\"))\n    print(f\"Number of images for patient: {num_images}\")\n    fig, axs = plt.subplots(5,3, figsize=(24,15))\n    axs = axs.flatten()\n    for i, img_path in enumerate(list(Path(patient_dir).iterdir())):\n        ds = pydicom.dcmread(img_path)\n        axs[i].imshow(rescale_img_to_hu(ds), cmap=\"bone\")\n        \ndef multi_class_labels(data, labels=[1]):\n    if data==1:\n        return [0,1]\n    return [1,0]\n\ndef _crop(img):\n    bin_img = self._binarize(img, threshold=5)\n    contour = self._extract_contour(bin_img)\n    img = self._erase_background(img, contour)\n    x1, x2 = np.min(contour[:, :, 0]), np.max(contour[:, :, 0])\n    y1, y2 = np.min(contour[:, :, 1]), np.max(contour[:, :, 1])\n    x1, x2 = int(0.99 * x1), int(1.01 * x2)\n    y1, y2 = int(0.99 * y1), int(1.01 * y2)\n    return img[y1:y2, x1:x2]\n\ndef image_resize(image, width = None, height = None, inter = cv2.INTER_LINEAR):\n\n    dim = None\n    (h, w) = image.shape[:2]\n\n    if width is None and height is None:\n        return image\n\n    if width is None:\n        r = height / float(h)\n        dim = (int(w * r), height)\n    else:\n        r = width / float(w)\n        dim = (width, int(h * r))\n    resized = cv2.resize(image, dim, interpolation = inter)\n\n    return resized\n\n# Copied from: https://www.kaggle.com/code/paulbacher/custom-preprocessor-rsna-breast-cancer\nclass MammographyPreprocessor():\n    def __init__(self,breast_side):\n        self.breast_side = breast_side\n        assert breast_side in ['L', 'R'], \"breast_side should be 'L' or 'R'\"\n        \n    # Adjust the contrast of an image\n    def _windowing(self, img, scan):\n        function = scan.VOILUTFunction\n        if type(scan.WindowWidth) == list:\n            center = scan.WindowCenter[0]\n            width = scan.WindowWidth[0]\n        else:\n            center = scan.WindowCenter\n            width = scan.WindowWidth\n        y_range = 2**scan.BitsStored - 1\n        if function == 'SIGMOID':\n            img = y_range / (1 + np.exp(-4 * (img - center) / width))\n        else: # LINEAR\n            below = img <= (center - width / 2)\n            above = img > (center + width / 2)\n            between = np.logical_and(~below, ~above)\n            img[below] = 0\n            img[above] = y_range\n            img[between] = ((img[between] - center) / width + 0.5) * y_range\n        return img\n    \n    # Interpret pixels in a consistant way\n    def _fix_photometric_interpretation(self, img, scan):\n        if scan == 'MLO':\n            return img.max() - img\n        elif scan == 'CC':\n            return img - img.min()\n        else:\n            raise ValueError(\"Invalid Photometric Interpretation: {}\"\n                               .format(scan.PhotometricInterpretation))\n    \n    # Cast into 8-bits for saving\n    def _normalize_to_255(self, img):\n        if img.max() != 0:\n            img = img / img.max()\n        img *= 255\n        return img.astype(np.uint8)\n    \n    # Flip the breast horizontally on the chosen side \n    def _flip_breast_side(self, img, scan):\n        img_breast_side = self._determine_breast_side(img)\n        if img_breast_side == self.breast_side:\n            return img\n        else:\n            return np.fliplr(img)    \n    \n    # Determine the current breast side\n    def _determine_breast_side(self, img):\n        col_sums_split = np.array_split(np.sum(img, axis=0), 2)\n        left_col_sum = np.sum(col_sums_split[0])\n        right_col_sum = np.sum(col_sums_split[1])\n        if left_col_sum > right_col_sum:\n            return 'L'\n        else:\n            return 'R'\n    \n    # Crop the useless background of the image\n    def _crop(self, img):\n        bin_img = self._binarize(img, threshold=5)\n        contour = self._extract_contour(bin_img)\n        img = self._erase_background(img, contour)\n        x1, x2 = np.min(contour[:, :, 0]), np.max(contour[:, :, 0])\n        y1, y2 = np.min(contour[:, :, 1]), np.max(contour[:, :, 1])\n        x1, x2 = int(0.99 * x1), int(1.01 * x2)\n        y1, y2 = int(0.99 * y1), int(1.01 * y2)\n        return img[y1:y2, x1:x2]\n    \n    # Binarize the image at the threshold\n    def _binarize(self, img, threshold):\n        return (img > threshold).astype(np.uint8)\n    \n    # Get contour points of the breast\n    def _extract_contour(self, bin_img):\n        contours, _ = cv2.findContours(\n            bin_img, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_NONE)\n        contour = max(contours, key=cv2.contourArea)\n        return contour\n    \n    # Set to background pixels of the image to zero\n    def _erase_background(self, img, contour):\n        mask = np.zeros(img.shape, np.uint8)\n        cv2.drawContours(mask, [contour], -1, 255, cv2.FILLED)\n        output = cv2.bitwise_and(img, mask)\n        return output\n    \n    # Resize the image to the preprocessor size\n    def _resize(self, img):\n        return cv2.resize(img, self.size)\n    \n    # Get the save path of a given dicom file\n    def _get_save_path(self, path, png):\n        patient = path.split('/')[-2]\n        filename = path.split('/')[-1]\n        if png:\n            filename = filename.replace('dcm', 'png')\n        else:\n            filename = filename.replace('dcm', 'jpeg')\n        save_path = os.path.join(os.getcwd(), patient, filename)\n        return save_path\n    \n    # Save the preprocessed image\n    def _save_image(self, img, path, png):\n        save_path = self._get_save_path(path, png)\n        patient_folder = os.path.split(save_path)[0]\n        os.makedirs(patient_folder, exist_ok=True)\n        cv2.imwrite(save_path, img)\n\n        \nmm_preprocessor=MammographyPreprocessor('L')\ntrain_csv = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\n\nimage=tf.io.read_file('/kaggle/input/rsna-mammography-images-as-pngs/images_as_pngs_cv2_256/train_images_processed_cv2_256/30509/642429521.png')\nimage = tf.io.decode_png(image).numpy()\nimage = mm_preprocessor._crop(image)\nimage = mm_preprocessor._flip_breast_side(image, 'R')\nprint(image.shape)\nplt.imshow(resize(image, (256,256)))\n\ntrain_csv[train_csv['patient_id']==30509]","metadata":{"execution":{"iopub.status.busy":"2023-01-17T20:45:13.947541Z","iopub.execute_input":"2023-01-17T20:45:13.947864Z","iopub.status.idle":"2023-01-17T20:45:14.273718Z","shell.execute_reply.started":"2023-01-17T20:45:13.947837Z","shell.execute_reply":"2023-01-17T20:45:14.272827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import clear_output\n\nclass Generator(tf.keras.utils.Sequence):\n    def __init__(self, image_list, view):\n        mm_preprocessor=MammographyPreprocessor('L')\n        np.random.seed(10)\n        train_csv = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\n        train_csv = train_csv[train_csv['view']==view]\n        base_path='/kaggle/input/rsna-mammography-images-as-pngs/images_as_pngs_cv2_256/'\n        # saving image path into train dataframe\n        train_csv['img_path']= f'{base_path}/train_images_processed_cv2_256'\\\n                            + '/' + train_csv.patient_id.astype(str)\\\n                            + '/' + train_csv.image_id.astype(str)\\\n                            + '.png'\n        train_csv = train_csv.sample(frac=1).reset_index(drop=True)\n        case = train_csv[train_csv['cancer']==1]\n        control = train_csv[train_csv['cancer']==0]\n\n        frames = [case, control.loc[:len(case)]]\n        train_csv = pd.concat(frames)\n        \n        train_csv = train_csv.sample(frac=1).reset_index(drop=True)\n        _image=[]\n        _label=[]\n        for encounter in tqdm(train_csv.iterrows()):\n            img_path = os.path.join(f'{base_path}/train_images_processed_cv2_256', str(encounter[1]['patient_id']), str(encounter[1]['image_id'])+ '.png')\n            image=tf.io.read_file(img_path)\n            image = tf.io.decode_png(image).numpy()\n            image = mm_preprocessor._crop(image)\n            image = mm_preprocessor._flip_breast_side(image, str(encounter[1]['laterality']))\n\n            _image.append(resize(image, (256,256)))\n            _label.append(encounter[1]['cancer'])\n        \n        \n        self.image=_image\n        self.label=_label\n        self.n_image=len(self.image)\n    \n    def __len__(self):\n        'Denotes the number of batches per epoch'\n        return 2\n\n    def __getitem__(self):\n        while True:\n            randomindex = random.randint(0, self.n_image-1)\n            X_placeholder = np.zeros((256, 256, 3, 1), dtype=np.float32)\n            y_placeholder = np.zeros((2), dtype=np.int16)\n            image = self.image[randomindex]\n            label = self.label[randomindex]\n                    \n            image=(image-np.min(image))/(np.max(image)-np.min(image))\n            X = np.array(np.stack((np.array(image),)*3, axis=2))\n            y = multi_class_labels(label, np.array([0,1]))\n            X_placeholder[:, :, :, 0] = X\n            y_placeholder = [label]\n            yield X_placeholder, y_placeholder\n    \n    def on_epoch_end(self):\n        print(\"Refreshing Epoch\")\n            \n\n\n","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2023-01-17T20:45:14.275349Z","iopub.execute_input":"2023-01-17T20:45:14.276312Z","iopub.status.idle":"2023-01-17T20:45:14.291026Z","shell.execute_reply.started":"2023-01-17T20:45:14.276273Z","shell.execute_reply":"2023-01-17T20:45:14.290265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom imgaug import augmenters as iaa\n\ntrain_df = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\n\ntest_df = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')\n\n\nbase_path='/kaggle/input/rsna-mammography-images-as-pngs/images_as_pngs_cv2_256/'\n\n# saving image path into train dataframe\ntrain_df['img_path']= f'{base_path}/train_images_processed_cv2_256'\\\n                    + '/' + train_df.patient_id.astype(str)\\\n                    + '/' + train_df.image_id.astype(str)\\\n                    + '.png'\n\nnp.random.seed(0)\n\nbatch_size=32\nepochs=2\nimg_path='/kaggle/input/rsna-mammography-images-as-pngs/images_as_pngs_cv2_256/train_images_processed_cv2_256/'\n\ndef get_train_val_generator(train_size=0.8, batch_size=batch_size, filename=train_df, image_dir=img_path):\n\n    patient_ids = train_df[\"patient_id\"].unique()\n    np.random.shuffle(patient_ids)\n    train_size = int(len(patient_ids) * train_size)\n    train_ids = patient_ids[:train_size]\n    val_ids = patient_ids[train_size:]\n\n    df_train = train_df[train_df['patient_id'].isin(train_ids)]\n    df_val = train_df[train_df['patient_id'].isin(val_ids)]\n\n    train_gen = DataGenerator(df_train, batch_size=batch_size, path=image_dir,aug=True)\n    val_gen = DataGenerator(df_val, batch_size=batch_size, path=image_dir,aug=False)\n    \n    return train_gen, val_gen\n\n\ndata_augmentation_layers = tf.keras.Sequential(\n    [\n        layers.experimental.preprocessing.RandomCrop(height=256, width=256),\n        layers.experimental.preprocessing.RandomFlip(\"horizontal_and_vertical\"),\n        layers.experimental.preprocessing.RandomRotation(0.25),\n        layers.experimental.preprocessing.RandomZoom((-0.2, 0)),\n        layers.experimental.preprocessing.RandomContrast((0.2,0.2)),\n])\n\nclass DataGenerator(tf.keras.utils.Sequence):\n    def __init__(self, df, path, batch_size=32, shuffle=True,aug=True,labels=True):\n        self.df = df.copy()\n        if 'prediction_id' not in df:\n            self.df['prediction_id'] = df[\"patient_id\"].astype(str) + '_' + df[\"laterality\"].astype(str)\n\n        self.prediction_ids = self.df['prediction_id'].unique()\n        self.labels = labels\n        if self.labels ==True:\n            self.labels = self.df.groupby('prediction_id')['cancer'].max()\n        self.path = path\n        self.batch_size = batch_size\n        self.aug=aug\n        self.shuffle = shuffle\n        self.on_epoch_end()\n\n    def __len__(self):\n        \"\"\"Denotes the number of batches per epoch\"\"\"\n        return int(len(self.prediction_ids) / self.batch_size)\n\n    def __getitem__(self, index):\n        \"\"\"Generate one batch of data\"\"\"\n        batch_indexes = self.prediction_ids[index * self.batch_size:(index + 1) * self.batch_size]\n        X, y = self.__data_generation(batch_indexes)\n        return X, y\n\n    def __get_input(self, path):\n        \n#         print('path   ',path)\n        image = tf.keras.preprocessing.image.load_img(path)\n        image_arr = tf.keras.preprocessing.image.img_to_array(image)\n        \n        if self.aug:\n            \n             image_arr=self.augmentor(image_arr)\n\n        \n        return image_arr\n\n    \n    def augmentor(self, images):\n        'Apply data augmentation'\n        images=data_augmentation_layers(images)\n\n        return images\n    \n    \n    def on_epoch_end(self):\n        \"\"\"Updates indexes after each epoch\"\"\"\n        if self.shuffle:\n            self.df = self.df.sample(frac=1).reset_index(drop=True)\n\n    def __data_generation(self, batch_indexes):\n        paths = self.get_paths_images(batch_indexes)\n        X = np.asarray([self.__get_input(path) for path in paths])\n        y = np.array([self.labels[batch_indexes]])\n        return X, y\n\n    def get_paths_images(self, batch_indexes):\n        batch = self.df[self.df['prediction_id'].isin(batch_indexes)]\n        rows_batch = self.get_rows(batch)\n        return self.path + rows_batch[\"patient_id\"].astype(str) + \"/\" + rows_batch[\"image_id\"].astype(\n            str) + \".png\"\n\n    def get_rows(self, batch):\n        \"\"\"Select only 1 MLO view picture per breast\"\"\"\n        only_MLO_view_images = batch[batch['view'] == 'MLO']\n        only_one_per_prediction_id = only_MLO_view_images.groupby('prediction_id')[['patient_id', 'image_id']].max()\n        return only_one_per_prediction_id\n    \n\nimage_dir = f'{base_path}/train_images_processed_cv2_256/'\ndataset_path = train_df\ntrain_gen, val_gen = get_train_val_generator(filename=dataset_path, image_dir=image_dir)\n","metadata":{"execution":{"iopub.status.busy":"2023-01-17T20:45:14.292656Z","iopub.execute_input":"2023-01-17T20:45:14.293074Z","iopub.status.idle":"2023-01-17T20:45:14.561339Z","shell.execute_reply.started":"2023-01-17T20:45:14.293038Z","shell.execute_reply":"2023-01-17T20:45:14.560317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# views=['MLO', 'CC']\n\n# for view in views:\n#     training_generator=Generator('/kaggle/input/rsna-breast-cancer-detection/train.csv', view)\n#     train_tf_gen = tf.data.Dataset.from_generator(training_generator.__getitem__, (tf.float32, tf.float32), (tf.TensorShape([256,256,3,1]), tf.TensorShape([1])))\n#     train_batches = train_tf_gen.batch(48)\n    \nBATCH_SIZE=300\nMETRICS = [\n    keras.metrics.TruePositives(name='tp'),\n    keras.metrics.FalsePositives(name='fp'),\n    keras.metrics.TrueNegatives(name='tn'),\n    keras.metrics.FalseNegatives(name='fn'), \n    keras.metrics.BinaryAccuracy(name='accuracy'),\n    keras.metrics.Precision(name='precision'),\n    keras.metrics.Recall(name='recall'),\n    keras.metrics.AUC(name='auc'),\n    keras.metrics.AUC(name='prc', curve='PR'), # precision-recall curve\n]\n\n# model=make_model((256,256,3), 1)\nmodel=get_model()\n\nmodel.compile(\n    optimizer=keras.optimizers.Adam(learning_rate=1e-3),\n    loss=keras.losses.BinaryCrossentropy(),\n    metrics=METRICS)\nearly_stopping = tf.keras.callbacks.EarlyStopping(\n    monitor='val_prc', \n    verbose=1,\n    patience=20,\n    mode='max',\n    restore_best_weights=True)\n\n!mkdir weights\n\ncheck_pointer = tf.keras.callbacks.ModelCheckpoint(filepath=\"/kaggle/working/weights/model.h5\", verbose=1, save_best_only=True, save_weights_only=True)\n\nhistory = model.fit(\n    train_gen,\n    steps_per_epoch=25,\n    epochs=200,\n    callbacks=[check_pointer],\n    validation_data=val_gen,\n    validation_steps=10,\n    verbose=0\n    )\nacc = history.history['accuracy']\nval_acc = history.history['val_accuracy']\nloss = history.history['loss']\nval_loss = history.history['val_loss']\n\nepochs = range(1,len(acc) + 1)\n\nplt.plot(epochs,acc,'bo',label = 'Training Accuracy')\nplt.plot(epochs,val_acc,'b',label = 'Validation Accuracy')\nplt.title('Training and Validation Accuracy')\nplt.legend()\nplt.figure()\n\nplt.plot(epochs,loss,'bo',label = 'Training loss')\nplt.plot(epochs,val_loss,'b',label = 'Validation Loss')\nplt.title('Training and Validation Loss')\nplt.legend()\nplt.show()\n\nclear_output()\n\n","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2023-01-17T20:45:14.562804Z","iopub.execute_input":"2023-01-17T20:45:14.56319Z","iopub.status.idle":"2023-01-17T20:45:52.191724Z","shell.execute_reply.started":"2023-01-17T20:45:14.563152Z","shell.execute_reply":"2023-01-17T20:45:52.19086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process(f, size=256, save_folder=None, dicom_process = True, extension=\"png\"):\n    \n    patient = f.split('/')[-2]\n    image_name = f.split('/')[-1][:-4]\n    if dicom_process:\n        dicom = pydicom.dcmread(f)\n        img = dicom.pixel_array\n\n        img = (img - img.min()) / (img.max() - img.min())\n\n        if dicom.PhotometricInterpretation == \"MONOCHROME1\":  \n            img = 1 - img\n            \n        image = (img * 255).astype(np.uint8)\n    else:\n        \n        dicom = dicoml.open(f)\n        img = dicom.pixelData()\n\n        img = (img - img.min()) / (img.max() - img.min())\n\n        if dicom.getPixelDataInfo()['PhotometricInterpretation'] == \"MONOCHROME1\":\n            img = 1 - img\n\n        image = (img * 255).astype(np.uint8)\n    \n    img = cv2.resize(image, (size, size))\n    \n    file_name = f'{save_folder}' + f\"{patient}_{image_name}.{extension}\"\n    \n    cv2.imwrite(file_name, img)","metadata":{"execution":{"iopub.status.busy":"2023-01-17T20:45:52.193329Z","iopub.execute_input":"2023-01-17T20:45:52.193969Z","iopub.status.idle":"2023-01-17T20:45:52.203345Z","shell.execute_reply.started":"2023-01-17T20:45:52.193913Z","shell.execute_reply":"2023-01-17T20:45:52.202455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Test Images\n\nDATASET_PATH='/kaggle/input/rsna-breast-cancer-detection/'\ntest_df = pd.read_csv(os.path.join(DATASET_PATH, \"test.csv\"))\ndisplay(test_df.head())\nprint(f'cases: {len(test_df)}')\n\n# Show sample submission example\n\ndf_sub = pd.read_csv(os.path.join(DATASET_PATH, \"sample_submission.csv\"))\ndisplay(df_sub.head())\nprint(f'cases: {len(df_sub)}')","metadata":{"execution":{"iopub.status.busy":"2023-01-17T20:46:02.507853Z","iopub.execute_input":"2023-01-17T20:46:02.508609Z","iopub.status.idle":"2023-01-17T20:46:02.586662Z","shell.execute_reply.started":"2023-01-17T20:46:02.508569Z","shell.execute_reply":"2023-01-17T20:46:02.585607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_dir_pydicom = '/kaggle/tmp/png_file_py/'\nimage_dir_dicomsdl = '/kaggle/tmp/png_file_dic/'\n\n# model_mlo=get_model()\n# model_mlo.load_weights(\"/kaggle/working/weights/model_MLO.h5\")\n# model_cc=get_model()\n# model_cc.load_weights(\"/kaggle/working/weights/model_CC.h5\")\nmodel=get_model()\nmodel.load_weights(\"/kaggle/working/weights/model.h5\")\nos.makedirs(image_dir_pydicom, exist_ok=True)\nos.makedirs(image_dir_dicomsdl, exist_ok=True)\n\n\n\ntest_images = glob.glob(\"/kaggle/input/rsna-breast-cancer-detection/test_images/*/*.dcm\")\nlen(test_images)\n\nstart_time = time.time()\n        \nParallel(n_jobs=4)(\n    delayed(process)(f, size = 256, save_folder = image_dir_pydicom)\n    for f in test_images\n)\n\ntest_dir = f'{image_dir_pydicom}'\ntest_df['img_path']= f'{test_dir}/'\\\n                    + '/' + test_df.patient_id.astype(str)\\\n                    + '_' + test_df.image_id.astype(str)\\\n                    + '.png'\n\n# def load_image(image_path, laterality):\n#     img = tf.io.read_file(image_path)\n#     img = tf.io.decode_png(img).numpy()\n#     img = mm_preprocessor._crop(img)\n#     img = mm_preprocessor._flip_breast_side(img, laterality)\n#     img=(img-np.min(img))/(np.max(img)-np.min(img))\n#     img = np.array(np.stack((np.array(img),)*3, axis=2))\n#     img = resize(img, (256,256))\n#     return img\n\npreds=[]\nfor i in range (len(test_df['img_path'])):\n    image = tf.keras.preprocessing.image.load_img(test_df['img_path'][i])\n    image=tf.expand_dims(np.array(image), 0)\n    if test_df['view'][i]=='MLO':\n        pred=model.predict(np.asarray(image))[0][0]\n    else:\n        pred=0\n    preds.append(pred)\n\n\npred_df = pd.DataFrame({'prediction_id':test_df.prediction_id,\n                        'cancer':preds}).groupby('prediction_id').max().reset_index()\n\n\n\npred_df.to_csv('submission.csv', index=False)\npred_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-17T20:53:41.747165Z","iopub.execute_input":"2023-01-17T20:53:41.74755Z","iopub.status.idle":"2023-01-17T20:53:53.796868Z","shell.execute_reply.started":"2023-01-17T20:53:41.747517Z","shell.execute_reply":"2023-01-17T20:53:53.795473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# image_dir_pydicom = './png_file_py/'\n# image_dir_dicomsdl = './png_file_dic/'\n\n# # model_mlo=get_model()\n# # model_mlo.load_weights(\"/kaggle/working/weights/model_MLO.h5\")\n# # model_cc=get_model()\n# # model_cc.load_weights(\"/kaggle/working/weights/model_CC.h5\")\n# model=get_model()\n# model.load_weights(\"/kaggle/working/weights/model.h5\")\n# os.makedirs(image_dir_pydicom, exist_ok=True)\n# os.makedirs(image_dir_dicomsdl, exist_ok=True)\n\n\n\n# test_images = glob.glob(\"/kaggle/input/rsna-breast-cancer-detection/train_images/*/*.dcm\")\n# test_images = test_images[:100]\n# # start_time = time.time()\n        \n# # Parallel(n_jobs=4)(\n# #     delayed(process)(f, size = 256, save_folder = image_dir_pydicom)\n# #     for f in test_images\n# # )\n\n# test_dir = \"/kaggle/input/rsna-mammography-images-as-pngs/images_as_pngs_cv2_256/train_images_processed_cv2_256/\"\n# test_df['img_path']= f'{test_dir}/'\\\n#                     + '/' + test_df.patient_id.astype(str)\\\n#                     + '/' + test_df.image_id.astype(str)\\\n#                     + '.png'\n\n# test_df=test_df.loc[:10]\n# # def load_image(image_path, laterality):\n# #     img = tf.io.read_file(image_path)\n# #     img = tf.io.decode_png(img).numpy()\n# #     img = mm_preprocessor._crop(img)\n# #     img = mm_preprocessor._flip_breast_side(img, laterality)\n# #     img=(img-np.min(img))/(np.max(img)-np.min(img))\n# #     img = np.array(np.stack((np.array(img),)*3, axis=2))\n# #     img = resize(img, (256,256))\n# #     return img\n\n# preds=[]\n# for i in range (len(test_df['img_path'])):\n#     image = tf.keras.preprocessing.image.load_img(test_df['img_path'][i])\n#     image=tf.expand_dims(np.array(image), 0)\n#     if test_df['view'][i]=='MLO':\n#         pred=model.predict(np.asarray(image))[0][0]\n#     else:\n#         pred=0\n#     preds.append(pred)\n\n\n# pred_df = pd.DataFrame({'prediction_id':test_df.patient_id+test_df.laterality,\n#                         'cancer':preds}).reset_index()\n\n\n\n# pred_df.to_csv('submission.csv', index=False)\n# pred_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-17T20:57:59.055347Z","iopub.execute_input":"2023-01-17T20:57:59.055725Z","iopub.status.idle":"2023-01-17T20:58:14.337474Z","shell.execute_reply.started":"2023-01-17T20:57:59.055693Z","shell.execute_reply":"2023-01-17T20:58:14.335863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds","metadata":{"execution":{"iopub.status.busy":"2023-01-17T20:57:31.583002Z","iopub.execute_input":"2023-01-17T20:57:31.583387Z","iopub.status.idle":"2023-01-17T20:57:31.590675Z","shell.execute_reply.started":"2023-01-17T20:57:31.583357Z","shell.execute_reply":"2023-01-17T20:57:31.589649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}