{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"%%capture\n\ntry:\n    import dicomsdl\nexcept:\n   !pip install /kaggle/input/read-dicom-set/dicom_read/dicomsdl-0.109.1-cp37-cp37m-manylinux_2_12_x86_64.manylinux2010_x86_64.whl\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-02-26T13:34:20.499662Z","iopub.execute_input":"2023-02-26T13:34:20.501035Z","iopub.status.idle":"2023-02-26T13:34:20.518965Z","shell.execute_reply.started":"2023-02-26T13:34:20.500965Z","shell.execute_reply":"2023-02-26T13:34:20.516814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install /kaggle/input/rsna-lib/protobuf-3.19.6-py2.py3-none-any.whl","metadata":{"execution":{"iopub.status.busy":"2023-02-26T13:34:20.520918Z","iopub.execute_input":"2023-02-26T13:34:20.521935Z","iopub.status.idle":"2023-02-26T13:34:30.074687Z","shell.execute_reply.started":"2023-02-26T13:34:20.521884Z","shell.execute_reply":"2023-02-26T13:34:30.073482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install /kaggle/input/d/markwijkhuizen/keras-cv-attention-models/keras_cv_attention_models-1.3.9-py3-none-any.whl","metadata":{"execution":{"iopub.status.busy":"2023-02-26T13:34:30.076841Z","iopub.execute_input":"2023-02-26T13:34:30.077294Z","iopub.status.idle":"2023-02-26T13:34:40.258861Z","shell.execute_reply.started":"2023-02-26T13:34:30.07725Z","shell.execute_reply":"2023-02-26T13:34:40.25623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport os\nimport pydicom\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport tensorflow as tf\nimport tensorflow_addons as tfa\nimport tensorflow_io as tfio\nimport wandb\nimport math\nimport warnings\nimport cv2\nimport dicomsdl as dicoml\nimport logging\nimport glob\nimport time\n\nfrom keras import backend as K\nfrom tensorflow.keras.layers import *\nfrom tensorflow import keras\nfrom tensorflow.keras import models\nfrom PIL import Image\nfrom pandas.core.common import SettingWithCopyWarning\nfrom joblib import Parallel, delayed\nfrom tensorflow.keras.applications import ResNet50\nfrom IPython.display import Image\nfrom torch.utils.data import Dataset\nfrom keras_cv_attention_models import convnext\n\nwarnings.simplefilter(action='ignore', category=SettingWithCopyWarning)\n\ndf = pd.read_csv(r'/kaggle/input/rsna-breast-cancer-detection/train.csv')","metadata":{"execution":{"iopub.status.busy":"2023-02-26T13:34:40.263175Z","iopub.execute_input":"2023-02-26T13:34:40.263628Z","iopub.status.idle":"2023-02-26T13:34:53.852601Z","shell.execute_reply.started":"2023-02-26T13:34:40.263589Z","shell.execute_reply":"2023-02-26T13:34:53.850465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN_BASE_PATH = '/kaggle/input/rsna-breast-cancer-512-pngs'\nTEST_BASE_PATH = '/kaggle/working/test'\nNUM_EPOCHS = 500\nIMG_PX_SIZE = 1024\nINPUT_SHAPE = (IMG_PX_SIZE, IMG_PX_SIZE, 3)\nLEARNING_RATE = 0.01\nBATCH_SIZE = 8\nIMAGE_SAVE_DIR = '/kaggle/working/tmp/'\nAUTOTUNE = tf.data.AUTOTUNE","metadata":{"execution":{"iopub.status.busy":"2023-02-26T13:34:53.854941Z","iopub.execute_input":"2023-02-26T13:34:53.855792Z","iopub.status.idle":"2023-02-26T13:34:53.862658Z","shell.execute_reply.started":"2023-02-26T13:34:53.855734Z","shell.execute_reply":"2023-02-26T13:34:53.861382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2023-02-26T13:34:53.864105Z","iopub.execute_input":"2023-02-26T13:34:53.864434Z","iopub.status.idle":"2023-02-26T13:34:53.921954Z","shell.execute_reply.started":"2023-02-26T13:34:53.864408Z","shell.execute_reply":"2023-02-26T13:34:53.920239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Cleaning","metadata":{}},{"cell_type":"code","source":"# Drop rows which are not MLO or CC views.\ndf = df[df['view'].str.contains('MLO|CC') == True]","metadata":{"execution":{"iopub.status.busy":"2023-02-26T13:34:53.923353Z","iopub.execute_input":"2023-02-26T13:34:53.923713Z","iopub.status.idle":"2023-02-26T13:34:53.970719Z","shell.execute_reply.started":"2023-02-26T13:34:53.923682Z","shell.execute_reply":"2023-02-26T13:34:53.969153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def remove_odd_occurances(df):\n    occur = df.groupby(['patient_id']).size()\n    for occ_index, occ_row in occur.items():\n        if occ_row % 2 != 0:\n            patient = df[df['patient_id']==occ_index]\n            lats = patient.groupby(['laterality']).size()\n            for lats_index, lats_row in lats.items():\n                if lats_row % 2 != 0:\n                    view = patient[patient['laterality']==lats_index]\n                    views = view.groupby(['view']).size()\n                    for views_index, views_row in views.items():\n                        if views_row % 2 == 0:\n\n                            to_drop = view[view['view']==views_index].index\n                            df.drop(to_drop[-1], axis=0, inplace=True)\n#remove_odd_occurances(df)","metadata":{"execution":{"iopub.status.busy":"2023-02-26T13:34:53.972867Z","iopub.execute_input":"2023-02-26T13:34:53.97323Z","iopub.status.idle":"2023-02-26T13:34:53.983176Z","shell.execute_reply.started":"2023-02-26T13:34:53.973201Z","shell.execute_reply":"2023-02-26T13:34:53.982017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_classes(df):\n    x = df['cancer'].value_counts().astype(np.int32)\n    sns.barplot(x=x.keys(), y=x)\nplot_classes(df)","metadata":{"execution":{"iopub.status.busy":"2023-02-26T13:34:53.984998Z","iopub.execute_input":"2023-02-26T13:34:53.985326Z","iopub.status.idle":"2023-02-26T13:34:54.157275Z","shell.execute_reply.started":"2023-02-26T13:34:53.985295Z","shell.execute_reply":"2023-02-26T13:34:54.156387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def generate_paths(df, path):\n    df['img_path'] = df.apply(lambda i: f'{path}/{i.patient_id}_{i.image_id}.png', axis=1)\ngenerate_paths(df, TRAIN_BASE_PATH)","metadata":{"execution":{"iopub.status.busy":"2023-02-26T13:34:54.160997Z","iopub.execute_input":"2023-02-26T13:34:54.161568Z","iopub.status.idle":"2023-02-26T13:34:55.303732Z","shell.execute_reply.started":"2023-02-26T13:34:54.161526Z","shell.execute_reply":"2023-02-26T13:34:55.302385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# PreProcessing","metadata":{}},{"cell_type":"code","source":"def crop_borders(img, l=0.02, r=0.02, u=0.04, d=0.04):\n    nrows, ncols, _ = img.shape\n\n    # Get the start and end rows and columns\n    l_crop = int(ncols * l)\n    r_crop = int(ncols * (1 - r))\n    u_crop = int(nrows * u)\n    d_crop = int(nrows * (1 - d))\n\n    cropped_img = img[u_crop:d_crop, l_crop:r_crop]\n\n    return cropped_img\n    \ndef clahe(img):\n    img = img.astype(np.uint8)\n    clahe = cv2.createCLAHE(clipLimit = 3)\n    final_img = clahe.apply(img)\n    return final_img\n    \ndef pad_image(img):\n    x_to_pad = int((IMG_PX_SIZE - img.shape[0]) / 2)\n    y_to_pad = int((IMG_PX_SIZE - img.shape[1]) / 2) \n\n        \n    final_image = np.pad(img, ((x_to_pad, x_to_pad+1), (y_to_pad, y_to_pad+1)), 'constant')\n    return final_image \n\ndef flip_left(img):\n    left_sum = np.sum(img[:, :int(IMG_PX_SIZE/2)])\n    right_sum = np.sum(img[:, int(IMG_PX_SIZE/2):])\n    if right_sum > left_sum:\n        img = np.fliplr(img)\n    return img\n\ndef crop_desired_region(img):\n    '''\n     Cropping desired region. \n    RSNA training images contain blank spaces. \n    To remove the region, apply non-zero pixels and cut.\n    And then resize to the original size.\n    '''\n    img = img.astype(np.uint8)\n    coords = cv2.findNonZero(img)\n    x, y, w, h = cv2.boundingRect(coords) \n    rect = img[y:y+h+20, x:x+w-20]\n    rect_originalSized = cv2.resize(rect, (IMG_PX_SIZE, IMG_PX_SIZE), interpolation=cv2.INTER_AREA)\n    return rect_originalSized[:,:,np.newaxis]\n\ndef normalize(image):\n    # Repeat channels to create 3 channel images required by pretrained ConvNextV2 models\n    image = np.repeat(image[:,:,np.newaxis], 3, -1)\n    # Cast to float 32\n    image = tf.cast(image, tf.float32)\n    # Normalize with respect to ImageNet mean/std\n    image = tf.keras.applications.imagenet_utils.preprocess_input(image, mode='torch')\n\n    return image\n    \ndef preprocess(img):\n    #Crop breast region\n    img = crop_desired_region(img)\n    # Crop borders\n    img = crop_borders(img)\n    # Fix constrast\n    img = clahe(img)\n    # Pad Image\n    img = pad_image(img)\n    # Flip left if laterality is right\n    img = flip_left(img) \n    \n    img = normalize(img)\n    return img\n\ndef get_image_data(path):\n    image = tf.keras.utils.load_img(path, \n                                    target_size=INPUT_SHAPE,\n                                    color_mode='grayscale',\n    )\n    \n    image_arr = tf.keras.utils.img_to_array(image)\n    plt.imshow(image_arr)\n    plt.show()\n    \n    image_arr = preprocess(image_arr)\n    plt.imshow(image_arr)\n    plt.show()\n\n    \n    return image_arr","metadata":{"execution":{"iopub.status.busy":"2023-02-26T13:34:55.305003Z","iopub.execute_input":"2023-02-26T13:34:55.306377Z","iopub.status.idle":"2023-02-26T13:34:55.326282Z","shell.execute_reply.started":"2023-02-26T13:34:55.306304Z","shell.execute_reply":"2023-02-26T13:34:55.32389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Dataset","metadata":{}},{"cell_type":"code","source":"class RSNAGenerator:\n    def __init__(self, df, path, batch_size=32, target_size=(32, 32), shuffle=True, labels=True):\n        self.df = df.copy()\n        if 'prediction_id' not in df:\n            self.df['prediction_id'] = df[\"patient_id\"].astype(str) + '_' + df[\"laterality\"].astype(str)    \n        self.labels = labels\n        if self.labels:\n            self.labels = self.df['cancer'].values\n        self.patient_ids = self.df['patient_id'].unique()\n        self.path = path\n        self.batch_size = batch_size\n        self.shuffle = shuffle\n        self.target_size = target_size\n        self.on_epoch_end()\n\n    def __len__(self):\n        \"\"\"Denotes the number of batches per epoch\"\"\"\n        return math.ceil(len(self.patient_ids) / self.batch_size)\n    \n    def on_epoch_end(self):\n        \"\"\"Updates indexes after each epoch\"\"\"\n        if self.shuffle:\n            self.df = self.df.sample(frac=1).reset_index(drop=True)\n\n    def __getitem__(self, index):\n        \"\"\"Generate one batch of data\"\"\"\n        batch_indexes = self.df.index[index * self.batch_size:(index + 1) * self.batch_size]\n        X, y = self.__data_generation(batch_indexes)\n        return X, y\n    \n    def __data_generation(self, batch_indexes):\n        batch = self.df.iloc[batch_indexes]\n        \n        X = np.asarray([get_image_data(path) for path in batch['img_path']])\n        y = np.asarray(batch['cancer'])\n\n        return X, y\n    \n    def __call__(self):\n        for i in range(self.__len__()):\n            yield self.__getitem__(i)\n            \n            if i == self.__len__()-1:\n                self.on_epoch_end()","metadata":{"execution":{"iopub.status.busy":"2023-02-26T13:34:55.328019Z","iopub.execute_input":"2023-02-26T13:34:55.328594Z","iopub.status.idle":"2023-02-26T13:34:55.345291Z","shell.execute_reply.started":"2023-02-26T13:34:55.32853Z","shell.execute_reply":"2023-02-26T13:34:55.344203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model","metadata":{}},{"cell_type":"markdown","source":"#### Augmentator","metadata":{}},{"cell_type":"code","source":"data_augmentation = keras.Sequential(\n    [\n        RandomFlip(\"horizontal\"),\n        RandomRotation(factor=0.02),\n    ],\n)","metadata":{"execution":{"iopub.status.busy":"2023-02-26T13:34:55.347041Z","iopub.execute_input":"2023-02-26T13:34:55.347447Z","iopub.status.idle":"2023-02-26T13:34:55.503069Z","shell.execute_reply.started":"2023-02-26T13:34:55.347407Z","shell.execute_reply":"2023-02-26T13:34:55.500391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training","metadata":{}},{"cell_type":"markdown","source":"#### Model","metadata":{}},{"cell_type":"code","source":"base_model = convnext.ConvNeXtV2Tiny(input_shape=INPUT_SHAPE,\n                                     num_classes=0)\n#base_model.load_weights('/kaggle/input/tfkeras-efficientnetsv2/21k_notop/efficientnetv2-s-21k_notop.h5')\nbase_model.trainable = False\n    \ndef get_model():\n    inputs = Input(shape=INPUT_SHAPE)\n    x = data_augmentation(inputs)\n    \n    x = base_model(x, training=False)\n    x = GlobalAveragePooling2D()(x)\n    x = Dropout(0.3)(x)\n    \n    x = Dense(1)(x)\n    out = Activation('sigmoid')(x)\n    \n    model = keras.Model(inputs=inputs, outputs=out)\n    \n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2023-02-26T13:34:55.505027Z","iopub.execute_input":"2023-02-26T13:34:55.505386Z","iopub.status.idle":"2023-02-26T13:35:18.513477Z","shell.execute_reply.started":"2023-02-26T13:34:55.505347Z","shell.execute_reply":"2023-02-26T13:35:18.512368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Evaluation","metadata":{}},{"cell_type":"code","source":"def pfbeta(y_true, y_pred, beta = 1):    \n    y_true_count = tf.reduce_sum(y_true)\n    ctp = tf.reduce_sum(y_true * y_pred)\n    cfp = tf.reduce_sum((1 - y_true) * y_pred)\n    zero = tf.constant([0], dtype=tf.float32)\n    beta_squared = beta * beta\n    c_precision = tf.where(ctp + cfp == zero, zero, ctp / (ctp + cfp))\n    c_recall =  tf.where(y_true_count == zero, zero, ctp / y_true_count)\n    return tf.where(c_precision + c_recall == zero, zero, (1 + beta_squared) * (c_precision * c_recall) / (beta_squared * c_precision + c_recall))","metadata":{"execution":{"iopub.status.busy":"2023-02-26T13:35:18.514991Z","iopub.execute_input":"2023-02-26T13:35:18.515425Z","iopub.status.idle":"2023-02-26T13:35:18.526172Z","shell.execute_reply.started":"2023-02-26T13:35:18.515382Z","shell.execute_reply":"2023-02-26T13:35:18.52381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Loss","metadata":{}},{"cell_type":"code","source":"def weighted_binary_crossentropy(weight_zero=0, weight_one=0):\n    def loss(y_true, y_pred):\n        y_true = tf.cast(y_true, dtype=tf.float64)\n        y_pred = tf.cast(y_pred, dtype=tf.float64)\n        \n        # calculate the binary cross entropy\n        bin_crossentropy = keras.backend.binary_crossentropy(y_true, y_pred)\n        # apply the weights\n        weights = y_true * weight_one + (1. - y_true) * weight_zero\n        weighted_bin_crossentropy = weights * bin_crossentropy \n\n        return keras.backend.mean(weighted_bin_crossentropy)\n    return loss","metadata":{"execution":{"iopub.status.busy":"2023-02-26T13:35:18.528859Z","iopub.execute_input":"2023-02-26T13:35:18.529573Z","iopub.status.idle":"2023-02-26T13:35:18.542772Z","shell.execute_reply.started":"2023-02-26T13:35:18.529524Z","shell.execute_reply":"2023-02-26T13:35:18.541204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Plots","metadata":{}},{"cell_type":"code","source":"def plot_results(loss, val_loss):\n    x = np.arange(0, len(loss)) \n         \n    sns.lineplot(x=x, y=loss)\n    sns.lineplot(x=x, y=val_loss)\n    \n    #sns.lineplot(ax=ax[1], x=x, y=history.history[\"pfbeta\"])\n    #sns.lineplot(ax=ax[1], x=x, y=history.history[\"val_pfbeta\"])\n    plt.legend()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-26T13:35:18.544337Z","iopub.execute_input":"2023-02-26T13:35:18.545198Z","iopub.status.idle":"2023-02-26T13:35:18.556691Z","shell.execute_reply.started":"2023-02-26T13:35:18.545152Z","shell.execute_reply":"2023-02-26T13:35:18.554932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_training_data(df, validation_split=0.15):\n    patient_ids = df['patient_id'].unique()\n    split = int((1 - validation_split)*len(patient_ids))\n    train_ids = patient_ids[:split]\n    val_ids = patient_ids[split:]\n    \n    train_df = df[df['patient_id'].isin(train_ids)]\n    val_df = df[df['patient_id'].isin(val_ids)]\n    \n    train_gen = RSNAGenerator(train_df, target_size=(IMG_PX_SIZE, IMG_PX_SIZE), batch_size=BATCH_SIZE, path=TRAIN_BASE_PATH)    \n    val_gen = RSNAGenerator(val_df, target_size=(IMG_PX_SIZE, IMG_PX_SIZE), batch_size=BATCH_SIZE, path=TRAIN_BASE_PATH)    \n    \n    train_ds = tf.data.Dataset.from_generator(train_gen,\n                                        output_types = (tf.float64, tf.float32),\n                                        output_shapes = ((tf.TensorShape((BATCH_SIZE, IMG_PX_SIZE, IMG_PX_SIZE, 3)), \n                                                          tf.TensorShape((BATCH_SIZE, )))\n                                                        )\n                                       )\n    \n    val_ds = tf.data.Dataset.from_generator(val_gen,\n                                        output_types = (tf.float64, tf.float32),\n                                        output_shapes = ((tf.TensorShape((BATCH_SIZE, IMG_PX_SIZE, IMG_PX_SIZE, 3)), \n                                                          tf.TensorShape((BATCH_SIZE, )))\n                                                        )\n                                       )\n    return train_ds, val_ds","metadata":{"execution":{"iopub.status.busy":"2023-02-26T13:35:18.558467Z","iopub.execute_input":"2023-02-26T13:35:18.558859Z","iopub.status.idle":"2023-02-26T13:35:18.5723Z","shell.execute_reply.started":"2023-02-26T13:35:18.558805Z","shell.execute_reply":"2023-02-26T13:35:18.570386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"@tf.function\ndef train_step(x, y):\n    with tf.GradientTape() as tape:\n        logits = model(x, training=True)\n        loss_value = loss_fn(y, logits)\n    grads = tape.gradient(loss_value, model.trainable_weights)\n    optimizer.apply_gradients(zip(grads, model.trainable_weights))\n    return loss_value\n\n@tf.function\ndef test_step(x, y):\n    val_logits = model(x, training=False)\n    val_loss_value = loss_fn(y, val_logits)\n    return val_loss_value","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-02-26T13:35:18.573968Z","iopub.execute_input":"2023-02-26T13:35:18.574581Z","iopub.status.idle":"2023-02-26T13:35:18.590508Z","shell.execute_reply.started":"2023-02-26T13:35:18.57453Z","shell.execute_reply":"2023-02-26T13:35:18.589563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ds, val_ds = get_training_data(df)\n\nmodel = get_model()\nloss_fn = weighted_binary_crossentropy(weight_zero=0.1, weight_one=0.9)\nval_loss_fn = weighted_binary_crossentropy(weight_zero=0.1, weight_one=0.9)\n#val_acc_metric = pfbeta\n\noptimizer = keras.optimizers.Adam(learning_rate=LEARNING_RATE\n)\n\ndef train(train_ds, val_ds, patience = 10): \n    all_loss = []\n    all_val_loss = []\n    all_metric = []\n    best = float('inf')\n    wait = 0\n    \n    for epoch in range(NUM_EPOCHS):\n        start_time = time.time()\n        print(\"\\nStart of epoch %d\" % (epoch,))\n      \n        # Iterate over the batches of the dataset.\n        for step, (x_batch_train, y_batch_train) in enumerate(train_ds):\n            y_batch_train = tf.reshape(y_batch_train, [BATCH_SIZE, 1])  \n        \n            loss_value = train_step(x_batch_train, y_batch_train)\n                                \n        for _, (x_batch_val, y_batch_val) in enumerate(val_ds):\n            y_batch_val = tf.reshape(y_batch_val, [BATCH_SIZE, 1])  \n            val_loss_value = test_step(x_batch_val, y_batch_val)\n            \n        print(\"Probabilistic F1 Score: %.4f\" % (float(val_loss_value),))\n        print(\"Time taken: %.2fs\" % (time.time() - start_time))\n        all_loss.append(loss_value.numpy())   \n        all_val_loss.append(val_loss_value.numpy())\n        \n        wait += 1\n        if val_loss_value < best:\n            best = val_loss_value\n            wait = 0\n            model.save('/kaggle/working/model/1.h5')\n        if wait >= patience:\n            break\n        \n    plot_results(all_loss, all_val_loss)\n    \n#train(train_ds, val_ds)","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-02-26T13:35:18.592121Z","iopub.execute_input":"2023-02-26T13:35:18.592788Z","iopub.status.idle":"2023-02-26T13:35:21.04444Z","shell.execute_reply.started":"2023-02-26T13:35:18.592745Z","shell.execute_reply":"2023-02-26T13:35:21.042768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"def process(row, size=256, save_folder=None, extension=\"png\"):\n    pat_id = row.patient_id\n    img_id = row.image_id\n\n    path = f'/kaggle/input/rsna-breast-cancer-detection/test_images/{pat_id}/{img_id}.dcm'\n\n    dicom = dicoml.open(path)\n    img = dicom.pixelData()\n\n    img = (img - img.min()) / (img.max() - img.min())\n\n    if dicom.getPixelDataInfo()['PhotometricInterpretation'] == \"MONOCHROME1\":\n        img = 1 - img\n\n    image = (img * 255).astype(np.uint8)\n    \n    img = cv2.resize(image, (size, size))\n\n    file_name = f'{save_folder}/{pat_id}_{img_id}.{extension}'\n    \n    cv2.imwrite(file_name, img)","metadata":{"execution":{"iopub.status.busy":"2023-02-26T13:35:21.046246Z","iopub.execute_input":"2023-02-26T13:35:21.046673Z","iopub.status.idle":"2023-02-26T13:35:21.0541Z","shell.execute_reply.started":"2023-02-26T13:35:21.046625Z","shell.execute_reply":"2023-02-26T13:35:21.053283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_csv(f'/kaggle/input/rsna-breast-cancer-detection/test.csv')\nmodel = get_model()\n\nos.makedirs(IMAGE_SAVE_DIR, exist_ok=True)\nParallel(n_jobs=-1)(\n    delayed(process)(row, size=IMG_PX_SIZE, save_folder=IMAGE_SAVE_DIR)\n    for row in test_df.itertuples()\n)\n\nbatch = np.empty((1, IMG_PX_SIZE, IMG_PX_SIZE, 3))\nfor index, row in enumerate(test_df.itertuples()):\n    path = f'/kaggle/working/tmp/{row.patient_id}_{row.image_id}.png'\n    batch[0,:,:,:] = get_image_data(path)\n    y_pred = model.predict(batch).squeeze()\n    test_df.at[index, 'cancer'] = y_pred\n        \n    batch = np.empty((1, IMG_PX_SIZE, IMG_PX_SIZE, 3))\n        \n!rm -r tmp\nsub = test_df[['prediction_id', 'cancer']]\nsub = sub.groupby('prediction_id').mean()\nsub = sub.sort_index()\nsub.to_csv('submission.csv',index=True)","metadata":{"execution":{"iopub.status.busy":"2023-02-26T13:35:21.055296Z","iopub.execute_input":"2023-02-26T13:35:21.055571Z","iopub.status.idle":"2023-02-26T13:35:40.52776Z","shell.execute_reply.started":"2023-02-26T13:35:21.055543Z","shell.execute_reply":"2023-02-26T13:35:40.525095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(sub)","metadata":{"execution":{"iopub.status.busy":"2023-02-26T13:35:40.530077Z","iopub.execute_input":"2023-02-26T13:35:40.530504Z","iopub.status.idle":"2023-02-26T13:35:40.540874Z","shell.execute_reply.started":"2023-02-26T13:35:40.530467Z","shell.execute_reply":"2023-02-26T13:35:40.538962Z"},"trusted":true},"execution_count":null,"outputs":[]}]}