{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install /kaggle/input/rsna-2022-whl/{pydicom-2.3.0-py3-none-any.whl,pylibjpeg-1.4.0-py3-none-any.whl,python_gdcm-3.0.15-cp37-cp37m-manylinux_2_17_x86_64.manylinux2014_x86_64.whl}\n!pip install /kaggle/input/fastai017-whl/dicomsdl-0.109.1-cp37-cp37m-manylinux_2_12_x86_64.manylinux2010_x86_64.whl\n#!pip install /kaggle/input/rsna-2022-whl/{torch-1.12.1-cp37-cp37m-manylinux1_x86_64.whl,torchvision-0.13.1-cp37-cp37m-manylinux1_x86_64.whl}","metadata":{"execution":{"iopub.status.busy":"2023-01-01T20:37:27.934783Z","iopub.execute_input":"2023-01-01T20:37:27.935329Z","iopub.status.idle":"2023-01-01T20:38:29.052628Z","shell.execute_reply.started":"2023-01-01T20:37:27.935221Z","shell.execute_reply":"2023-01-01T20:38:29.051395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pylibjpeg\nfrom tensorflow.keras.optimizers import Adam\nimport cv2\nimport pydicom\nimport pandas as pd\nimport numpy as np\nfrom glob import glob\nfrom tqdm import tqdm\n\ntrain_csv = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')","metadata":{"execution":{"iopub.status.busy":"2023-01-01T20:38:29.054944Z","iopub.execute_input":"2023-01-01T20:38:29.05537Z","iopub.status.idle":"2023-01-01T20:38:34.709076Z","shell.execute_reply.started":"2023-01-01T20:38:29.055328Z","shell.execute_reply":"2023-01-01T20:38:34.708067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import dicomsdl as dicoml\ndef preprocessFast(f, size=256, dicom_process = True, extension=\"png\"):\n    \n    patient = f.split('/')[-2]\n    image_name = f.split('/')[-1][:-4]\n    if dicom_process:\n        dicom = pydicom.dcmread(f)\n        img = dicom.pixel_array\n\n        img = (img - img.min()) / (img.max() - img.min())\n\n        if dicom.PhotometricInterpretation == \"MONOCHROME1\":  \n            img = 1 - img\n            \n        image = (img * 255).astype(np.uint8)\n    else:\n        \n        dicom = dicoml.open(f)\n        img = dicom.pixelData()\n\n        img = (img - img.min()) / (img.max() - img.min())\n\n        if dicom.getPixelDataInfo()['PhotometricInterpretation'] == \"MONOCHROME1\":\n            img = 1 - img\n\n        image = (img * 255).astype(np.uint8)\n    \n    img = cv2.resize(image, (size, size))\n    img = np.expand_dims(img,-1)\n    return img","metadata":{"execution":{"iopub.status.busy":"2023-01-01T20:38:34.71079Z","iopub.execute_input":"2023-01-01T20:38:34.711202Z","iopub.status.idle":"2023-01-01T20:38:34.727457Z","shell.execute_reply.started":"2023-01-01T20:38:34.711163Z","shell.execute_reply":"2023-01-01T20:38:34.726163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1 = train_csv[train_csv['cancer']==0]\nprint(len(df1))\ndf1.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-01T20:38:34.730422Z","iopub.execute_input":"2023-01-01T20:38:34.730779Z","iopub.status.idle":"2023-01-01T20:38:34.768583Z","shell.execute_reply.started":"2023-01-01T20:38:34.730744Z","shell.execute_reply":"2023-01-01T20:38:34.767647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df2 = train_csv[train_csv['cancer']==1]\nprint(len(df2))\ndf2.head(100)","metadata":{"execution":{"iopub.status.busy":"2023-01-01T20:38:34.770069Z","iopub.execute_input":"2023-01-01T20:38:34.770759Z","iopub.status.idle":"2023-01-01T20:38:34.796046Z","shell.execute_reply.started":"2023-01-01T20:38:34.770715Z","shell.execute_reply":"2023-01-01T20:38:34.795155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"healthy_paths = []\ncancer_paths =  []\nfor i in tqdm(glob('/kaggle/input/rsna-breast-cancer-256-pngs/*')):\n    \n    id = i.split('_')[1][:-4]\n    #print(id)\n    if (int(id) in df2['image_id'].values):\n        cancer_paths.append(i)\n    else:\n        healthy_paths.append(i)","metadata":{"execution":{"iopub.status.busy":"2023-01-01T20:38:34.79744Z","iopub.execute_input":"2023-01-01T20:38:34.798011Z","iopub.status.idle":"2023-01-01T20:38:35.962547Z","shell.execute_reply.started":"2023-01-01T20:38:34.797975Z","shell.execute_reply":"2023-01-01T20:38:35.961518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(cancer_paths)","metadata":{"execution":{"iopub.status.busy":"2023-01-01T20:38:35.963916Z","iopub.execute_input":"2023-01-01T20:38:35.964794Z","iopub.status.idle":"2023-01-01T20:38:35.972161Z","shell.execute_reply.started":"2023-01-01T20:38:35.964755Z","shell.execute_reply":"2023-01-01T20:38:35.97097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom PIL import Image\nfrom matplotlib.image import imread\n\nX_train = []\nY_train = []\n\n\nfor i in tqdm(healthy_paths[0:1158]):\n    img = imread(i)\n    X_train.append(img)\n    Y_train.append(0)\n\nfor i in tqdm(cancer_paths):  \n    img = imread(i)\n    X_train.append(img)\n    Y_train.append(1)\n      ","metadata":{"execution":{"iopub.status.busy":"2023-01-01T20:38:35.973897Z","iopub.execute_input":"2023-01-01T20:38:35.974311Z","iopub.status.idle":"2023-01-01T20:38:50.229429Z","shell.execute_reply.started":"2023-01-01T20:38:35.974275Z","shell.execute_reply":"2023-01-01T20:38:50.228443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(np.unique(Y_train))","metadata":{"execution":{"iopub.status.busy":"2023-01-01T20:38:50.230913Z","iopub.execute_input":"2023-01-01T20:38:50.231747Z","iopub.status.idle":"2023-01-01T20:38:50.238229Z","shell.execute_reply.started":"2023-01-01T20:38:50.231709Z","shell.execute_reply":"2023-01-01T20:38:50.237148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import pydicom\n\n# X_train = []\n# Y_train = []\n\n# for patient in tqdm(df1['patient_id'].unique()[0:150]):\n#     healthy_path = (f'/kaggle/input/rsna-breast-cancer-detection/train_images/{patient}/*')\n#     for i in glob(healthy_path):\n#             print(i)\n\n#             ds = pydicom.dcmread(i)\n#             pixel_array_numpy = ds.pixel_array\n#             pixel_array_numpy = cv2.resize(pixel_array_numpy, dsize=(256, 256))\n            \n\n#             X_train.append(pixel_array_numpy)\n#             Y_train.append(0)\n#             print(len(X_train))\n#             #plt.imshow(pixel_array_numpy, cmap='gray')\n#             #plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-01T20:38:50.243062Z","iopub.execute_input":"2023-01-01T20:38:50.243433Z","iopub.status.idle":"2023-01-01T20:38:50.270563Z","shell.execute_reply.started":"2023-01-01T20:38:50.243406Z","shell.execute_reply":"2023-01-01T20:38:50.269497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for patient in tqdm(df2['patient_id'].unique()[0:150]):\n#     sick_path = (f'/kaggle/input/rsna-breast-cancer-detection/train_images/{patient}/*')\n#     for i in glob(sick_path):\n#             print(i)\n\n#             ds = pydicom.dcmread(i)\n#             pixel_array_numpy = ds.pixel_array\n#             pixel_array_numpy = cv2.resize(pixel_array_numpy, dsize=(256, 256))\n#             #print(pixel_array_numpy.shape)\n\n#             X_train.append(pixel_array_numpy)\n#             Y_train.append(1)\n#             print(len(X_train))\n#             #plt.imshow(pixel_array_numpy, cmap='gray')\n#             #plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-01T20:38:50.272135Z","iopub.execute_input":"2023-01-01T20:38:50.272551Z","iopub.status.idle":"2023-01-01T20:38:50.280571Z","shell.execute_reply.started":"2023-01-01T20:38:50.272516Z","shell.execute_reply":"2023-01-01T20:38:50.279622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = np.array(X_train)\nY_train = np.array(Y_train) \n\n# X_train_rgb = np.repeat(X_train[..., np.newaxis], 3, -1)\n# print(X_train_rgb.shape)  ","metadata":{"execution":{"iopub.status.busy":"2023-01-01T20:38:50.283607Z","iopub.execute_input":"2023-01-01T20:38:50.283886Z","iopub.status.idle":"2023-01-01T20:38:50.466571Z","shell.execute_reply.started":"2023-01-01T20:38:50.28386Z","shell.execute_reply":"2023-01-01T20:38:50.465564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = np.expand_dims(X_train,-1)\nX_train.shape","metadata":{"execution":{"iopub.status.busy":"2023-01-01T20:38:50.468006Z","iopub.execute_input":"2023-01-01T20:38:50.468484Z","iopub.status.idle":"2023-01-01T20:38:50.476605Z","shell.execute_reply.started":"2023-01-01T20:38:50.468434Z","shell.execute_reply":"2023-01-01T20:38:50.474471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Dense","metadata":{}},{"cell_type":"code","source":"# from tensorflow.keras.optimizers import Adam\n# from keras.models import Sequential\n# from tensorflow.keras.applications import DenseNet121\n# from keras import layers\n\n# effnet = DenseNet121(\n#         weights='imagenet',\n#         include_top=False,\n#         input_shape=(256,256,3)\n# )\n\n# Dense_model = Sequential()\n# Dense_model.add(effnet)\n# Dense_model.add(layers.GlobalAveragePooling2D())\n# Dense_model.add(layers.Dropout(0.5))\n# Dense_model.add(layers.Dense(1024,activation='relu'))\n# Dense_model.add(layers.Dense(256,activation='relu'))\n# Dense_model.add(layers.Dense(1, activation='sigmoid'))\n\n# Dense_model.compile(\n#         loss='binary_crossentropy',\n#         optimizer=Adam(lr=0.001),\n#         metrics=['accuracy'],\n# )\n\n# Dense_model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-01-01T20:38:50.478393Z","iopub.execute_input":"2023-01-01T20:38:50.479117Z","iopub.status.idle":"2023-01-01T20:38:50.485346Z","shell.execute_reply.started":"2023-01-01T20:38:50.479066Z","shell.execute_reply":"2023-01-01T20:38:50.484105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.model_selection import train_test_split\n# x_train, x_test, y_train, y_test = train_test_split(X_train_rgb, Y_train, test_size=0.20, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-01-01T20:38:50.486982Z","iopub.execute_input":"2023-01-01T20:38:50.487404Z","iopub.status.idle":"2023-01-01T20:38:50.497611Z","shell.execute_reply.started":"2023-01-01T20:38:50.48737Z","shell.execute_reply":"2023-01-01T20:38:50.496644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Dense_model.fit(x_train, y_train, validation_data=(x_test,y_test), epochs=20, batch_size=64)","metadata":{"execution":{"iopub.status.busy":"2023-01-01T20:38:50.498791Z","iopub.execute_input":"2023-01-01T20:38:50.499723Z","iopub.status.idle":"2023-01-01T20:38:50.507411Z","shell.execute_reply.started":"2023-01-01T20:38:50.499684Z","shell.execute_reply":"2023-01-01T20:38:50.506402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# index = 58\n# print(Dense_model.predict(np.expand_dims(x_test[index],0)))\n# plt.imshow(x_train[index])\n# y_test[index]","metadata":{"execution":{"iopub.status.busy":"2023-01-01T20:38:50.508865Z","iopub.execute_input":"2023-01-01T20:38:50.509284Z","iopub.status.idle":"2023-01-01T20:38:50.518156Z","shell.execute_reply.started":"2023-01-01T20:38:50.509208Z","shell.execute_reply":"2023-01-01T20:38:50.5168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test = []\n# for i in glob('/kaggle/input/rsna-breast-cancer-detection/test_images/10008/*'):\n#             print(i)\n\n#             ds = pydicom.dcmread(i)\n#             pixel_array_numpy = ds.pixel_array\n#             pixel_array_numpy = cv2.resize(pixel_array_numpy, dsize=(256, 256))\n#             test.append(pixel_array_numpy)\n# test = np.array(test)\n\n# test = np.repeat(test[..., np.newaxis], 3, -1)\n# print(test.shape)  ","metadata":{"execution":{"iopub.status.busy":"2023-01-01T20:38:50.519645Z","iopub.execute_input":"2023-01-01T20:38:50.519992Z","iopub.status.idle":"2023-01-01T20:38:50.527813Z","shell.execute_reply.started":"2023-01-01T20:38:50.519955Z","shell.execute_reply":"2023-01-01T20:38:50.526929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(Dense_model.predict(np.expand_dims(x_test[0],0)))\n# print(Dense_model.predict(np.expand_dims(x_test[1],0)))\n# print(Dense_model.predict(np.expand_dims(x_test[2],0)))\n# print(Dense_model.predict(np.expand_dims(x_test[3],0)))","metadata":{"execution":{"iopub.status.busy":"2023-01-01T20:38:50.529494Z","iopub.execute_input":"2023-01-01T20:38:50.529763Z","iopub.status.idle":"2023-01-01T20:38:50.537335Z","shell.execute_reply.started":"2023-01-01T20:38:50.529738Z","shell.execute_reply":"2023-01-01T20:38:50.536439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## My Model","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import BatchNormalization\nfrom tensorflow.keras.layers import SeparableConv2D\nfrom tensorflow.keras.layers import MaxPooling2D\nfrom tensorflow.keras.layers import Conv2D\nfrom tensorflow.keras.layers import Activation\nfrom tensorflow.keras.layers import Flatten\nfrom tensorflow.keras.layers import Dropout\nfrom tensorflow.keras.layers import Dense\nimport tensorflow as tf\nfrom tensorflow import keras\n\n\nmodel = Sequential()\ninput_shape = (256, 256, 1)\n\nmodel.add(Conv2D(filters=32, kernel_size = (3,3), strides =1, padding = 'same', activation = 'relu', input_shape = input_shape))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Conv2D(filters=64, kernel_size = (3,3), strides =1, padding = 'same', activation = 'relu'))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Conv2D(filters=128, kernel_size = (3,3), strides =1, padding = 'same', activation = 'relu'))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Conv2D(filters=256, kernel_size = (3,3), strides =1, padding = 'same', activation = 'relu'))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\n\nmodel.add(Dropout(0.13))\nmodel.add(Flatten())\nmodel.add(Dense(256, activation = 'relu'))\nmodel.add(Dropout(0.13))\nmodel.add(Dense(100, activation = 'relu'))\nmodel.add(Dense(50, activation = 'relu'))\nmodel.add(Dense(1 , activation=\"sigmoid\"))\nmodel.summary()\n# chanDim =-1\n\n# # CONV => RELU => POOL\n# model = Sequential()\n# model.add(SeparableConv2D(32, (3, 3), padding=\"same\",input_shape=(256,256,1)))\n# model.add(Activation(\"relu\"))\n# model.add(BatchNormalization(axis=chanDim))\n# model.add(MaxPooling2D(pool_size=(2, 2)))\n# model.add(Dropout(0.25))\n# # (CONV => RELU => POOL) * 2\n# model.add(SeparableConv2D(64, (3, 3), padding=\"same\"))\n# model.add(Activation(\"relu\"))\n# model.add(BatchNormalization(axis=chanDim))\n# model.add(SeparableConv2D(64, (3, 3), padding=\"same\"))\n# model.add(Activation(\"relu\"))\n# model.add(BatchNormalization(axis=chanDim))\n# model.add(MaxPooling2D(pool_size=(2, 2)))\n# model.add(Dropout(0.25))\n# # (CONV => RELU => POOL) * 3\n# model.add(SeparableConv2D(128, (3, 3), padding=\"same\"))\n# model.add(Activation(\"relu\"))\n# model.add(BatchNormalization(axis=chanDim))\n# model.add(SeparableConv2D(128, (3, 3), padding=\"same\"))\n# model.add(Activation(\"relu\"))\n# model.add(BatchNormalization(axis=chanDim))\n# model.add(SeparableConv2D(128, (3, 3), padding=\"same\"))\n# model.add(Activation(\"relu\"))\n# model.add(BatchNormalization(axis=chanDim))\n# model.add(MaxPooling2D(pool_size=(2, 2)))\n# model.add(Dropout(0.25))\n\n# model.add(Flatten())\n# model.add(Dense(256))\n# model.add(Activation(\"relu\"))\n# model.add(BatchNormalization())\n# model.add(Dropout(0.5))\n# # softmax classifier\n# model.add(Dense(1))\n        \n# model.add(Activation('sigmoid'))\n# model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-01-01T20:38:50.538853Z","iopub.execute_input":"2023-01-01T20:38:50.539283Z","iopub.status.idle":"2023-01-01T20:38:53.232484Z","shell.execute_reply.started":"2023-01-01T20:38:50.539249Z","shell.execute_reply":"2023-01-01T20:38:53.230841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras import metrics\ndef tf_pfbeta(from_logits=True, beta=1.0, epsilon=1e-07):\n    \n    def pfbeta(y_true, y_pred):\n        y_pred = tf.cond(\n            tf.cast(from_logits, dtype=tf.bool),\n            lambda: tf.nn.sigmoid(y_pred),\n            lambda: y_pred,\n        )\n        y_true = tf.reshape(y_true, [-1])\n        y_pred = tf.reshape(y_pred, [-1])\n\n        ctp = tf.reduce_sum(y_true * y_pred, axis=-1)\n        cfp = tf.reduce_sum(y_pred, axis=-1) - ctp\n\n        c_precision = ctp / (ctp + cfp)\n        c_recall = ctp / tf.reduce_sum(y_true)\n        \n        def compute_fractions():\n            numerator = c_precision * c_recall\n            denominator = beta**2 * c_precision + c_recall\n            return (1 + beta**2) * tf.math.divide_no_nan(numerator, denominator)\n        \n        return tf.cond(\n            tf.logical_and(\n                tf.greater(c_precision, 0.), tf.greater(c_recall, 0.)\n            ),\n            compute_fractions,\n            lambda: tf.constant(0, dtype=tf.float32)\n        )\n    \n    return pfbeta\n\ndef tf_auc(from_logits=True):\n    auc_fn = metrics.AUC()\n    \n    def auc(y_true, y_pred):\n        y_pred = tf.cond(\n            tf.cast(from_logits, dtype=tf.bool),\n            lambda: tf.nn.sigmoid(y_pred),\n            lambda: y_pred,\n        )\n        return auc_fn(y_true, y_pred)\n    \n    return auc\n\n\ndef weighted_binary_loss(weight, from_logits=True, reduction=\"mean\"):\n    def inverse_sigmoid(sigmoidal):\n        return - tf.math.log(1. / sigmoidal - 1.)\n\n    def weighted_loss(labels, predictions):\n        predictions = tf.convert_to_tensor(predictions)\n        labels = tf.cast(labels, predictions.dtype)\n        num_samples = tf.cast(tf.shape(labels)[-1], dtype=labels.dtype)\n\n        logits = tf.cond(\n            tf.cast(from_logits, dtype=tf.bool),\n            lambda: predictions,\n            lambda: inverse_sigmoid(sigmoidal=predictions),\n        )\n        loss = tf.nn.weighted_cross_entropy_with_logits(\n            tf.cast(labels, dtype=tf.float32), logits, pos_weight=weight\n        )\n        \n        if reduction.lower() == \"mean\":\n            return tf.reduce_mean(loss)\n        elif reduction.lower() == \"sum\":\n            return tf.reduce_sum(loss) / num_samples\n        elif reduction.lower() == \"none\":\n            return loss\n        else:\n            raise ValueError(\n                'Reduction type is should be `mean` or `sum` or `none`. ',\n                f'But, received {reduction}'\n            )\n    return weighted_loss\n\ndef binary_focal_loss(\n    alpha=0.25, \n    gamma=2.0, \n    label_smoothing=0, \n    from_logits=False,\n    apply_class_balancing=False,\n    apply_positive_weight=1,\n    reduction=\"mean\"\n):\n    '''\n    alpha: A weight balancing factor for class 1, default is 0.25. \n        The weight for class 0 is 1.0 - alpha.\n    \n    gamma: A focusing parameter used to compute the focal factor, default is 2.0\n    \n    apply_class_balancing: A bool, whether to apply weight balancing on the binary \n        classes 0 and 1.\n    '''\n    \n    def smooth_labels(labels):\n        return labels * (1.0 - label_smoothing) + 0.5 * label_smoothing\n    \n    def compute_loss(labels, logits):\n        logits = tf.convert_to_tensor(logits)\n        labels = tf.cast(labels, logits.dtype)\n        labels = tf.cond(\n            tf.cast(label_smoothing, dtype=tf.bool),\n            lambda: smooth_labels(labels),\n            lambda: labels,\n        )\n        num_samples = tf.cast(tf.shape(labels)[-1], dtype=labels.dtype)\n        cross_entropy = weighted_binary_loss(\n            apply_positive_weight, from_logits, reduction='none'\n        )(labels, logits)\n        \n        sigmoidal = tf.cond(\n            tf.cast(from_logits, dtype=tf.bool),\n            lambda: tf.nn.sigmoid(logits),\n            lambda: logits,\n        )\n        pt = labels * sigmoidal + (1.0 - labels) * (1.0 - sigmoidal)\n        focal_factor = tf.pow(1.0 - pt, gamma)\n        focal_bce =  focal_factor * cross_entropy\n        \n        if apply_class_balancing:\n            weight = labels * alpha + (1 - labels) * (1 - alpha)\n            focal_bce = weight * focal_bce\n\n        if reduction == 'mean':\n            return tf.reduce_mean(focal_bce)\n        elif reduction == 'sum':\n            return tf.reduce_sum(focal_bce) / num_samples\n        else:\n            raise ValueError(\n                'Reduction type should be `mean` or `sum` ',\n                f'But, received {reduction}'\n            )\n    return compute_loss\n","metadata":{"execution":{"iopub.status.busy":"2023-01-01T20:38:53.234028Z","iopub.execute_input":"2023-01-01T20:38:53.234433Z","iopub.status.idle":"2023-01-01T20:38:53.256617Z","shell.execute_reply.started":"2023-01-01T20:38:53.234384Z","shell.execute_reply":"2023-01-01T20:38:53.25547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(loss='binary_crossentropy',\n              \n            optimizer=Adam(lr=0.001),metrics = [\n            tf_pfbeta(beta=1.0, from_logits=True),\n            tf_auc(from_logits=True)\n        ])","metadata":{"execution":{"iopub.status.busy":"2023-01-01T20:38:53.258337Z","iopub.execute_input":"2023-01-01T20:38:53.258811Z","iopub.status.idle":"2023-01-01T20:38:53.283341Z","shell.execute_reply.started":"2023-01-01T20:38:53.258776Z","shell.execute_reply":"2023-01-01T20:38:53.282317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nx_train, x_test, y_train, y_test = train_test_split(X_train, Y_train, test_size=0.20, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-01-01T20:38:53.284839Z","iopub.execute_input":"2023-01-01T20:38:53.285219Z","iopub.status.idle":"2023-01-01T20:38:53.727016Z","shell.execute_reply.started":"2023-01-01T20:38:53.285184Z","shell.execute_reply":"2023-01-01T20:38:53.725958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.utils import compute_class_weight\n# train_classes = Y_train\n# class_weights = compute_class_weight(\n#                                         class_weight = \"balanced\",\n#                                         classes = np.unique(train_classes),\n#                                         y = train_classes                                                    \n#                                     )\n# class_weights = dict(zip(np.unique(train_classes), class_weights))\n# class_weights","metadata":{"execution":{"iopub.status.busy":"2023-01-01T20:38:53.728581Z","iopub.execute_input":"2023-01-01T20:38:53.729046Z","iopub.status.idle":"2023-01-01T20:38:53.734004Z","shell.execute_reply.started":"2023-01-01T20:38:53.729005Z","shell.execute_reply":"2023-01-01T20:38:53.733083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(y_train.dtype)","metadata":{"execution":{"iopub.status.busy":"2023-01-01T20:38:53.735377Z","iopub.execute_input":"2023-01-01T20:38:53.73631Z","iopub.status.idle":"2023-01-01T20:38:53.747523Z","shell.execute_reply.started":"2023-01-01T20:38:53.736274Z","shell.execute_reply":"2023-01-01T20:38:53.746393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(x_train, y_train, validation_data=(x_test,y_test), epochs=20, batch_size=64)","metadata":{"execution":{"iopub.status.busy":"2023-01-01T20:38:53.750662Z","iopub.execute_input":"2023-01-01T20:38:53.750932Z","iopub.status.idle":"2023-01-01T20:39:49.959154Z","shell.execute_reply.started":"2023-01-01T20:38:53.750908Z","shell.execute_reply":"2023-01-01T20:39:49.95821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.evaluate(x_test,y_test)","metadata":{"execution":{"iopub.status.busy":"2023-01-01T20:39:49.965644Z","iopub.execute_input":"2023-01-01T20:39:49.965932Z","iopub.status.idle":"2023-01-01T20:39:50.758012Z","shell.execute_reply.started":"2023-01-01T20:39:49.965906Z","shell.execute_reply":"2023-01-01T20:39:50.757113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"index = 50\nprint(model.predict(np.expand_dims(x_test[index],0)))\nplt.imshow(x_train[index], cmap='gray')\ny_test[index]","metadata":{"execution":{"iopub.status.busy":"2023-01-01T20:39:50.763243Z","iopub.execute_input":"2023-01-01T20:39:50.76353Z","iopub.status.idle":"2023-01-01T20:39:51.188917Z","shell.execute_reply.started":"2023-01-01T20:39:50.763496Z","shell.execute_reply":"2023-01-01T20:39:51.188001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Test","metadata":{}},{"cell_type":"code","source":"import pydicom\n#TAKEN FROM : https://www.kaggle.com/code/vslaykovsky/infer-effnetv2-aux-targets-weighted-loss-thres/notebook\nfrom joblib import Parallel, delayed\nimport os \n\ndef process(f, size=256, save_folder=\"\", extension=\"png\"):\n    patient = f.split('/')[-2]\n    image = f.split('/')[-1][:-4]\n\n    dicom = pydicom.dcmread(f)\n    img = dicom.pixel_array\n\n    img = (img - img.min()) / (img.max() - img.min())\n\n    if dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        img = 1 - img\n\n    img = cv2.resize(img, (size, size))\n\n    cv2.imwrite(save_folder + f\"/{patient}_{image}.{extension}\", (img * 255).astype(np.uint8))\n    \n\ntest_images = glob(\"/kaggle/input/rsna-breast-cancer-detection/test_images/*/*.dcm\")\n\n!mkdir -p test\n_ = Parallel(n_jobs=4)(\n    delayed(process)(uid, save_folder='test')\n    for uid in tqdm(test_images)\n)\n\n\nroot_path = '/kaggle/working/test'\n\ndef getpath(root_path, parent_folder, row_df):\n    path = os.path.join(root_path, str(row_df['patient_id']) + '_' +str(row_df['image_id']) + '.png')\n    return path","metadata":{"execution":{"iopub.status.busy":"2023-01-01T20:39:51.190395Z","iopub.execute_input":"2023-01-01T20:39:51.193505Z","iopub.status.idle":"2023-01-01T20:39:57.967686Z","shell.execute_reply.started":"2023-01-01T20:39:51.193463Z","shell.execute_reply":"2023-01-01T20:39:57.966472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"samp_sub = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/sample_submission.csv')\ntest_csv = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')\npareF = 'test_images' \n\ntest_csv['Filepath']  = test_csv.apply(lambda row : getpath(root_path, pareF, row), axis=1)\n\ntest_csv.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-01T20:39:57.972788Z","iopub.execute_input":"2023-01-01T20:39:57.975655Z","iopub.status.idle":"2023-01-01T20:39:58.006816Z","shell.execute_reply.started":"2023-01-01T20:39:57.97561Z","shell.execute_reply":"2023-01-01T20:39:58.005947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = []\nfor i in glob('/kaggle/working/test/*'):\n            output = cv2.imread(i,0)\n            test.append(output)\ntest = np.array(test)\ntest = np.expand_dims(test,-1)\ntest.shape","metadata":{"execution":{"iopub.status.busy":"2023-01-01T20:39:58.009857Z","iopub.execute_input":"2023-01-01T20:39:58.010145Z","iopub.status.idle":"2023-01-01T20:39:58.024139Z","shell.execute_reply.started":"2023-01-01T20:39:58.010117Z","shell.execute_reply":"2023-01-01T20:39:58.023156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def preprocess(file_path):\n#     ds = pydicom.dcmread(i)\n#     pixel_array_numpy = ds.pixel_array\n#     pixel_array_numpy = cv2.resize(pixel_array_numpy, dsize=(256, 256))\n#     pixel_array_numpy = np.expand_dims(pixel_array_numpy,-1)\n#     return pixel_array_numpy\n\n\n# test = []\n# for i in glob('/kaggle/input/rsna-breast-cancer-detection/test_images/10008/*'):\n#             output = preprocess(i)\n#             test.append(output)\n# test = np.array(test)\n# test.shape","metadata":{"execution":{"iopub.status.busy":"2023-01-01T20:39:58.025402Z","iopub.execute_input":"2023-01-01T20:39:58.025833Z","iopub.status.idle":"2023-01-01T20:39:58.030319Z","shell.execute_reply.started":"2023-01-01T20:39:58.025797Z","shell.execute_reply":"2023-01-01T20:39:58.029362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test = []\n# for i in glob('/kaggle/input/rsna-breast-cancer-detection/test_images/10008/*'):\n#             output = preprocessFast(i, size = 256, dicom_process = False)\n#             test.append(output)\n# test = np.array(test)\n# test.shape","metadata":{"execution":{"iopub.status.busy":"2023-01-01T20:39:58.031715Z","iopub.execute_input":"2023-01-01T20:39:58.032318Z","iopub.status.idle":"2023-01-01T20:39:58.040057Z","shell.execute_reply.started":"2023-01-01T20:39:58.032276Z","shell.execute_reply":"2023-01-01T20:39:58.039307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = model.predict(test)\npreds","metadata":{"execution":{"iopub.status.busy":"2023-01-01T20:39:58.041303Z","iopub.execute_input":"2023-01-01T20:39:58.042128Z","iopub.status.idle":"2023-01-01T20:39:58.27191Z","shell.execute_reply.started":"2023-01-01T20:39:58.042076Z","shell.execute_reply":"2023-01-01T20:39:58.270225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/test.csv\")\ndf['cancer'] = 0\n\nTHRESHOLD = 0.50\n\npreds = (preds > THRESHOLD).astype(int)\ndf[\"cancer\"] = preds","metadata":{"execution":{"iopub.status.busy":"2023-01-01T20:39:58.27331Z","iopub.execute_input":"2023-01-01T20:39:58.273734Z","iopub.status.idle":"2023-01-01T20:39:58.284666Z","shell.execute_reply.started":"2023-01-01T20:39:58.273697Z","shell.execute_reply":"2023-01-01T20:39:58.283717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['prediction_id'] = df['patient_id'].astype(str) + \"_\" + df['laterality']\n\nsub = df[['prediction_id', 'cancer']].groupby(\"prediction_id\").mean().reset_index()\n\nsub.to_csv('/kaggle/working/submission.csv', index=False)\n\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-01T20:39:58.286341Z","iopub.execute_input":"2023-01-01T20:39:58.28674Z","iopub.status.idle":"2023-01-01T20:39:58.307653Z","shell.execute_reply.started":"2023-01-01T20:39:58.286705Z","shell.execute_reply":"2023-01-01T20:39:58.306648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot confusion matrices for benchmark and transfer learning models\nfrom sklearn.metrics import confusion_matrix\nimport seaborn as sns\n\nplt.figure(figsize=(15, 5))\n\npreds = model.predict(x_test)\npreds = (preds >= 0.5).astype(np.int32)\n\ncm = confusion_matrix(y_test, preds)\ndf_cm = pd.DataFrame(cm, index=['no-cancer', 'cancer'], columns=['no-cancer', 'cancer'])\nplt.subplot(121)\nplt.title(\"Confusion matrix for benchmark model\\n\")\nsns.heatmap(df_cm, annot=True, fmt=\"d\", cmap=\"YlGnBu\")\nplt.ylabel(\"Predicted\")\nplt.xlabel(\"Actual\")","metadata":{"execution":{"iopub.status.busy":"2023-01-01T20:39:58.309307Z","iopub.execute_input":"2023-01-01T20:39:58.30966Z","iopub.status.idle":"2023-01-01T20:39:59.688854Z","shell.execute_reply.started":"2023-01-01T20:39:58.309626Z","shell.execute_reply":"2023-01-01T20:39:59.687945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"true_negative = 0\ntrue_positive = 0\nfor i,j in zip(preds,y_test):\n    if (i == 0 and j==0):\n        true_negative = true_negative+1\n    if (i == 1 and j==1):\n        true_positive = true_positive+1    \n        \nprint(true_negative)\nprint(true_positive)","metadata":{"execution":{"iopub.status.busy":"2023-01-01T20:39:59.691957Z","iopub.execute_input":"2023-01-01T20:39:59.69451Z","iopub.status.idle":"2023-01-01T20:39:59.708645Z","shell.execute_reply.started":"2023-01-01T20:39:59.69447Z","shell.execute_reply":"2023-01-01T20:39:59.707685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}