{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-02-12T09:14:13.661647Z","iopub.execute_input":"2023-02-12T09:14:13.66215Z","iopub.status.idle":"2023-02-12T09:14:13.667022Z","shell.execute_reply.started":"2023-02-12T09:14:13.662083Z","shell.execute_reply":"2023-02-12T09:14:13.666104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Let's download the data ","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-12T09:14:13.691071Z","iopub.execute_input":"2023-02-12T09:14:13.691436Z","iopub.status.idle":"2023-02-12T09:14:13.962617Z","shell.execute_reply.started":"2023-02-12T09:14:13.691399Z","shell.execute_reply":"2023-02-12T09:14:13.961567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Exploring the data","metadata":{}},{"cell_type":"code","source":"print(f'Length of train dataframe: {len(train_df)}\\n')\nprint(f'Number of NaN values:\\n{train_df.isna().sum()}\\n')","metadata":{"execution":{"iopub.status.busy":"2023-02-12T09:14:13.967792Z","iopub.execute_input":"2023-02-12T09:14:13.975477Z","iopub.status.idle":"2023-02-12T09:14:14.014595Z","shell.execute_reply.started":"2023-02-12T09:14:13.975439Z","shell.execute_reply":"2023-02-12T09:14:14.013584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> 37 missing value on Age, and more than  25000 value on BIRAFS and density ","metadata":{}},{"cell_type":"code","source":"#->For simple model fitting and future predictions I'm gonna use only patient scans.\n#let's consider one particular patient\npatient_id = train_df[train_df.cancer == 1].iloc[0].patient_id\n\none_patient_df = train_df[train_df.patient_id == patient_id]\none_patient_df","metadata":{"execution":{"iopub.status.busy":"2023-02-12T09:14:14.016177Z","iopub.execute_input":"2023-02-12T09:14:14.016796Z","iopub.status.idle":"2023-02-12T09:14:14.066207Z","shell.execute_reply.started":"2023-02-12T09:14:14.016753Z","shell.execute_reply":"2023-02-12T09:14:14.065183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install dicomsdl","metadata":{"execution":{"iopub.status.busy":"2023-02-12T09:14:14.067902Z","iopub.execute_input":"2023-02-12T09:14:14.06938Z","iopub.status.idle":"2023-02-12T09:14:27.775882Z","shell.execute_reply.started":"2023-02-12T09:14:14.069339Z","shell.execute_reply":"2023-02-12T09:14:27.77479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"images_dir = '/kaggle/input/rsna-breast-cancer-detection/{}_images/{}/{}.dcm'\ntrain = 'train'\ntest = 'test'\n\ntrain_df[\"image_path\"] =  '/kaggle/input/rsna-breast-cancer-detection/train_images/'+train_df[\"patient_id\"].astype(str) +\"/\"+ train_df [\"image_id\"].astype(str)+\".dcm\"","metadata":{"execution":{"iopub.status.busy":"2023-02-12T09:14:27.779565Z","iopub.execute_input":"2023-02-12T09:14:27.779886Z","iopub.status.idle":"2023-02-12T09:14:27.877431Z","shell.execute_reply.started":"2023-02-12T09:14:27.779855Z","shell.execute_reply":"2023-02-12T09:14:27.876473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport dicomsdl\nimport seaborn as sns\n\nn_rows = len(one_patient_df)\n\nplt.figure(figsize=(5 * n_rows, 5))\nfor i in range(n_rows):\n    row = one_patient_df.iloc[i]\n    \n    plt.subplot(1, n_rows, i + 1)\n    \n    img_arr = dicomsdl.open(images_dir.format(train, row.patient_id, row.image_id)).pixelData()\n    plt.imshow(img_arr, cmap = plt.cm.bone)\n    plt.text(200, 300, row['view'], fontsize = 14, bbox={'facecolor': 'white', 'pad' : 5})\n    plt.text(200, 700, row['cancer'], fontsize = 14, bbox={'facecolor': 'white', 'pad' : 5})","metadata":{"execution":{"iopub.status.busy":"2023-02-12T09:15:45.574403Z","iopub.execute_input":"2023-02-12T09:15:45.574764Z","iopub.status.idle":"2023-02-12T09:15:53.063613Z","shell.execute_reply.started":"2023-02-12T09:15:45.574732Z","shell.execute_reply":"2023-02-12T09:15:53.062592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Frequency of cancer and normal patients","metadata":{}},{"cell_type":"code","source":"# Let's visualize how many patient in the dataset have cancer and how many don't\ntemp = train_df.groupby('patient_id')['cancer'].max().to_frame()\ntemp.reset_index(inplace=True)\ntemp = temp['cancer'].value_counts().to_frame()\ntemp.reset_index(inplace=True)\ntemp.columns = ['cancer', 'patient_count']\nfig,ax = plt.subplots(figsize=(12, 8))\nax = sns.barplot(data=temp, x=temp.columns[0], y=temp.columns[1])\nax.bar_label(ax.containers[0])\nplt.title('Frequency of normal patients and breast cancer patients')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-12T09:15:53.065514Z","iopub.execute_input":"2023-02-12T09:15:53.06594Z","iopub.status.idle":"2023-02-12T09:15:53.27516Z","shell.execute_reply.started":"2023-02-12T09:15:53.065906Z","shell.execute_reply":"2023-02-12T09:15:53.274196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Among all the patients 486 patients were affected by cancer.","metadata":{}},{"cell_type":"markdown","source":"**Realtionship between age and cancer**","metadata":{}},{"cell_type":"code","source":"# Now let's identified cancer patients and relationship with age\ntemp = train_df.groupby('patient_id')[['age', 'cancer']].max()\nm = {\n    \"30-40\": [0, 0],\n    \"40-50\": [0, 0],\n    \"50-60\": [0, 0],\n    \"60-70\": [0, 0],\n    \"70-80\": [0, 0],\n    \"80-90\": [0, 0]\n}\nfor index, row in temp.iterrows():\n    age = row['age']\n    if age>=30 and age<40: \n        m['30-40'][0] += 1\n        if row['cancer']==1: m['30-40'][1] += 1\n    elif age>=40 and age<50: \n        m['40-50'][0] += 1\n        if row['cancer']==1: m['40-50'][1] += 1\n    elif age>=50 and age<60: \n        m['50-60'][0] += 1\n        if row['cancer']==1: m['50-60'][1] += 1\n    elif age>=60 and age<70: \n        m['60-70'][0] += 1\n        if row['cancer']==1: m['60-70'][1] += 1\n    elif age>=70 and age<80: \n        m['70-80'][0] += 1\n        if row['cancer']==1: m['70-80'][1] += 1\n    else: \n        m['80-90'][0] += 1\n        if row['cancer']==1: m['80-90'][1] += 1\ntemp = {}\nfor key, val in m.items(): temp[key] = val[1] / val[0]\nfig, ax = plt.subplots(figsize=(12,10))\nax = sns.barplot(x=list(temp.keys()), y=list(temp.values()))\nplt.title(\"Percentage of cancer patient of different age groups\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-12T09:15:53.278218Z","iopub.execute_input":"2023-02-12T09:15:53.278501Z","iopub.status.idle":"2023-02-12T09:15:54.061246Z","shell.execute_reply.started":"2023-02-12T09:15:53.278476Z","shell.execute_reply":"2023-02-12T09:15:54.059925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It is clear that, age is an important factor which influence that a patient can have cancer or not. With increasing age chances of breast cancer increases in woman.","metadata":{}},{"cell_type":"markdown","source":"**Correlation between all the variables**","metadata":{}},{"cell_type":"code","source":"temp = train_df.drop(columns=['patient_id', 'image_id', 'site_id', 'machine_id'])\nfig, ax = plt.subplots(figsize=(20, 12))\ndataplot = sns.heatmap(temp.corr(method='spearman'), cmap=\"YlGnBu\", annot=True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-12T09:15:54.064708Z","iopub.execute_input":"2023-02-12T09:15:54.065078Z","iopub.status.idle":"2023-02-12T09:15:54.635734Z","shell.execute_reply.started":"2023-02-12T09:15:54.065042Z","shell.execute_reply":"2023-02-12T09:15:54.634622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# pre-processing the dataset","metadata":{}},{"cell_type":"markdown","source":"**Split the data into train and test**","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX_train, X_val = train_test_split(train_df, test_size = 0.4, random_state = 42)\nlen(X_train), len(X_val)","metadata":{"execution":{"iopub.status.busy":"2023-02-12T09:15:54.637053Z","iopub.execute_input":"2023-02-12T09:15:54.638013Z","iopub.status.idle":"2023-02-12T09:15:54.752969Z","shell.execute_reply.started":"2023-02-12T09:15:54.637973Z","shell.execute_reply":"2023-02-12T09:15:54.751862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train","metadata":{"execution":{"iopub.status.busy":"2023-02-12T09:15:54.754406Z","iopub.execute_input":"2023-02-12T09:15:54.755075Z","iopub.status.idle":"2023-02-12T09:15:54.81326Z","shell.execute_reply.started":"2023-02-12T09:15:54.755028Z","shell.execute_reply":"2023-02-12T09:15:54.812025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Pre-processing","metadata":{}},{"cell_type":"code","source":"import tensorflow.keras as K\n\nfrom tensorflow.keras import Sequential\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.layers import (Conv2D, \n                                     MaxPooling2D, \n                                     BatchNormalization, \n                                     Dense, \n                                     Dropout,\n                                     GlobalMaxPooling2D)","metadata":{"execution":{"iopub.status.busy":"2023-02-12T09:15:54.818014Z","iopub.execute_input":"2023-02-12T09:15:54.821757Z","iopub.status.idle":"2023-02-12T09:15:59.588036Z","shell.execute_reply.started":"2023-02-12T09:15:54.821705Z","shell.execute_reply":"2023-02-12T09:15:59.587069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# callbacks\nfrom tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau\n\nearly_stop = EarlyStopping(patience = 5, restore_best_weights = True, verbose = 1) # val_loss\nreduce_lr = ReduceLROnPlateau(factor = 0.1, patience = 2, mode = 'min', verbose = 1) # val_loss ","metadata":{"execution":{"iopub.status.busy":"2023-02-12T09:21:47.82769Z","iopub.execute_input":"2023-02-12T09:21:47.828042Z","iopub.status.idle":"2023-02-12T09:21:47.833936Z","shell.execute_reply.started":"2023-02-12T09:21:47.828012Z","shell.execute_reply":"2023-02-12T09:21:47.832517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd \nimport numpy as np\nimport matplotlib.pyplot as plt\nimport os\nimport glob\nimport gc\nimport time \nimport shutil\nfrom sklearn.model_selection import train_test_split\nfrom kaggle_datasets import KaggleDatasets\n\nimport tensorflow as tf\nfrom tensorflow.keras.applications.resnet50 import ResNet50\nfrom tensorflow.python.client import device_lib\nfrom tensorflow.keras import Model, layers, Input, Sequential\nfrom sklearn.metrics import log_loss\n\nfrom joblib import Parallel, delayed\nimport cv2","metadata":{"execution":{"iopub.status.busy":"2023-02-12T09:21:47.83653Z","iopub.execute_input":"2023-02-12T09:21:47.837027Z","iopub.status.idle":"2023-02-12T09:21:47.845449Z","shell.execute_reply.started":"2023-02-12T09:21:47.836986Z","shell.execute_reply":"2023-02-12T09:21:47.844174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"x -> les image dialna \ny-> cancer or not \n\nsize of images 7dadnah tahowa \n\ncreena wa7d la list 3amra bles zeros\n from : row = self.df.iloc[pos]\n            patient_id = row.patient_id\n            img_id = row.image_id\n            \n            file_name = images_dir.format(self.mode_str, patient_id, img_id)\n            \n            img = dicomsdl.open(file_name)\n            img_arr = img.pixelData()\n            \nwe are reading the images\n","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\n\nfrom enum import Enum, auto\n\nclass Mode(Enum):\n    TRAIN = auto()\n    TEST = auto()\n    \nimg_height = 300\nimg_width = 250\nimg_shape = (img_height, img_width, 3)\n\nclass ImageDataGen(tf.keras.utils.Sequence):\n    \n    def __init__(self,\n                 df,\n                 batch_size,\n                 mode = Mode.TRAIN):\n\n        self.df = df\n        self.batch_size = batch_size\n        self.mode = mode\n        self.mode_str = train if mode == Mode.TRAIN else test\n        \n        self.len = len(df)\n        \n    def __getitem__(self, index):\n        \n        start, end = index * self.batch_size, (index + 1) * self.batch_size\n        \n        X = np.zeros((self.batch_size, ) + img_shape)\n        y = np.zeros((self.batch_size, 1))\n        \n        for i , pos in enumerate(range(start, end)):\n            if pos >= self.len: break\n                     \n            row = self.df.iloc[pos]\n            patient_id = row.patient_id\n            img_id = row.image_id\n            \n            file_name = images_dir.format(self.mode_str, patient_id, img_id)\n            \n            img = dicomsdl.open(file_name)\n            img_arr = img.pixelData()\n            \n            # standartize all scans ( Hyperparameter tuning )\n            img_arr = (img_arr - img_arr.min()) / (img_arr.max() - img_arr.min())\n            \n            if img.PhotometricInterpretation == \"RGB\":\n                img_arr = 1 - img_arr\n\n            img_arr = np.expand_dims(img_arr, axis = -1)\n            img_arr = tf.image.resize(img_arr, img_shape[:-1], method = 'nearest').numpy()\n                 \n            X[i,...] = img_arr\n                \n            \n            if self.mode == Mode.TRAIN:\n                y[i] = row.cancer\n                \n        return (X, y) if self.mode == Mode.TRAIN else X\n                \n    \n    def __len__(self):\n        return self.len // self.batch_size + bool(self.len % self.batch_size)\n","metadata":{"execution":{"iopub.status.busy":"2023-02-12T09:21:47.8482Z","iopub.execute_input":"2023-02-12T09:21:47.848616Z","iopub.status.idle":"2023-02-12T09:21:47.864578Z","shell.execute_reply.started":"2023-02-12T09:21:47.848581Z","shell.execute_reply":"2023-02-12T09:21:47.863547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\ntrain_df= train_df.sample(1000)\nX_train, X_val = train_test_split(train_df, test_size = 0.4, random_state = 42)\nlen(X_train), len(X_val)\ntrain_gen = ImageDataGen(X_train, 50)\nval_gen = ImageDataGen(X_val, 50)","metadata":{"execution":{"iopub.status.busy":"2023-02-12T09:21:47.866015Z","iopub.execute_input":"2023-02-12T09:21:47.866995Z","iopub.status.idle":"2023-02-12T09:21:47.879004Z","shell.execute_reply.started":"2023-02-12T09:21:47.866959Z","shell.execute_reply":"2023-02-12T09:21:47.877919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"    Rescaling = layers.Rescaling(1./255,name = \"Rescaling0_1\")\n    RandomFlip = layers.RandomFlip('horizontal')\n    RandomRotation = layers.RandomRotation(0.2) ","metadata":{"execution":{"iopub.status.busy":"2023-02-12T09:21:47.882486Z","iopub.execute_input":"2023-02-12T09:21:47.883854Z","iopub.status.idle":"2023-02-12T09:21:47.897947Z","shell.execute_reply.started":"2023-02-12T09:21:47.883815Z","shell.execute_reply":"2023-02-12T09:21:47.896762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_augmentation  = Sequential([\n    #layers.InputLayer(input_shape=(RESIZE,RESIZE,3)), \n    RandomFlip,\n    RandomRotation])","metadata":{"execution":{"iopub.status.busy":"2023-02-12T09:21:47.900257Z","iopub.execute_input":"2023-02-12T09:21:47.90086Z","iopub.status.idle":"2023-02-12T09:21:47.908966Z","shell.execute_reply.started":"2023-02-12T09:21:47.900825Z","shell.execute_reply":"2023-02-12T09:21:47.907853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Model**","metadata":{}},{"cell_type":"code","source":"    ResNet = ResNet50(weights='imagenet',\n                  include_top=False,\n                  input_shape=img_shape,\n                  pooling = 'avg')\n    ResNet.trainable = False\n    \n    model = Sequential(\n        [layers.InputLayer(input_shape=img_shape),         \n         layers.experimental.preprocessing.Rescaling(1./255,name = \"Rescaling0_1\"),\n         ResNet,\n         layers.Flatten(),\n         layers.Dense(32, activation = \"relu\"),\n            layers.Dense(1,activation =\"sigmoid\",dtype='float32')\n        ]\n    )\n\nrecall_thresholds = [0.4, 0.5, 0.6, 0.8]\nmodel.compile(optimizer=\"adam\",loss= tf.keras.losses.BinaryCrossentropy(), \n              metrics = [tf.keras.metrics.BinaryAccuracy(threshold = 0.5), \n                         tf.keras.metrics.Recall(thresholds = recall_thresholds),\n                         tf.keras.metrics.Precision(thresholds = recall_thresholds)])","metadata":{"execution":{"iopub.status.busy":"2023-02-12T09:21:47.912452Z","iopub.execute_input":"2023-02-12T09:21:47.912717Z","iopub.status.idle":"2023-02-12T09:21:49.628011Z","shell.execute_reply.started":"2023-02-12T09:21:47.912692Z","shell.execute_reply":"2023-02-12T09:21:49.626981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history =  model.fit(train_gen,\n                    validation_data = val_gen,\n                    epochs = 8,\n                    verbose = 1,\n                    workers = 8,\n                    callbacks = [reduce_lr])","metadata":{"execution":{"iopub.status.busy":"2023-02-12T09:21:49.63072Z","iopub.execute_input":"2023-02-12T09:21:49.631011Z","iopub.status.idle":"2023-02-12T11:40:56.790942Z","shell.execute_reply.started":"2023-02-12T09:21:49.630985Z","shell.execute_reply":"2023-02-12T11:40:56.780212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"loss = history.history['loss']\nval_loss = history.history['val_loss']\n\nacc = history.history['binary_accuracy']\nval_acc = history.history['val_binary_accuracy']\n\n\nepochs = range(1, len(loss) + 1)\n\nplt.figure(figsize=(16, 5))\n#accuracy\nplt.subplot(1,2,1)\nplt.plot(epochs, acc, 'b', label = 'Training accuracy')\nplt.plot(epochs, val_acc, 'r', label = 'Validation accuracy')\nplt.legend()\n\n#loss\nplt.subplot(1,2,2)\nplt.plot(epochs, loss, 'b', label = 'Trainig loss')\nplt.plot(epochs, val_loss, 'r', label = 'Validation loss')\nplt.legend()\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-12T12:05:36.649053Z","iopub.execute_input":"2023-02-12T12:05:36.650062Z","iopub.status.idle":"2023-02-12T12:05:36.999875Z","shell.execute_reply.started":"2023-02-12T12:05:36.650017Z","shell.execute_reply":"2023-02-12T12:05:36.998892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Test","metadata":{}},{"cell_type":"code","source":"test_df = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-12T11:43:02.541745Z","iopub.execute_input":"2023-02-12T11:43:04.372532Z","iopub.status.idle":"2023-02-12T11:43:10.796251Z","shell.execute_reply.started":"2023-02-12T11:43:04.372437Z","shell.execute_reply":"2023-02-12T11:43:10.532323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_gen = ImageDataGen(test_df, 16, mode= Mode.TEST)\n","metadata":{"execution":{"iopub.status.busy":"2023-02-12T11:43:10.797779Z","iopub.execute_input":"2023-02-12T11:43:11.000258Z","iopub.status.idle":"2023-02-12T11:43:15.176794Z","shell.execute_reply.started":"2023-02-12T11:43:11.000183Z","shell.execute_reply":"2023-02-12T11:43:15.123947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = model.predict(test_gen)\n","metadata":{"execution":{"iopub.status.busy":"2023-02-12T11:43:15.178684Z","iopub.execute_input":"2023-02-12T11:43:15.179077Z","iopub.status.idle":"2023-02-12T11:45:39.007964Z","shell.execute_reply.started":"2023-02-12T11:43:15.17904Z","shell.execute_reply":"2023-02-12T11:45:39.006892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred","metadata":{"execution":{"iopub.status.busy":"2023-02-12T11:45:39.009745Z","iopub.execute_input":"2023-02-12T11:45:39.011394Z","iopub.status.idle":"2023-02-12T11:45:39.019941Z","shell.execute_reply.started":"2023-02-12T11:45:39.011354Z","shell.execute_reply":"2023-02-12T11:45:39.018734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['cancer'] = pred[:len(test_df)]\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-12T11:45:39.02206Z","iopub.execute_input":"2023-02-12T11:45:39.022527Z","iopub.status.idle":"2023-02-12T11:45:39.041166Z","shell.execute_reply.started":"2023-02-12T11:45:39.022488Z","shell.execute_reply":"2023-02-12T11:45:39.040238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = test_df.groupby('prediction_id')['cancer'].max().to_frame().reset_index()\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-12T11:45:39.043072Z","iopub.execute_input":"2023-02-12T11:45:39.043836Z","iopub.status.idle":"2023-02-12T11:45:39.069835Z","shell.execute_reply.started":"2023-02-12T11:45:39.043795Z","shell.execute_reply":"2023-02-12T11:45:39.068711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv(\"submission.csv\", index = False)","metadata":{"execution":{"iopub.status.busy":"2023-02-12T11:45:39.074796Z","iopub.execute_input":"2023-02-12T11:45:39.076599Z","iopub.status.idle":"2023-02-12T11:45:39.087389Z","shell.execute_reply.started":"2023-02-12T11:45:39.076561Z","shell.execute_reply":"2023-02-12T11:45:39.086095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Evaluation the model**","metadata":{}},{"cell_type":"code","source":"# Extract y_test and X_test from val_gen\ny_test = []\nX_test = []\nfor i in range(len(val_gen)):\n    X, y = val_gen[i]\n    y_test.append(y)\n    X_test.append(X)\n    \ny_test = np.concatenate(y_test)\nX_test = np.concatenate(X_test)","metadata":{"execution":{"iopub.status.busy":"2023-02-12T11:45:39.08866Z","iopub.execute_input":"2023-02-12T11:45:39.090354Z","iopub.status.idle":"2023-02-12T11:49:37.676391Z","shell.execute_reply.started":"2023-02-12T11:45:39.090315Z","shell.execute_reply":"2023-02-12T11:49:37.675143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport tensorflow as tf\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.metrics import roc_auc_score\n\n# Get predicted probabilities for X_test\ny_pred = model.predict(X_test)\n\n# Convert predicted probabilities to binary predictions (0 or 1)\ny_pred_bin = (y_pred > 0.5).astype(int)\n\n# Calculate confusion matrix\nfrom sklearn.metrics import confusion_matrix\ncm = confusion_matrix(y_test, y_pred_bin)\nprint(cm)","metadata":{"execution":{"iopub.status.busy":"2023-02-12T11:49:37.677708Z","iopub.execute_input":"2023-02-12T11:49:37.678574Z","iopub.status.idle":"2023-02-12T11:49:42.491123Z","shell.execute_reply.started":"2023-02-12T11:49:37.678535Z","shell.execute_reply":"2023-02-12T11:49:42.489862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import f1_score, roc_auc_score\n\ny_pred = model.predict(X_test)\nf1 = f1_score(y_test, y_pred.round(), average='binary')\nroc_auc = roc_auc_score(y_test, y_pred)\n\nprint(\"F1 score: \", f1)\nprint(\"AUC-ROC: \", roc_auc)\n\n# Calculate the mean and standard deviation of the errors\nerror = y_pred - y_test\nerror_mean = np.mean(error)\nerror_std = np.std(error)\nprint(\"Error mean:\", error_mean)\nprint(\"Error standard deviation:\", error_std)\n\n# Plot the error distribution\nplt.hist(error, bins=20)\nplt.xlabel(\"Error\")\nplt.ylabel(\"Frequency\")\n\n# Calculate the confidence interval\nalpha = 0.95\nz = 1.96  # Z-score for 95% confidence interval\nerror_lower = error_mean - z * error_std / np.sqrt(len(error))\nerror_upper = error_mean + z * error_std / np.sqrt(len(error))\nprint(\"Confidence interval: [\", error_lower, \",\", error_upper, \"]\")","metadata":{"execution":{"iopub.status.busy":"2023-02-12T11:49:42.492621Z","iopub.execute_input":"2023-02-12T11:49:42.494603Z","iopub.status.idle":"2023-02-12T11:49:45.656705Z","shell.execute_reply.started":"2023-02-12T11:49:42.494565Z","shell.execute_reply":"2023-02-12T11:49:45.65579Z"},"trusted":true},"execution_count":null,"outputs":[]}]}