{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"⚠️ **Here I'm going to use the model that was fit in this notebook [Breast Cancer Detection: tf, CNN (train)](https://www.kaggle.com/code/maryiaznak/breast-cancer-detection-tf-cnn-train/notebook)**","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)","metadata":{"execution":{"iopub.status.busy":"2023-02-06T19:57:36.219326Z","iopub.execute_input":"2023-02-06T19:57:36.220489Z","iopub.status.idle":"2023-02-06T19:57:36.228334Z","shell.execute_reply.started":"2023-02-06T19:57:36.220402Z","shell.execute_reply":"2023-02-06T19:57:36.226196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1. DataGenerator","metadata":{}},{"cell_type":"code","source":"%%capture\n\n!pip install /kaggle/input/rsnamodules/dicomsdl-0.109.1-cp37-cp37m-manylinux_2_12_x86_64.manylinux2010_x86_64.whl \n\ntry:\n    import pylibjpeg\nexcept:\n    !pip install /kaggle/input/rsna-2022-whl/{pylibjpeg-1.4.0-py3-none-any.whl,python_gdcm-3.0.15-cp37-cp37m-manylinux_2_17_x86_64.manylinux2014_x86_64.whl}","metadata":{"execution":{"iopub.status.busy":"2023-01-05T18:43:23.718257Z","iopub.execute_input":"2023-01-05T18:43:23.718725Z","iopub.status.idle":"2023-01-05T18:44:28.588414Z","shell.execute_reply.started":"2023-01-05T18:43:23.718694Z","shell.execute_reply":"2023-01-05T18:44:28.587024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"images_dir = '/kaggle/input/rsna-breast-cancer-detection/test_images/{}/{}.dcm'","metadata":{"execution":{"iopub.status.busy":"2023-02-06T19:57:36.230102Z","iopub.execute_input":"2023-02-06T19:57:36.230592Z","iopub.status.idle":"2023-02-06T19:57:36.24267Z","shell.execute_reply.started":"2023-02-06T19:57:36.23055Z","shell.execute_reply":"2023-02-06T19:57:36.241275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nimport dicomsdl","metadata":{"execution":{"iopub.status.busy":"2023-02-06T19:57:36.244718Z","iopub.execute_input":"2023-02-06T19:57:36.245153Z","iopub.status.idle":"2023-02-06T19:57:36.256508Z","shell.execute_reply.started":"2023-02-06T19:57:36.245097Z","shell.execute_reply":"2023-02-06T19:57:36.255369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_height = 200\nimg_width = 150\nimg_shape = (img_height, img_width, 1)\n\nclass ImageDataGen(tf.keras.utils.Sequence):\n    \n    def __init__(self,\n                 df,\n                 batch_size):\n\n        self.df = df\n        self.batch_size = batch_size\n        self.len = len(df)\n        \n        \n    def __getitem__(self, index):\n        \n        start, end = index * self.batch_size, (index + 1) * self.batch_size\n        \n        X = np.zeros((self.batch_size, ) + img_shape)\n        for i , pos in enumerate(range(start, end)):\n            if pos >= self.len: break\n                     \n            row = self.df.iloc[pos]\n            patient_id = row.patient_id\n            img_id = row.image_id\n            \n            file_name = images_dir.format(patient_id, img_id)\n            img_arr = self.__get_img(file_name)\n                 \n            X[i,...] = img_arr  \n                \n        return X\n                \n    \n    def __len__(self):\n        return self.len // self.batch_size + bool(self.len % self.batch_size)\n    \n    \n    def __get_img(self, file_name):\n        \n        img = dicomsdl.open(file_name)\n        img_arr = img.pixelData()\n            \n        # standartize all scans\n        img_arr = (img_arr - img_arr.min()) / (img_arr.max() - img_arr.min())\n        if img.PhotometricInterpretation == \"MONOCHROME1\":\n            img_arr = 1 - img_arr\n            \n        #crop image\n        img_arr = img_arr[:, ~np.all(img_arr == 0, axis = 0)]\n        img_arr = img_arr[~np.all(img_arr == 0, axis = 1), :]\n        \n        # resize image\n        img_arr = np.expand_dims(img_arr, axis = -1)\n        img_arr = tf.image.resize(img_arr, img_shape[:-1], method = 'nearest').numpy()\n        \n        return img_arr","metadata":{"execution":{"iopub.status.busy":"2023-02-06T19:57:39.215892Z","iopub.execute_input":"2023-02-06T19:57:39.216429Z","iopub.status.idle":"2023-02-06T19:57:40.640445Z","shell.execute_reply.started":"2023-02-06T19:57:39.216376Z","shell.execute_reply":"2023-02-06T19:57:40.639006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2. Model","metadata":{}},{"cell_type":"code","source":"model = tf.keras.models.load_model('/kaggle/input/pretrained-model/model.h5')\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-01-05T18:44:35.470084Z","iopub.execute_input":"2023-01-05T18:44:35.470752Z","iopub.status.idle":"2023-01-05T18:44:36.691538Z","shell.execute_reply.started":"2023-01-05T18:44:35.470706Z","shell.execute_reply":"2023-01-05T18:44:36.690227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3. Test Dataset","metadata":{}},{"cell_type":"code","source":"test_df = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-06T19:57:51.567219Z","iopub.execute_input":"2023-02-06T19:57:51.567707Z","iopub.status.idle":"2023-02-06T19:57:51.622392Z","shell.execute_reply.started":"2023-02-06T19:57:51.567668Z","shell.execute_reply":"2023-02-06T19:57:51.621367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_gen = ImageDataGen(test_df, 50)","metadata":{"execution":{"iopub.status.busy":"2023-02-06T19:57:52.50912Z","iopub.execute_input":"2023-02-06T19:57:52.510025Z","iopub.status.idle":"2023-02-06T19:57:52.516213Z","shell.execute_reply.started":"2023-02-06T19:57:52.509968Z","shell.execute_reply":"2023-02-06T19:57:52.514573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = model.predict(test_gen)","metadata":{"execution":{"iopub.status.busy":"2023-01-05T18:45:58.578478Z","iopub.execute_input":"2023-01-05T18:45:58.579239Z","iopub.status.idle":"2023-01-05T18:46:29.202339Z","shell.execute_reply.started":"2023-01-05T18:45:58.5792Z","shell.execute_reply":"2023-01-05T18:46:29.200839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['cancer'] = (pred[:len(test_df)] > 0.5).astype('int')\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-05T18:46:51.253029Z","iopub.execute_input":"2023-01-05T18:46:51.253464Z","iopub.status.idle":"2023-01-05T18:46:51.269554Z","shell.execute_reply.started":"2023-01-05T18:46:51.253431Z","shell.execute_reply":"2023-01-05T18:46:51.268391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = test_df.groupby('prediction_id')['cancer'].max().to_frame().reset_index()\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-05T18:46:56.015602Z","iopub.execute_input":"2023-01-05T18:46:56.016119Z","iopub.status.idle":"2023-01-05T18:46:56.038038Z","shell.execute_reply.started":"2023-01-05T18:46:56.01608Z","shell.execute_reply":"2023-01-05T18:46:56.036501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv(\"submission.csv\", index = False)","metadata":{"execution":{"iopub.status.busy":"2023-01-05T18:46:57.989064Z","iopub.execute_input":"2023-01-05T18:46:57.990146Z","iopub.status.idle":"2023-01-05T18:46:57.998267Z","shell.execute_reply.started":"2023-01-05T18:46:57.990104Z","shell.execute_reply":"2023-01-05T18:46:57.996968Z"},"trusted":true},"execution_count":null,"outputs":[]}]}