{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# import libraries\nfrom tensorflow import keras\nimport tensorflow as tf\nimport pydicom\nimport numpy as np\nimport pandas as pd\nimport os","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-12-08T07:07:51.641565Z","iopub.execute_input":"2021-12-08T07:07:51.642215Z","iopub.status.idle":"2021-12-08T07:07:55.693568Z","shell.execute_reply.started":"2021-12-08T07:07:51.642114Z","shell.execute_reply":"2021-12-08T07:07:55.6928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get test image path\npath = '../input/rsna-intracranial-hemorrhage-detection/rsna-intracranial-hemorrhage-detection/stage_2_train/'\nlabels = pd.read_csv('../input/rsna-intracranial-hemorrhage-detection/rsna-intracranial-hemorrhage-detection/stage_2_train.csv')","metadata":{"execution":{"iopub.status.busy":"2021-12-08T07:07:55.695232Z","iopub.execute_input":"2021-12-08T07:07:55.695495Z","iopub.status.idle":"2021-12-08T07:07:59.39752Z","shell.execute_reply.started":"2021-12-08T07:07:55.69546Z","shell.execute_reply":"2021-12-08T07:07:59.396787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SAMPLE_SIZE = 215866\nSEED = 42\nBATCH_SIZE = 32\nNUM_CLASSES_BINARY = 1\nNUM_CLASSES_MULTI = 5\nSUBCLASSES = ['any', 'epidural', 'intraparenchymal', 'intraventricular', 'subarachnoid', 'subdural']","metadata":{"execution":{"iopub.status.busy":"2021-12-08T07:07:59.39879Z","iopub.execute_input":"2021-12-08T07:07:59.399087Z","iopub.status.idle":"2021-12-08T07:07:59.404368Z","shell.execute_reply.started":"2021-12-08T07:07:59.399052Z","shell.execute_reply":"2021-12-08T07:07:59.40365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# define image size\nIMAGE_SIZE = (224,224)\n\n# correct dcmd\ndef correct_dcm(dcm):\n    x = dcm.pixel_array + 1000\n    px_mode = 4096\n    x[x>=px_mode] = x[x>=px_mode] - px_mode\n    dcm.PixelData = x.tobytes()\n    dcm.RescaleIntercept = -1000\n    \n# convert dicom field values to integers \ndef get_first_of_dicom_field_as_int(x):\n    if type(x) == pydicom.multival.MultiValue:\n        return int(x[0])\n    return int(x)\n    \n# get windowing values \ndef get_windowing(data):\n    dicom_fields = [data[('0028','1050')].value, # window center\n                    data[('0028','1051')].value, # window width\n                    data[('0028','1052')].value, # intercept\n                    data[('0028','1053')].value, # slope\n                   ]\n    return [get_first_of_dicom_field_as_int(x) for x in dicom_fields]\n    \n# get min and max of the window values\ndef get_min_max_of_window_value(window_center, window_width):\n    mini = window_center - (window_width // 2)\n    maxi = window_center + (window_width // 2) \n    return mini, maxi\n\n# change windowing \ndef window_image(img, window_center, window_width):\n    try:\n        # call get_windowing function to get window values\n        _,_, intercept, slope = get_windowing(img) \n        # change window values \n        img = img.pixel_array * slope + intercept\n        img_min, img_max = get_min_max_of_window_value(window_center, window_width)\n        img[img < img_min] = img_min\n        img[img > img_max] = img_max\n    except:\n        img = img_min * np.ones(IMAGE_SIZE)\n        \n    return img\n\n\n# normalize\ndef normalize(channel, wc_ww: tuple, norm_type = 'none'):\n    if norm_type.lower() == 'none':\n        return channel\n    if norm_type.lower() == 'min_max':\n        mini, maxi = get_min_max_of_window_value(wc_ww[0], wc_ww[1])\n        resulted_channel = (channel - mini) / (maxi - mini)\n        return resulted_channel\n    \n\ndef bsb_window(img, third_window):\n    '''\n    this function preprocesses the DICOM image\n\n        Parameters:\n        - img: DICOM image\n\n        Returns:\n        - bsb_image: image array after preproessing  \n    '''\n    if third_window == \"bone\":\n        third = (600, 2000)\n    else:\n        third = (50, 350)\n\n    bsb_config = {'brain': (40,80),     # brain channel\n             'subdural': (80,200),      # subdural channel\n             third_window: third}       # bone channel\n\n    brain_img = window_image(img, *bsb_config['brain'])         # image with brain channel\n    subdural_img = window_image(img,*bsb_config['subdural'])    # image with subdural channel\n    third_img = window_image(img, *bsb_config[third_window])           # image with bone channel\n    \n    brain_img = normalize(brain_img, bsb_config['brain'], 'min_max')                # normalize image with brain channel\n    subdural_img = normalize(subdural_img, bsb_config['subdural'], 'min_max')       # normalize image with subdural channel\n    third_img = normalize(third_img, bsb_config[third_window], 'min_max')                   # normalize image with bone channel\n\n    # preprocessed image\n    bsb_img = np.zeros((brain_img.shape[0], brain_img.shape[1], 3)) \n    bsb_img[:, :, 0] = brain_img\n    bsb_img[:, :, 1] = subdural_img\n    bsb_img[:, :, 2] = third_img\n    \n    if (np.any(np.isnan(bsb_img))):\n        bsb_img = np.ones((*IMAGE_SIZE,3)) # reshape image \n        \n    return bsb_img\n\ndef preprocess_img_soft(dcm):\n  if (dcm.BitsStored == 12) and (dcm.PixelRepresentation == 0) and (int(dcm.RescaleIntercept) > -100):\n          correct_dcm(dcm)\n  img = bsb_window(dcm, third_window=\"soft\")\n  img = tf.convert_to_tensor(img, dtype=tf.float64)\n  return img\n\ndef preprocess_img_bone(dcm):\n  if (dcm.BitsStored == 12) and (dcm.PixelRepresentation == 0) and (int(dcm.RescaleIntercept) > -100):\n          correct_dcm(dcm)\n  img = bsb_window(dcm, third_window=\"bone\")\n  img = tf.convert_to_tensor(img, dtype=tf.float64)\n  return img","metadata":{"execution":{"iopub.status.busy":"2021-12-08T07:07:59.406542Z","iopub.execute_input":"2021-12-08T07:07:59.407057Z","iopub.status.idle":"2021-12-08T07:07:59.428663Z","shell.execute_reply.started":"2021-12-08T07:07:59.407021Z","shell.execute_reply":"2021-12-08T07:07:59.427966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class ImageGenerator_soft(tf.keras.utils.Sequence):\n    def __init__(self, dataframe,batch_size,shuffle, num_classes):\n        self.dataframe = dataframe\n        self.num_classes = num_classes\n        self.batch_size = batch_size\n        self.shuffle = shuffle\n        \n    def __len__(self):\n        return math.ceil(len(self.dataframe) / self.batch_size)\n    \n    def __getitem__(self, index):\n        batch_df = self.dataframe.iloc[index * self.batch_size: (index+1) * self.batch_size]\n        paths = path + batch_df.index.astype(str)\n        X = np.empty((len(batch_df), *IMAGE_SIZE, 3))\n        y = np.empty((len(batch_df), self.num_classes))\n        for i, p in enumerate(paths):\n            dcm = pydicom.dcmread(p)\n            # correct dcm\n            if (dcm.BitsStored == 12) and (dcm.PixelRepresentation == 0) and (int(dcm.RescaleIntercept) > -100):\n                correct_dcm(dcm)\n\n            img = bsb_window(dcm, third_window=\"soft\")\n            img = tf.convert_to_tensor(img, dtype=tf.float64)\n            X[i] = tf.image.resize(img, IMAGE_SIZE)\n            y[i] = batch_df.iloc[i].values\n            \n        return X, y\n    \n    def on_epoch_end(self):\n        if self.shuffle:\n            self.dataframe = self.dataframe.sample(len(self.dataframe), replace = False, random_state = SEED)\n        self.current_epoch += 1","metadata":{"execution":{"iopub.status.busy":"2021-12-08T07:07:59.429922Z","iopub.execute_input":"2021-12-08T07:07:59.43017Z","iopub.status.idle":"2021-12-08T07:08:00.264194Z","shell.execute_reply.started":"2021-12-08T07:07:59.430138Z","shell.execute_reply":"2021-12-08T07:08:00.26347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class ImageGenerator_bone(tf.keras.utils.Sequence):\n    def __init__(self, dataframe,batch_size,shuffle, num_classes):\n        self.dataframe = dataframe\n        self.num_classes = num_classes\n        self.batch_size = batch_size\n        self.shuffle = shuffle\n        \n    def __len__(self):\n        return math.ceil(len(self.dataframe) / self.batch_size)\n    \n    def __getitem__(self, index):\n        batch_df = self.dataframe.iloc[index * self.batch_size: (index+1) * self.batch_size]\n        paths = path + batch_df.index.astype(str)\n        X = np.empty((len(batch_df), *IMAGE_SIZE, 3))\n        y = np.empty((len(batch_df), self.num_classes))\n        for i, p in enumerate(paths):\n            dcm = pydicom.dcmread(p)\n            # correct dcm\n            if (dcm.BitsStored == 12) and (dcm.PixelRepresentation == 0) and (int(dcm.RescaleIntercept) > -100):\n                correct_dcm(dcm)\n\n            img = bsb_window(dcm, third_window=\"bone\")\n            img = tf.convert_to_tensor(img, dtype=tf.float64)\n            X[i] = tf.image.resize(img, IMAGE_SIZE)\n            y[i] = batch_df.iloc[i].values\n            \n        return X, y\n    \n    def on_epoch_end(self):\n        if self.shuffle:\n            self.dataframe = self.dataframe.sample(len(self.dataframe), replace = False, random_state = SEED)\n        self.current_epoch += 1","metadata":{"execution":{"iopub.status.busy":"2021-12-08T07:08:00.265608Z","iopub.execute_input":"2021-12-08T07:08:00.265884Z","iopub.status.idle":"2021-12-08T07:08:00.280067Z","shell.execute_reply.started":"2021-12-08T07:08:00.265848Z","shell.execute_reply":"2021-12-08T07:08:00.278769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#load saved Binary model\nBinary_model = keras.models.load_model('../input/final-binary/best_model_densenet201.h5')","metadata":{"execution":{"iopub.status.busy":"2021-12-08T07:08:00.283059Z","iopub.execute_input":"2021-12-08T07:08:00.283288Z","iopub.status.idle":"2021-12-08T07:08:15.618609Z","shell.execute_reply.started":"2021-12-08T07:08:00.283258Z","shell.execute_reply":"2021-12-08T07:08:15.617873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# custom loss function <ref?> \ndef np_multilabel_loss(class_weights=None):\n    def single_class_crossentropy(y_true, y_pred):\n        y_true = tf.cast(y_true, tf.float32)\n        y_pred = tf.cast(y_pred, tf.float32)\n        \n        y_pred = tf.where(y_pred > 1-(1e-07), 1-1e-07, y_pred)\n        y_pred = tf.where(y_pred < 1e-07, 1e-07, y_pred)\n        single_class_cross_entropies = - tf.reduce_mean(y_true * tf.math.log(y_pred) + (1-y_true) * tf.math.log(1-y_pred), axis=0)\n\n        if class_weights is None:\n            loss = tf.reduce_mean(single_class_cross_entropies)\n        else:\n            loss = tf.reduce_sum(class_weights*single_class_cross_entropies)\n        return loss\n    return single_class_crossentropy","metadata":{"execution":{"iopub.status.busy":"2021-12-08T07:08:15.622026Z","iopub.execute_input":"2021-12-08T07:08:15.622221Z","iopub.status.idle":"2021-12-08T07:08:15.632071Z","shell.execute_reply.started":"2021-12-08T07:08:15.622197Z","shell.execute_reply":"2021-12-08T07:08:15.631344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#load saved Multilabel model\nMultilabel = keras.models.load_model('../input/multilabel/multilabel.h5', custom_objects={\"single_class_crossentropy\": np_multilabel_loss})","metadata":{"execution":{"iopub.status.busy":"2021-12-08T07:08:15.633603Z","iopub.execute_input":"2021-12-08T07:08:15.634427Z","iopub.status.idle":"2021-12-08T07:08:22.33639Z","shell.execute_reply.started":"2021-12-08T07:08:15.63439Z","shell.execute_reply":"2021-12-08T07:08:22.335646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-08T07:08:22.339794Z","iopub.execute_input":"2021-12-08T07:08:22.340009Z","iopub.status.idle":"2021-12-08T07:08:22.358428Z","shell.execute_reply.started":"2021-12-08T07:08:22.339984Z","shell.execute_reply":"2021-12-08T07:08:22.357681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label = labels.Label\nlabels = labels.ID.str.rsplit('_', n=1, expand = True)\nlabels['label'] = label\nlabels.rename({0:'id', 1: 'subtype'}, axis =1, inplace=True)\nlabels.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-08T07:08:22.359981Z","iopub.execute_input":"2021-12-08T07:08:22.36023Z","iopub.status.idle":"2021-12-08T07:08:32.838379Z","shell.execute_reply.started":"2021-12-08T07:08:22.360196Z","shell.execute_reply":"2021-12-08T07:08:32.837654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = pd.pivot_table(labels, index='id', columns='subtype', values = 'label')\nlabels.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-08T07:08:32.839818Z","iopub.execute_input":"2021-12-08T07:08:32.840256Z","iopub.status.idle":"2021-12-08T07:08:40.704504Z","shell.execute_reply.started":"2021-12-08T07:08:32.840217Z","shell.execute_reply":"2021-12-08T07:08:40.703685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels.index = labels.index.astype(str) + '.dcm'\nlabels.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-08T07:08:40.705587Z","iopub.execute_input":"2021-12-08T07:08:40.705853Z","iopub.status.idle":"2021-12-08T07:08:41.475353Z","shell.execute_reply.started":"2021-12-08T07:08:40.705798Z","shell.execute_reply":"2021-12-08T07:08:41.474686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"normal_df = labels[labels['any'] == 0]       # normal scans\nabnormal_df = labels[labels['any'] == 1]     # abnormal scans","metadata":{"execution":{"iopub.status.busy":"2021-12-08T07:08:41.476648Z","iopub.execute_input":"2021-12-08T07:08:41.476972Z","iopub.status.idle":"2021-12-08T07:08:41.528863Z","shell.execute_reply.started":"2021-12-08T07:08:41.476927Z","shell.execute_reply":"2021-12-08T07:08:41.528119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"normal_sample = normal_df.sample(SAMPLE_SIZE//2, replace = False, random_state = SEED, axis = 0)\nabnormal_sample = abnormal_df.sample(SAMPLE_SIZE//2, replace = False, random_state = SEED, axis = 0)\nsample_df = normal_sample.append(abnormal_sample)\nsample_df = sample_df.sample(frac = 1, random_state = SEED, axis = 0)\nsample_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-08T07:08:41.530156Z","iopub.execute_input":"2021-12-08T07:08:41.530438Z","iopub.status.idle":"2021-12-08T07:08:41.640669Z","shell.execute_reply.started":"2021-12-08T07:08:41.530402Z","shell.execute_reply":"2021-12-08T07:08:41.639878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"binary_df = pd.DataFrame(sample_df['any'])","metadata":{"execution":{"iopub.status.busy":"2021-12-08T07:08:41.641971Z","iopub.execute_input":"2021-12-08T07:08:41.642296Z","iopub.status.idle":"2021-12-08T07:08:41.647647Z","shell.execute_reply.started":"2021-12-08T07:08:41.642259Z","shell.execute_reply":"2021-12-08T07:08:41.646869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_generator = ImageGenerator_soft(binary_df, BATCH_SIZE, shuffle=False, num_classes=NUM_CLASSES_BINARY)","metadata":{"execution":{"iopub.status.busy":"2021-12-08T07:08:41.648978Z","iopub.execute_input":"2021-12-08T07:08:41.649576Z","iopub.status.idle":"2021-12-08T07:08:41.656849Z","shell.execute_reply.started":"2021-12-08T07:08:41.649537Z","shell.execute_reply":"2021-12-08T07:08:41.656051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data_binary = tf.data.Dataset.from_generator(lambda: map(tuple, img_generator), \n                                          output_types=(tf.float64, tf.uint8),\n                                          output_shapes = (\n                                                    tf.TensorShape((None, *IMAGE_SIZE,3)),\n                                                    tf.TensorShape((None, NUM_CLASSES_BINARY))\n                                          ))","metadata":{"execution":{"iopub.status.busy":"2021-12-08T07:08:41.658266Z","iopub.execute_input":"2021-12-08T07:08:41.658628Z","iopub.status.idle":"2021-12-08T07:08:41.709842Z","shell.execute_reply.started":"2021-12-08T07:08:41.658593Z","shell.execute_reply":"2021-12-08T07:08:41.70923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import math\n\n#predict normal/abnormal \nbinaryPred = Binary_model.predict(test_data_binary)\nbinary_res = pd.DataFrame({'prob': binaryPred.flatten()}, index=binary_df.index)\nbinary_res.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-08T07:08:41.711108Z","iopub.execute_input":"2021-12-08T07:08:41.711365Z","iopub.status.idle":"2021-12-08T08:40:27.055366Z","shell.execute_reply.started":"2021-12-08T07:08:41.71133Z","shell.execute_reply":"2021-12-08T08:40:27.054659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"abnormal_pred = binary_res[binary_res['prob'] > 0.5]\nabnormal_pred.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-08T08:41:04.765634Z","iopub.execute_input":"2021-12-08T08:41:04.766559Z","iopub.status.idle":"2021-12-08T08:41:04.787806Z","shell.execute_reply.started":"2021-12-08T08:41:04.766518Z","shell.execute_reply":"2021-12-08T08:41:04.787049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"-----------------------------","metadata":{}},{"cell_type":"code","source":"multi_df = sample_df.loc[abnormal_pred.index]\nmulti_df","metadata":{"execution":{"iopub.status.busy":"2021-12-08T08:41:06.307312Z","iopub.execute_input":"2021-12-08T08:41:06.307866Z","iopub.status.idle":"2021-12-08T08:41:06.42251Z","shell.execute_reply.started":"2021-12-08T08:41:06.307808Z","shell.execute_reply":"2021-12-08T08:41:06.421687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"multi_df = multi_df.drop(columns=['any'])","metadata":{"execution":{"iopub.status.busy":"2021-12-08T08:41:06.687092Z","iopub.execute_input":"2021-12-08T08:41:06.687333Z","iopub.status.idle":"2021-12-08T08:41:06.693286Z","shell.execute_reply.started":"2021-12-08T08:41:06.687304Z","shell.execute_reply":"2021-12-08T08:41:06.692153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_generator_multi = ImageGenerator_bone(multi_df, BATCH_SIZE, shuffle=False, num_classes=NUM_CLASSES_MULTI)","metadata":{"execution":{"iopub.status.busy":"2021-12-08T08:41:08.058117Z","iopub.execute_input":"2021-12-08T08:41:08.058597Z","iopub.status.idle":"2021-12-08T08:41:08.062869Z","shell.execute_reply.started":"2021-12-08T08:41:08.058558Z","shell.execute_reply":"2021-12-08T08:41:08.061998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data_multi = tf.data.Dataset.from_generator(lambda: map(tuple, img_generator_multi), \n                                          output_types=(tf.float64, tf.uint8),\n                                          output_shapes = (\n                                                    tf.TensorShape((None, *IMAGE_SIZE,3)),\n                                                    tf.TensorShape((None, NUM_CLASSES_MULTI))\n                                          ))","metadata":{"execution":{"iopub.status.busy":"2021-12-08T08:41:08.786285Z","iopub.execute_input":"2021-12-08T08:41:08.78707Z","iopub.status.idle":"2021-12-08T08:41:08.812838Z","shell.execute_reply.started":"2021-12-08T08:41:08.787033Z","shell.execute_reply":"2021-12-08T08:41:08.812196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# if abnormality is detected, predict the type of the hemorrhage \nmultiPred = Multilabel.predict(test_data_multi)\n","metadata":{"execution":{"iopub.status.busy":"2021-12-08T08:41:09.522889Z","iopub.execute_input":"2021-12-08T08:41:09.523224Z","iopub.status.idle":"2021-12-08T09:17:02.110711Z","shell.execute_reply.started":"2021-12-08T08:41:09.523189Z","shell.execute_reply":"2021-12-08T09:17:02.109848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"multi_res = pd.DataFrame(multiPred, columns = SUBCLASSES[1:], index=multi_df.index)\nmulti_res.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-08T09:19:49.051478Z","iopub.execute_input":"2021-12-08T09:19:49.052044Z","iopub.status.idle":"2021-12-08T09:19:49.064603Z","shell.execute_reply.started":"2021-12-08T09:19:49.052004Z","shell.execute_reply":"2021-12-08T09:19:49.063659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"multi_res['any'] = abnormal_pred.prob\nmulti_res.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-08T09:19:49.790734Z","iopub.execute_input":"2021-12-08T09:19:49.791587Z","iopub.status.idle":"2021-12-08T09:19:49.807254Z","shell.execute_reply.started":"2021-12-08T09:19:49.791533Z","shell.execute_reply":"2021-12-08T09:19:49.806476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"normal_pred = binary_res[binary_res['prob'] <= 0.5]\nnormal_pred.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-08T09:19:50.049622Z","iopub.execute_input":"2021-12-08T09:19:50.050111Z","iopub.status.idle":"2021-12-08T09:19:50.072417Z","shell.execute_reply.started":"2021-12-08T09:19:50.050079Z","shell.execute_reply":"2021-12-08T09:19:50.071723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"normal_pred = normal_pred.rename(columns={'prob':'any'})\nnormal_pred.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-08T09:19:50.286634Z","iopub.execute_input":"2021-12-08T09:19:50.286945Z","iopub.status.idle":"2021-12-08T09:19:50.315927Z","shell.execute_reply.started":"2021-12-08T09:19:50.286909Z","shell.execute_reply":"2021-12-08T09:19:50.314628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"normal_pred[SUBCLASSES[1:]] = 0\nnormal_pred.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-08T09:19:50.58429Z","iopub.execute_input":"2021-12-08T09:19:50.584589Z","iopub.status.idle":"2021-12-08T09:19:50.607236Z","shell.execute_reply.started":"2021-12-08T09:19:50.584555Z","shell.execute_reply":"2021-12-08T09:19:50.606624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"whole_pred = pd.concat((normal_pred, multi_res))\nwhole_pred","metadata":{"execution":{"iopub.status.busy":"2021-12-08T09:19:50.880563Z","iopub.execute_input":"2021-12-08T09:19:50.880804Z","iopub.status.idle":"2021-12-08T09:19:50.910491Z","shell.execute_reply.started":"2021-12-08T09:19:50.880776Z","shell.execute_reply":"2021-12-08T09:19:50.909697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_df = sample_df.loc[whole_pred.index]\nsample_df","metadata":{"execution":{"iopub.status.busy":"2021-12-08T09:19:51.086205Z","iopub.execute_input":"2021-12-08T09:19:51.086521Z","iopub.status.idle":"2021-12-08T09:19:51.227659Z","shell.execute_reply.started":"2021-12-08T09:19:51.086494Z","shell.execute_reply":"2021-12-08T09:19:51.226964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_df.to_csv('testing_data.csv')\nwhole_pred.to_csv('predicted_data.csv')","metadata":{"execution":{"iopub.status.busy":"2021-12-08T09:19:51.432547Z","iopub.execute_input":"2021-12-08T09:19:51.433327Z","iopub.status.idle":"2021-12-08T09:19:53.550791Z","shell.execute_reply.started":"2021-12-08T09:19:51.433284Z","shell.execute_reply":"2021-12-08T09:19:53.550057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # check the results of the binary prediction \n# if binaryPred > 0.5:\n#     # if abnormality is detected, predict the type of the hemorrhage \n#     multiPred = Multilabel.predict(img)\n    \n# else:\n#     print('No hemorrhage detected')","metadata":{"execution":{"iopub.status.busy":"2021-12-08T09:19:53.552434Z","iopub.execute_input":"2021-12-08T09:19:53.552669Z","iopub.status.idle":"2021-12-08T09:19:53.558173Z","shell.execute_reply.started":"2021-12-08T09:19:53.552637Z","shell.execute_reply":"2021-12-08T09:19:53.557496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Testing","metadata":{}},{"cell_type":"code","source":"y_pred = pd.read_csv('predicted_data.csv')\ny_test = pd.read_csv('testing_data.csv')","metadata":{"execution":{"iopub.status.busy":"2021-12-08T09:19:56.791418Z","iopub.execute_input":"2021-12-08T09:19:56.792213Z","iopub.status.idle":"2021-12-08T09:19:57.221936Z","shell.execute_reply.started":"2021-12-08T09:19:56.792176Z","shell.execute_reply":"2021-12-08T09:19:57.221169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"columns = list(y_test.columns)\ncolumns = columns[1:]","metadata":{"execution":{"iopub.status.busy":"2021-12-08T09:19:57.223359Z","iopub.execute_input":"2021-12-08T09:19:57.223608Z","iopub.status.idle":"2021-12-08T09:19:57.230311Z","shell.execute_reply.started":"2021-12-08T09:19:57.223574Z","shell.execute_reply":"2021-12-08T09:19:57.22966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = pd.pivot_table(y_pred, index='id')\ny_test = pd.pivot_table(y_test, index='id')","metadata":{"execution":{"iopub.status.busy":"2021-12-08T09:19:57.332751Z","iopub.execute_input":"2021-12-08T09:19:57.33327Z","iopub.status.idle":"2021-12-08T09:19:58.51932Z","shell.execute_reply.started":"2021-12-08T09:19:57.333241Z","shell.execute_reply":"2021-12-08T09:19:58.516915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = (y_pred > 0.5) \n","metadata":{"execution":{"iopub.status.busy":"2021-12-08T09:19:58.520986Z","iopub.execute_input":"2021-12-08T09:19:58.521379Z","iopub.status.idle":"2021-12-08T09:19:58.532909Z","shell.execute_reply.started":"2021-12-08T09:19:58.521343Z","shell.execute_reply":"2021-12-08T09:19:58.532226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import classification_report\nprint(classification_report(y_test, y_pred,target_names=columns))\n","metadata":{"execution":{"iopub.status.busy":"2021-12-08T09:19:58.536758Z","iopub.execute_input":"2021-12-08T09:19:58.537206Z","iopub.status.idle":"2021-12-08T09:19:59.454323Z","shell.execute_reply.started":"2021-12-08T09:19:58.537173Z","shell.execute_reply":"2021-12-08T09:19:59.453535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score\n# accuracy: (tp + tn) / (p + n)\naccuracy = accuracy_score(y_test,y_pred)\nprint('Accuracy: %f' % accuracy)\n# precision tp / (tp + fp)\nprecision = precision_score(y_test,y_pred,average='samples')\nprint('Precision: %f' % precision)\n# recall: tp / (tp + fn)\nrecall = recall_score(y_test,y_pred,average='samples')\nprint('Recall: %f' % recall)\n# f1: 2 tp / (2 tp + fp + fn)\nf1 = f1_score(y_test,y_pred,average='samples')\nprint('F1 score: %f' % f1)\n","metadata":{"execution":{"iopub.status.busy":"2021-12-08T09:19:59.456122Z","iopub.execute_input":"2021-12-08T09:19:59.456687Z","iopub.status.idle":"2021-12-08T09:20:00.078202Z","shell.execute_reply.started":"2021-12-08T09:19:59.456646Z","shell.execute_reply":"2021-12-08T09:20:00.077492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}}]}