{"cells":[{"metadata":{},"cell_type":"markdown","source":"# Intro\nWelcome to the [RSNA Intracranial Hemorrhage Detection](https://www.kaggle.com/c/rsna-intracranial-hemorrhage-detection).\n\n![](https://storage.googleapis.com/kaggle-competitions/kaggle/13451/logos/header.png)\n\nThis notebook is a starter code for all beginners and easy to understand. Used is a image generator based on this [template](https://stanford.edu/~shervine/blog/keras-how-to-generate-data-on-the-fly)\n\nThe hemorrhage types are explained [here](https://www.kaggle.com/c/rsna-intracranial-hemorrhage-detection/overview/hemorrhage-types).\n\nThe model is based on ResNet50 and runs on GPU.\n\n<span style=\"color: royalblue;\">Please vote the notebook up if it helps you. Thank you. </span>"},{"metadata":{},"cell_type":"markdown","source":"# Load Libraries"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport random\nimport os\nimport matplotlib.pyplot as plt","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import pydicom as dicom\nimport cv2","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import MultiLabelBinarizer","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from keras.utils import to_categorical, Sequence\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Dropout, Flatten, Conv2D, MaxPool2D, Activation\nfrom keras.optimizers import RMSprop,Adam\nfrom keras.applications import VGG19, VGG16, ResNet50","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import tensorflow as tf","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import warnings\nwarnings.filterwarnings(\"ignore\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Path\nDefine the path for the subfolders with the data."},{"metadata":{"trusted":true},"cell_type":"code","source":"path_in = \"../input/rsna-intracranial-hemorrhage-detection/rsna-intracranial-hemorrhage-detection/\"\nos.listdir(path_in)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Define the sub paths with images"},{"metadata":{"trusted":true},"cell_type":"code","source":"path_train_img = path_in + 'stage_2_train'\npath_test_img = path_in + 'stage_2_test'","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Path to the pretrained data set."},{"metadata":{"trusted":true},"cell_type":"code","source":"path_models = '../input/models' \nos.listdir(path_models)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Functions\nWe define some helper functions."},{"metadata":{"trusted":true},"cell_type":"code","source":"def rescale_pixelarray(dataset):\n    image = dataset.pixel_array\n    rescaled_image = image * dataset.RescaleSlope + dataset.RescaleIntercept\n    rescaled_image[rescaled_image < -1024] = -1024\n    return rescaled_image","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def plot_example(data, sub_type='subdural'):\n    \"\"\" Plot 5 examples of a given subtype \"\"\"\n    \n    fig, axs = plt.subplots(1, 5, figsize=(25, 12))\n    fig.subplots_adjust(hspace = .2, wspace=.2)\n    axs = axs.ravel()\n    for i in range(5):\n        idx = data[(data['Label']==1)&(data['sub_type']==sub_type)].index[i]\n        data_file = dicom.dcmread(path_train_img+'/ID_'+data.loc[idx, 'PatientID']+'.dcm')\n        #img = data_file.pixel_array\n        img = rescale_pixelarray(data_file)\n        if type(data_file.WindowCenter) == dicom.multival.MultiValue:\n            window_center = int(data_file.WindowCenter[0])\n        else: \n            window_center = int(data_file.WindowCenter)\n            \n        if type(data_file.WindowWidth) == dicom.multival.MultiValue:\n            window_width = int(data_file.WindowWidth[0])\n        else:\n            window_width = int(data_file.WindowWidth)\n        img_min = window_center - window_width // 2\n        img_max = window_center + window_width // 2\n        window_image = img.copy()\n        window_image[window_image < img_min] = img_min\n        window_image[window_image > img_max] = img_max\n        axs[i].imshow(window_image, cmap=plt.cm.gray)\n        axs[i].set_title(data.loc[idx, 'PatientID']+'_'+data.loc[idx, 'sub_type'])\n        axs[i].set_xticklabels([])\n        axs[i].set_yticklabels([])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def plot_types(data, num_types):\n    \"\"\" Plot image of patient with given number of sub types\"\"\"\n    \n    temp = data[(data['sub_type']!='any')&\n           (data['Label']==1)].groupby('PatientID').sum()\n    fig, ax = plt.subplots(1, 1, figsize=(15, 6))\n   \n    idx = temp[temp['Label']==num_types].index[0]\n    sub_types = list(data[(data['PatientID']==idx)&\n                     (data['Label']!=0)&\n                     (data['sub_type']!='any')]['sub_type'].values)\n    \n    title = idx+':'\n    for sub_type in sub_types:\n        title = title+' '+sub_type\n        if sub_types.index(sub_type) < len(sub_types)-1:\n            title = title+','\n    data_file = dicom.dcmread(path_train_img+'/ID_'+idx+'.dcm')\n    img = rescale_pixelarray(data_file)\n    if type(data_file.WindowCenter) == dicom.multival.MultiValue:\n        window_center = int(data_file.WindowCenter[0])\n    else: \n        window_center = int(data_file.WindowCenter)\n            \n    if type(data_file.WindowWidth) == dicom.multival.MultiValue:\n        window_width = int(data_file.WindowWidth[0])\n    else:\n        window_width = int(data_file.WindowWidth)\n    img_min = window_center - window_width // 2\n    img_max = window_center + window_width // 2\n    window_image = img.copy()\n    window_image[window_image < img_min] = img_min\n    window_image[window_image > img_max] = img_max\n    ax.imshow(window_image, cmap=plt.cm.gray)\n    ax.set_title(title)\n    ax.set_xticklabels([])\n    ax.set_yticklabels([])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#plot_example(train_data, sub_type='subdural')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Parameters"},{"metadata":{"trusted":true},"cell_type":"code","source":"q_size = 200\nimg_channel = 3\nnum_classes = 6","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Read Image Name"},{"metadata":{"trusted":true},"cell_type":"code","source":"list_train_img = os.listdir(path_train_img)\nlist_test_img = os.listdir(path_test_img)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Read Input Data"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_data = pd.read_csv(path_in + 'stage_2_train.csv')\nsub_org = pd.read_csv(path_in + 'stage_2_sample_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_data.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Modify Input Data"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_data['sub_type'] = train_data['ID'].str.split(\"_\", n = 3, expand = True)[2]\ntrain_data['PatientID'] = train_data['ID'].str.split(\"_\", n = 3, expand = True)[1]\nsub_org['sub_type'] = sub_org['ID'].str.split(\"_\", n = 3, expand = True)[2]\nsub_org['PatientID'] = sub_org['ID'].str.split(\"_\", n = 3, expand = True)[1]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_data['sub_type'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Overview"},{"metadata":{"trusted":true},"cell_type":"code","source":"print('number of (unique) train patient ids:', len(train_data['PatientID'].unique()))\nprint('number of train images: ', len(list_train_img))\nprint('number of (unique) test patient ids:', len(sub_org['PatientID'].unique()))\nprint('number of test images: ', len(list_test_img))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# EDA"},{"metadata":{},"cell_type":"markdown","source":"## Intraparenchymal\n* **Location**: Inside of the brain.\n* **Mechanism**: Hight blood pressure, trauma, arteriovenous, malformation, tumor, etc.\n* **Source**: Arterial or venous.\n* **Shape**: Typically rounded.\n* **Presentation**: Acute (sudden onest of headache, nausea, vomiting)."},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_example(train_data, sub_type='intraparenchymal')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Intraventricular\n* **Location**: Inside of the ventricle.\n* **Mechanism**: Can be associated with both intraparenchymal and subarachnoid hermorrhages.\n* **Source**: Arterial or venous.\n* **Shape**: Conforms to ventricular shape.\n* **Presentation**: Acute (sudden onest of headache, nausea, vomiting)."},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_example(train_data, sub_type='intraventricular')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Subarachnoid\n* **Location**: Between the arachonid and the pia mater.\n* **Mechanism**: Rupture of aneurysms or arteriovenous malformations or trauma.\n* **Source**: Predominantly arterial.\n* **Shape**: Tracks along the sulci and fissures.\n* **Presentation**: Acute (worst headache of life)."},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_example(train_data, sub_type='subarachnoid')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Subdural\n* **Location**: Between the Dura and the arachnoid.\n* **Mechanism**: Trauma.\n* **Source**: Venous (bridging veins).\n* **Shape**: Crescent.\n* **Presentation**: May be insidous (worsening headache)."},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_example(train_data, sub_type='subdural')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Epidural\n* **Location**: Between the dura and the skull.\n* **Mechanism**: Trauma or after surgery.\n* **Source**: Arterial.\n* **Shape**: Lentiform.\n* **Presentation**: Acute (skull fracture and altered mental status)"},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_example(train_data, sub_type='epidural')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Group Subtypes\nThere are 5 subtyps and the addditional label any, which should always be true if any of the sub-type labels is true."},{"metadata":{"trusted":true},"cell_type":"code","source":"group_type = train_data.groupby('sub_type').sum()\nfig = plt.figure(figsize=(9, 5))\nax = fig.add_subplot(111)\nax.bar(group_type.index, group_type['Label'])\nax.set_xticklabels(group_type.index, rotation=45)\nplt.grid()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Multilabel\nThere are a lot of patients with a multilabel."},{"metadata":{"trusted":true},"cell_type":"code","source":"train_data[(train_data['sub_type']!='any')&\n           (train_data['Label']==1)].groupby('PatientID').sum()['Label'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"So we can see there are 23 patients which have all labels."},{"metadata":{},"cell_type":"markdown","source":"### 1 Type"},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_types(train_data, 1)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### 2 Types"},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_types(train_data, 2)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### 3 Types"},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_types(train_data, 3)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### 4 Types"},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_types(train_data, 4)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### 5 Types (all types)"},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_types(train_data, 5)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Train And Test Pivot"},{"metadata":{"trusted":true},"cell_type":"code","source":"column_names = ['Label', 'PatientID', 'sub_type']\ntrain_data_pivot = train_data[column_names].drop_duplicates().pivot(index='PatientID',\n                                                                    columns='sub_type',\n                                                                    values='Label')\ntest_data_pivot = sub_org[column_names].drop_duplicates().pivot(index='PatientID',\n                                                                columns='sub_type',\n                                                                values='Label')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Select Subset Input Data For Training\nThis is a big dataset. So we select a smaller subset for the training."},{"metadata":{"trusted":true},"cell_type":"code","source":"percentage = 0.25\nnum_train_img = int(percentage*len(train_data_pivot.index))\nnum_test_img = len(test_data_pivot.index)\nprint('num_train_data:', len(list_train_img), num_train_img)\nprint('num_test_data:', len(list_test_img))\nlist_train_img = list(train_data_pivot.index)\nlist_test_img = list(test_data_pivot.index)\nrandom_train_img = random.sample(list_train_img, num_train_img)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y_train_org = train_data_pivot.loc[random_train_img]","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Split Train And Val"},{"metadata":{"trusted":true},"cell_type":"code","source":"y_train, y_val = train_test_split(y_train_org, test_size=0.3)\ny_test = test_data_pivot","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Calculate Class Weights"},{"metadata":{"trusted":true},"cell_type":"code","source":"class_weight = dict(zip(range(0, num_classes), y_train.sum()/y_train.sum().sum()))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class_weight","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Data Generator"},{"metadata":{"trusted":true},"cell_type":"code","source":"class DataGenerator(Sequence):\n    def __init__(self, path, list_IDs, labels, batch_size,\n                 img_size, img_channel, num_classes, shuffle=True):\n        self.path = path\n        self.list_IDs = list_IDs\n        self.labels = labels\n        self.batch_size = batch_size\n        self.img_size = img_size\n        self.img_channel = img_channel\n        self.num_classes = num_classes\n        self.shuffle = shuffle\n        self.on_epoch_end()\n     \n    \n    def __len__(self):\n        return int(np.floor(len(self.list_IDs)/self.batch_size))\n    \n    \n    def __getitem__(self, index):\n        indexes = self.indexes[index*self.batch_size:(index+1)*self.batch_size]\n        list_IDs_temp = [self.list_IDs[k] for k in indexes]\n        X, y = self.__data_generation(list_IDs_temp)\n        return X, y\n    \n    \n    def on_epoch_end(self):\n        self.indexes = np.arange(len(self.list_IDs))\n        if self.shuffle == True:\n            np.random.shuffle(self.indexes)\n    \n    \n    def rescale_pixelarray(self, dataset):\n        image = dataset.pixel_array\n        rescaled_image = image * dataset.RescaleSlope + dataset.RescaleIntercept\n        rescaled_image[rescaled_image < -1024] = -1024\n        return rescaled_image\n\n    \n    def __data_generation(self, list_IDs_temp):\n        X = np.empty((self.batch_size, self.img_size, self.img_size))\n        y = np.empty((self.batch_size, self.num_classes), dtype=int)\n        for i, ID in enumerate(list_IDs_temp):\n            data_file = dicom.dcmread(self.path+'/ID_'+ID+'.dcm')\n            img = self.rescale_pixelarray(data_file)\n            img = cv2.resize(img, (self.img_size, self.img_size))\n            X[i, ] = img\n            y[i, ] = self.labels.loc[ID]\n        X = np.repeat(X[..., np.newaxis], 3, -1)\n        X = X.astype('float32')\n        X -= X.mean(axis=0)\n        std = X.std(axis=0)\n        X /= X.std(axis=0)\n        return X, y","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Load Pretrained Model"},{"metadata":{"trusted":true},"cell_type":"code","source":"conv_base = ResNet50(weights='../input/models/model_weights_resnet.h5',\n                     include_top=False,\n                     input_shape=(q_size, q_size, img_channel))\nconv_base.trainable = True","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Define train and validation data via Data Generator"},{"metadata":{"trusted":true},"cell_type":"code","source":"batch_size = 32\ntrain_generator = DataGenerator(path_train_img, list(y_train.index), y_train,\n                                batch_size, q_size, img_channel, num_classes)\nval_generator = DataGenerator(path_train_img, list(y_val.index), y_val,\n                                batch_size, q_size, img_channel, num_classes)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Define the model"},{"metadata":{"trusted":true},"cell_type":"code","source":"model = Sequential()\nmodel.add(conv_base)\nmodel.add(Flatten())\nmodel.add(Dense(64, activation='relu'))\nmodel.add(Dense(64, activation='relu'))\nmodel.add(Dropout(0.3))\nmodel.add(Dense(6, activation='sigmoid'))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Compile the model"},{"metadata":{"trusted":true},"cell_type":"code","source":"model.compile(optimizer = RMSprop(lr=1e-5),\n              loss='binary_crossentropy',\n              metrics=['binary_accuracy'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"epochs = 5","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Fit the model with the fit_generator method"},{"metadata":{"trusted":true},"cell_type":"code","source":"history = model.fit_generator(generator=train_generator,\n                              validation_data=val_generator,\n                              epochs = epochs,\n                              class_weight = class_weight,\n                              workers=4)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Plot the loss values"},{"metadata":{"trusted":true},"cell_type":"code","source":"loss = history.history['loss']\nloss_val = history.history['val_loss']\nepochs = range(1, len(loss)+1)\nplt.plot(epochs, loss, 'bo', label='loss_train')\nplt.plot(epochs, loss_val, 'b', label='loss_val')\nplt.title('value of the loss function')\nplt.xlabel('epochs')\nplt.ylabel('value of the loss functio')\nplt.legend()\nplt.grid()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Plot the accuracy values"},{"metadata":{"trusted":true},"cell_type":"code","source":"acc = history.history['binary_accuracy']\nacc_val = history.history['val_binary_accuracy']\nepochs = range(1, len(loss)+1)\nplt.plot(epochs, acc, 'bo', label='accuracy_train')\nplt.plot(epochs, acc_val, 'b', label='accuracy_val')\nplt.title('accuracy')\nplt.xlabel('epochs')\nplt.ylabel('value of accuracy')\nplt.legend()\nplt.grid()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Define the test data via Data Generator"},{"metadata":{"trusted":true},"cell_type":"code","source":"batch_size = 16\ntest_generator = DataGenerator(path_test_img, list(y_test.index), y_test,\n                                batch_size, q_size, img_channel, num_classes, shuffle=False)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Predict the test images with the generator class"},{"metadata":{"trusted":true},"cell_type":"code","source":"predict = model.predict_generator(test_generator, verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"assert(len(predict) == len(test_data_pivot))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Prepare the prediction data by the export format"},{"metadata":{"trusted":true},"cell_type":"code","source":"submission = pd.DataFrame(predict, columns=y_train_org.columns)\nsubmission.insert(loc=0, column='PatientID', value=test_data_pivot.index)\nsubmission.index=submission['PatientID']\nsubmission = submission.drop(['PatientID'], axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission = submission.stack().reset_index()\nsubmission = submission.rename(columns={0: 'Label'})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission.insert(loc=0, column='ID', value='ID_'+submission['PatientID'].astype(str)+'_'+submission['sub_type'].astype(str))\nsubmission = submission.drop(['PatientID', 'sub_type'], axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission.index = submission['ID']\nsubmission = submission.reindex(sub_org['ID'])\nsubmission.index = range(len(submission))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Export the prediction data"},{"metadata":{"trusted":true},"cell_type":"code","source":"submission.to_csv('submission.csv', header = True, index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}