{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Stroke Blood Clot Origin (Model Training using Tensorflow-Keras): My approach\n\n* Firstly, credits to Tmyok's notebook for converting large TIF files to JPG images: https://www.kaggle.com/code/tmyok1984/mayo-convert-tif-to-jpg \n\n* This notebook focuses on optimizing the data pipeline and model performance by trying different combinations of several already availabe pretrained models and data augmentations.\n\n* If you like my work then do consider upvoting this notebook. Also if you have any better approach, feel free to enlighten me in the comments.","metadata":{}},{"cell_type":"markdown","source":"## Imports","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nimport matplotlib.pyplot as plt\nimport tensorflow as tf\nimport cv2\nfrom tensorflow import keras\nfrom tensorflow.keras.utils import Sequence\nimport warnings\nfrom tqdm import tqdm\nimport gc\ngc.enable()\nimport PIL\nfrom sklearn.model_selection import train_test_split\nPIL.Image.MAX_IMAGE_PIXELS = None\ntf.__version__, PIL.__version__","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-18T08:43:39.767088Z","iopub.execute_input":"2022-08-18T08:43:39.767533Z","iopub.status.idle":"2022-08-18T08:43:45.96578Z","shell.execute_reply.started":"2022-08-18T08:43:39.767447Z","shell.execute_reply":"2022-08-18T08:43:45.964705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def set_seed(seed=31415):\n    np.random.seed(seed)\n    tf.random.set_seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    os.environ['TF_DETERMINISTIC_OPS'] = '1'\n    os.environ['TF_CPP_MINLOG_LEVEL'] = '2'\n    warnings.simplefilter('ignore')\nset_seed()","metadata":{"execution":{"iopub.status.busy":"2022-08-18T08:43:45.96852Z","iopub.execute_input":"2022-08-18T08:43:45.969663Z","iopub.status.idle":"2022-08-18T08:43:45.978658Z","shell.execute_reply.started":"2022-08-18T08:43:45.969619Z","shell.execute_reply":"2022-08-18T08:43:45.975498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cfg = {\n    'paths':{\n        'train_csv': '../input/mayo-clinic-strip-ai/train.csv',\n        'test_csv': '../input/mayo-clinic-strip-ai/test.csv',\n        'other_csv': '../input/mayo-clinic-strip-ai/other.csv',\n        'train_img': '../input/sbco-train-data/train/'\n    },\n    'img_shape':(512, 512),\n    'epochs': 50,\n    'lr': 1e-02,\n    'batch_size': 16,\n    'valid_split': 0.2\n}","metadata":{"execution":{"iopub.status.busy":"2022-08-18T08:43:45.98006Z","iopub.execute_input":"2022-08-18T08:43:45.980535Z","iopub.status.idle":"2022-08-18T08:43:45.996288Z","shell.execute_reply.started":"2022-08-18T08:43:45.980484Z","shell.execute_reply":"2022-08-18T08:43:45.995096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(cfg['paths']['train_csv'])\ndf['image_id'] = df['image_id'] + '.jpg'\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-18T08:43:45.998782Z","iopub.execute_input":"2022-08-18T08:43:45.999758Z","iopub.status.idle":"2022-08-18T08:43:46.032893Z","shell.execute_reply.started":"2022-08-18T08:43:45.999649Z","shell.execute_reply":"2022-08-18T08:43:46.031912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"enc = {label: str(idx) for idx, label in enumerate(df.label.unique())}\ndf['encoded_labels'] = [enc[i] for i in tqdm(df.label)]\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-18T08:43:48.134068Z","iopub.execute_input":"2022-08-18T08:43:48.136955Z","iopub.status.idle":"2022-08-18T08:43:48.163915Z","shell.execute_reply.started":"2022-08-18T08:43:48.136912Z","shell.execute_reply":"2022-08-18T08:43:48.16299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df, valid_df = train_test_split(df, test_size=0.1, random_state=42, shuffle=True)\ntrain_df.reset_index(drop=True); valid_df.reset_index(drop=True)\nprint(train_df.shape, valid_df.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-18T08:43:48.419322Z","iopub.execute_input":"2022-08-18T08:43:48.419712Z","iopub.status.idle":"2022-08-18T08:43:48.430179Z","shell.execute_reply.started":"2022-08-18T08:43:48.419649Z","shell.execute_reply":"2022-08-18T08:43:48.42904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model ","metadata":{}},{"cell_type":"code","source":"def create_final_model(\n    augments=False,\n    trainable=False,\n    weights=None \n):\n    '''\n    augments: Bool, whether to augment the data or not\n    trainable: Bool, whether to make the resnet model trainable or not\n    weights: str, takes either None or 'imagenet', whether to use pretrained imagenet weights or not \n    '''\n    ip = tf.keras.Input(shape=(cfg['img_shape'][0], cfg['img_shape'][1], 3))\n    \n    if augments:\n        data_augs = tf.keras.Sequential([\n            tf.keras.layers.experimental.preprocessing.RandomFlip('horizontal', seed=0),\n            tf.keras.layers.experimental.preprocessing.RandomRotation(factor=0.25, seed=0),\n            tf.keras.layers.experimental.preprocessing.RandomZoom(height_factor=0.2, width_factor= 0.1, seed=0)\n        ])\n        x = data_augs(ip)\n        x = tf.keras.applications.EfficientNetB4(include_top=False, weights=weights, input_tensor=x)\n    else:\n        x = tf.keras.applications.EfficientNetB4(include_top=False, weights=weights, input_tensor=ip)\n\n    x.trainable = trainable\n    x = tf.keras.layers.GlobalAveragePooling2D()(x.output)\n    x = tf.keras.layers.Dense(512, activation='relu')(x)\n    x = tf.keras.layers.Dropout(0.25)(x)\n    x = tf.keras.layers.Dense(128, activation='relu')(x)\n    x = tf.keras.layers.Dropout(0.25)(x)\n    x = tf.keras.layers.Dense(32, activation='relu')(x)\n    x = tf.keras.layers.Dropout(0.25)(x)\n    #x = tf.keras.layers.Reshape((2048, 1))(x)\n    #x = tf.keras.layers.Bidirectional(tf.keras.layers.LSTM(256, activation='relu'))(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.Dense(1, activation='sigmoid')(x)       \n    model = tf.keras.models.Model(inputs=ip, outputs=x)\n    return model\n\nmodel = create_final_model(augments=False, trainable=False, weights='imagenet')\n#model.summary()","metadata":{"execution":{"iopub.status.busy":"2022-08-18T08:47:10.28621Z","iopub.execute_input":"2022-08-18T08:47:10.286594Z","iopub.status.idle":"2022-08-18T08:47:16.726622Z","shell.execute_reply.started":"2022-08-18T08:47:10.286562Z","shell.execute_reply":"2022-08-18T08:47:16.72555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Creating Data Pipeline","metadata":{}},{"cell_type":"markdown","source":"### Approach 1","metadata":{}},{"cell_type":"code","source":"# dataset_gen = tf.keras.preprocessing.image.ImageDataGenerator(\n#     featurewise_center=False,\n#     samplewise_center=False,\n#     featurewise_std_normalization=False,\n#     samplewise_std_normalization=False,\n#     zca_whitening=False,\n#     zca_epsilon=1e-06,\n#     rotation_range=40,\n#     width_shift_range=0.2,\n#     height_shift_range=0.2,\n#     brightness_range=None,\n#     shear_range=0,\n#     zoom_range=20,\n#     channel_shift_range=0.0,\n#     fill_mode=\"nearest\",\n#     cval=0.0,\n#     horizontal_flip=True,\n#     vertical_flip=False,\n#     rescale=1.0 / 255.0,                #Image normalization\n#     preprocessing_function=None,\n#     data_format=None,\n#     validation_split=0.2,\n#     dtype=None,\n# )\n\n# train_data = dataset_gen.flow_from_dataframe(\n#                              train_df,\n#                              directory = '/kaggle/input/sbco-train-data/train/',\n#                              seed=42,\n#                              subset='training',\n#                              x_col = 'image_id',\n#                              y_col = 'label',\n#                              target_size = cfg['img_shape'],\n#                              color_mode= 'rgb',\n#                              class_mode = 'binary',\n#                              interpolation = 'nearest',\n#                              shuffle = True,\n#                              batch_size = cfg['batch_size']\n# )\n\n# valid_data = dataset_gen.flow_from_dataframe(\n#                              train_df,\n#                              directory = '/kaggle/input/sbco-train-data/train/',\n#                              seed=42,\n#                              subset='validation',\n#                              x_col = 'image_id',\n#                              y_col = 'label',\n#                              target_size = cfg['img_shape'],\n#                              color_mode= 'rgb',\n#                              class_mode = 'binary',\n#                              interpolation = 'nearest',\n#                              shuffle = True,\n#                              batch_size = cfg['batch_size']\n# )","metadata":{"execution":{"iopub.status.busy":"2022-08-18T08:47:18.916255Z","iopub.execute_input":"2022-08-18T08:47:18.916662Z","iopub.status.idle":"2022-08-18T08:47:18.926422Z","shell.execute_reply.started":"2022-08-18T08:47:18.916629Z","shell.execute_reply":"2022-08-18T08:47:18.925329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Approach 2","metadata":{}},{"cell_type":"code","source":"class DatasetGen(Sequence):\n    def __init__(\n        self, df,\n        img_shape=cfg['img_shape'][0],\n        batch_size=cfg['batch_size'],\n        img_path = cfg['paths']['train_img'],\n        shuffle = True\n    ):\n        self.df = df\n        self.img_shape = img_shape\n        self.batch_size = batch_size\n        self.img_path = img_path\n        self.shuffle = shuffle\n        \n    def one_epoch_end(self):\n        if self.shuffle:\n            self.df = self.df.sample(frac=1).reset_index(drop=True)\n        return self.df\n        \n    def __len__(self):\n        return len(self.df) \n    \n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        return self.__get_data(row)\n    \n    def __get_data(self, row):\n        img_path = os.path.join(self.img_path, row['image_id'])\n        img = cv2.imread(img_path, cv2.COLOR_BGR2RGB)\n        img = cv2.resize(img, (self.img_shape, self.img_shape), interpolation=cv2.INTER_AREA)\n        img = img / 255.0\n        #label = [0] * 2; label[int(row['encoded_labels'])] = 1\n        label = [int(row['encoded_labels'])]\n        return img, np.array(label)\n    \n    def __call__(self):\n        for idx in range(self.__len__()):\n            yield self.__getitem__(idx)","metadata":{"execution":{"iopub.status.busy":"2022-08-18T08:47:19.754145Z","iopub.execute_input":"2022-08-18T08:47:19.754947Z","iopub.status.idle":"2022-08-18T08:47:19.766048Z","shell.execute_reply.started":"2022-08-18T08:47:19.754894Z","shell.execute_reply":"2022-08-18T08:47:19.764956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = tf.data.Dataset.from_generator(\n    DatasetGen(train_df),\n    output_types=(tf.float32, tf.int16),\n    output_shapes=((cfg['img_shape'][0], cfg['img_shape'][1], 3), (1,))\n)\n\nvalid_data = tf.data.Dataset.from_generator(\n    DatasetGen(valid_df),\n    output_types=(tf.float32, tf.int16),\n    output_shapes=((cfg['img_shape'][0], cfg['img_shape'][1], 3), (1,))\n)\n\nat = tf.data.AUTOTUNE\ntrain_data = train_data.prefetch(at).batch(cfg['batch_size'])\nvalid_data = valid_data.prefetch(at).batch(cfg['batch_size'])","metadata":{"execution":{"iopub.status.busy":"2022-08-18T08:47:19.956743Z","iopub.execute_input":"2022-08-18T08:47:19.957085Z","iopub.status.idle":"2022-08-18T08:47:20.029682Z","shell.execute_reply.started":"2022-08-18T08:47:19.957056Z","shell.execute_reply":"2022-08-18T08:47:20.028417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model compilation and training","metadata":{}},{"cell_type":"code","source":"model.compile(\n    optimizer=tf.keras.optimizers.SGD(lr=cfg['lr'], momentum=0.9, nesterov=True), #Adam(learning_rate=cfg['lr']\n    loss= tf.keras.losses.BinaryCrossentropy(from_logits=True),\n    metrics = ['Accuracy']\n)\n\ncheckpoint = tf.keras.callbacks.ModelCheckpoint(\n    filepath='/kaggle/working/best_model.h5',\n    monitor='val_loss',\n    verbose=0,\n    save_best_only=True,\n    mode='min'\n)\n\nreduce_lr = tf.keras.callbacks.ReduceLROnPlateau(\n    monitor='val_loss',\n    factor=0.25,\n    patience=3,\n    verbose=0,\n    mode='min'\n)\n\nes = tf.keras.callbacks.EarlyStopping(\n    patience=3,\n    min_delta=0,\n    monitor='val_loss',\n    restore_best_weights=True,\n    verbose=0,\n    mode='min',\n    baseline=None\n)","metadata":{"execution":{"iopub.status.busy":"2022-08-18T08:47:20.560617Z","iopub.execute_input":"2022-08-18T08:47:20.561819Z","iopub.status.idle":"2022-08-18T08:47:20.58754Z","shell.execute_reply.started":"2022-08-18T08:47:20.561766Z","shell.execute_reply":"2022-08-18T08:47:20.586539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(\n            train_data,\n            validation_data=valid_data,\n            epochs = cfg['epochs'],\n            callbacks=[es,reduce_lr,checkpoint],\n            verbose=1\n        )","metadata":{"execution":{"iopub.status.busy":"2022-08-18T08:47:21.402971Z","iopub.execute_input":"2022-08-18T08:47:21.403663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(history.history['Accuracy'])\nplt.plot(history.history['val_Accuracy'])\nplt.title('model accuracy')\nplt.ylabel('accuracy')\nplt.xlabel('epoch')\nplt.legend(['train', 'test'], loc='upper left')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(history.history['loss'])\nplt.plot(history.history['val_loss'])\nplt.title('model loss')\nplt.ylabel('loss')\nplt.xlabel('epoch')\nplt.legend(['train', 'test'], loc='upper left')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Final word","metadata":{}},{"cell_type":"markdown","source":"* The training and validation accuracy when I trained the model stagnated after reaching a particular value, it neither increased nor decreased and I couldn't figure out why was that the case. If anyone has any solutions, let me in the comment section. \n\n* Looks like more work on data preprocessing needs to be done because (although I'm not entirely sure about it), what I'm doing here is resizing the huge files to (512, 512) sized images and that I think means losing a lot of data.\n\n* Sorry for bringing this up once again: Do consider upvoting this notebook.","metadata":{}}]}