{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":52254,"databundleVersionId":8756537,"sourceType":"competition"},{"sourceId":6211844,"sourceType":"datasetVersion","datasetId":3567114}],"dockerImageVersionId":30699,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# <center>Medical image ...</center>\n\n","metadata":{"papermill":{"duration":0.025483,"end_time":"2022-02-01T10:21:43.84374","exception":false,"start_time":"2022-02-01T10:21:43.818257","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"## To commence this project, the neccesary libraries need to be installed.\n## We would be using the **TensorFlow** framework and the pretrained **EfficientNetB1**\n","metadata":{}},{"cell_type":"code","source":"# Imporitng libraries\nimport pandas as pd\nimport os\nimport random\nimport numpy as np\nimport tensorflow as tf\nimport cv2\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom IPython.display import clear_output\nfrom sklearn.metrics import confusion_matrix, ConfusionMatrixDisplay\nfrom sklearn.model_selection import StratifiedKFold\nfrom tensorflow.keras import backend as K\n\nrandom.seed(42)","metadata":{"id":"fUPKjZaaJHkM","papermill":{"duration":5.021373,"end_time":"2022-02-01T10:21:48.88737","exception":false,"start_time":"2022-02-01T10:21:43.865997","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-08-07T15:53:26.751885Z","iopub.execute_input":"2024-08-07T15:53:26.752157Z","iopub.status.idle":"2024-08-07T15:53:40.122406Z","shell.execute_reply.started":"2024-08-07T15:53:26.752132Z","shell.execute_reply":"2024-08-07T15:53:40.121398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Loading the training dataset\ntrain_img = \"/kaggle/input/rsna-atd-512x512-png-v2-dataset/train_images\"","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:53:40.124759Z","iopub.execute_input":"2024-08-07T15:53:40.125484Z","iopub.status.idle":"2024-08-07T15:53:40.129485Z","shell.execute_reply.started":"2024-08-07T15:53:40.125449Z","shell.execute_reply":"2024-08-07T15:53:40.128626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in os.listdir(train_img):\n    for j in os.listdir(os.path.join(train_img,i)):\n        print(len(os.listdir(os.path.join(train_img,i,j))))","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:53:40.130556Z","iopub.execute_input":"2024-08-07T15:53:40.130819Z","iopub.status.idle":"2024-08-07T15:53:44.748672Z","shell.execute_reply.started":"2024-08-07T15:53:40.130796Z","shell.execute_reply":"2024-08-07T15:53:44.747765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Part A\n## Import and explore the data","metadata":{"papermill":{"duration":0.02193,"end_time":"2022-02-01T10:21:48.93356","exception":false,"start_time":"2022-02-01T10:21:48.91163","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Making a list containing all unique classes in the training set\n\n# Reading the training labels\ntraining_labels = pd.read_csv(\"/kaggle/input/rsna-2023-abdominal-trauma-detection/train_2024.csv\")","metadata":{"id":"PBlH8ae3LbG0","papermill":{"duration":0.069238,"end_time":"2022-02-01T10:21:49.024476","exception":false,"start_time":"2022-02-01T10:21:48.955238","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-08-07T15:53:44.749672Z","iopub.execute_input":"2024-08-07T15:53:44.749957Z","iopub.status.idle":"2024-08-07T15:53:44.767128Z","shell.execute_reply.started":"2024-08-07T15:53:44.749933Z","shell.execute_reply":"2024-08-07T15:53:44.766262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Viewing the dataset in a structured format\ntraining_labels","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:53:44.770115Z","iopub.execute_input":"2024-08-07T15:53:44.770479Z","iopub.status.idle":"2024-08-07T15:53:44.800241Z","shell.execute_reply.started":"2024-08-07T15:53:44.770456Z","shell.execute_reply":"2024-08-07T15:53:44.799381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Part B\n## In the next few cells, we chose to explore the dataframe to check for missing values","metadata":{}},{"cell_type":"code","source":"training_labels.columns","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:53:44.801245Z","iopub.execute_input":"2024-08-07T15:53:44.801533Z","iopub.status.idle":"2024-08-07T15:53:44.807173Z","shell.execute_reply.started":"2024-08-07T15:53:44.801511Z","shell.execute_reply":"2024-08-07T15:53:44.80625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels['patient_id'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:53:44.808128Z","iopub.execute_input":"2024-08-07T15:53:44.80842Z","iopub.status.idle":"2024-08-07T15:53:44.823694Z","shell.execute_reply.started":"2024-08-07T15:53:44.808396Z","shell.execute_reply":"2024-08-07T15:53:44.822853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels['bowel_healthy'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:53:44.826399Z","iopub.execute_input":"2024-08-07T15:53:44.8267Z","iopub.status.idle":"2024-08-07T15:53:44.833325Z","shell.execute_reply.started":"2024-08-07T15:53:44.826679Z","shell.execute_reply":"2024-08-07T15:53:44.832436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels['bowel_injury'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:53:44.83448Z","iopub.execute_input":"2024-08-07T15:53:44.83475Z","iopub.status.idle":"2024-08-07T15:53:44.844389Z","shell.execute_reply.started":"2024-08-07T15:53:44.834728Z","shell.execute_reply":"2024-08-07T15:53:44.843444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels['extravasation_healthy'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:53:44.845561Z","iopub.execute_input":"2024-08-07T15:53:44.845898Z","iopub.status.idle":"2024-08-07T15:53:44.855915Z","shell.execute_reply.started":"2024-08-07T15:53:44.845867Z","shell.execute_reply":"2024-08-07T15:53:44.855047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels['extravasation_injury'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:53:44.856958Z","iopub.execute_input":"2024-08-07T15:53:44.857247Z","iopub.status.idle":"2024-08-07T15:53:44.869402Z","shell.execute_reply.started":"2024-08-07T15:53:44.857225Z","shell.execute_reply":"2024-08-07T15:53:44.868597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels['kidney_healthy'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:53:44.870368Z","iopub.execute_input":"2024-08-07T15:53:44.870638Z","iopub.status.idle":"2024-08-07T15:53:44.880497Z","shell.execute_reply.started":"2024-08-07T15:53:44.870615Z","shell.execute_reply":"2024-08-07T15:53:44.87969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels['kidney_low'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:53:44.881461Z","iopub.execute_input":"2024-08-07T15:53:44.882173Z","iopub.status.idle":"2024-08-07T15:53:44.893137Z","shell.execute_reply.started":"2024-08-07T15:53:44.882146Z","shell.execute_reply":"2024-08-07T15:53:44.8923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels['kidney_high'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:53:44.897858Z","iopub.execute_input":"2024-08-07T15:53:44.898138Z","iopub.status.idle":"2024-08-07T15:53:44.905011Z","shell.execute_reply.started":"2024-08-07T15:53:44.898116Z","shell.execute_reply":"2024-08-07T15:53:44.904193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels['liver_healthy'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:53:44.905991Z","iopub.execute_input":"2024-08-07T15:53:44.906276Z","iopub.status.idle":"2024-08-07T15:53:44.916974Z","shell.execute_reply.started":"2024-08-07T15:53:44.906254Z","shell.execute_reply":"2024-08-07T15:53:44.916089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels['liver_low'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:53:44.918082Z","iopub.execute_input":"2024-08-07T15:53:44.918664Z","iopub.status.idle":"2024-08-07T15:53:44.929309Z","shell.execute_reply.started":"2024-08-07T15:53:44.918634Z","shell.execute_reply":"2024-08-07T15:53:44.928443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels['liver_high'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:53:44.930314Z","iopub.execute_input":"2024-08-07T15:53:44.930595Z","iopub.status.idle":"2024-08-07T15:53:44.942957Z","shell.execute_reply.started":"2024-08-07T15:53:44.930573Z","shell.execute_reply":"2024-08-07T15:53:44.942011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels['spleen_healthy'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:53:44.944023Z","iopub.execute_input":"2024-08-07T15:53:44.944359Z","iopub.status.idle":"2024-08-07T15:53:44.952508Z","shell.execute_reply.started":"2024-08-07T15:53:44.94433Z","shell.execute_reply":"2024-08-07T15:53:44.951578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels['spleen_low'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:53:44.953574Z","iopub.execute_input":"2024-08-07T15:53:44.954276Z","iopub.status.idle":"2024-08-07T15:53:44.964977Z","shell.execute_reply.started":"2024-08-07T15:53:44.954252Z","shell.execute_reply":"2024-08-07T15:53:44.964135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels['spleen_high'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:53:44.966202Z","iopub.execute_input":"2024-08-07T15:53:44.966486Z","iopub.status.idle":"2024-08-07T15:53:44.976783Z","shell.execute_reply.started":"2024-08-07T15:53:44.966464Z","shell.execute_reply":"2024-08-07T15:53:44.975981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels['any_injury'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:53:44.97794Z","iopub.execute_input":"2024-08-07T15:53:44.978277Z","iopub.status.idle":"2024-08-07T15:53:44.988359Z","shell.execute_reply.started":"2024-08-07T15:53:44.978248Z","shell.execute_reply":"2024-08-07T15:53:44.987414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Metrics(tf.keras.metrics.Metric):\n    def __init__(self, name='metrics', **kwargs):\n        super(Metrics, self).__init__(name=name, **kwargs)\n        self.precision = tf.keras.metrics.Precision()\n        self.recall = tf.keras.metrics.Recall()\n\n    def update_state(self, y_true, y_pred, sample_weight=None):\n        self.precision.update_state(y_true, y_pred, sample_weight)\n        self.recall.update_state(y_true, y_pred, sample_weight)\n\n    def result(self):\n        return {\n            \"precision\": self.precision.result(),\n            \"recall\": self.recall.result(),\n            \"f1_score\": 2 * ((self.precision.result() * self.recall.result()) / (self.precision.result() + self.recall.result() + K.epsilon()))\n        }\n\n    def reset_states(self):\n        self.precision.reset_states()\n        self.recall.reset_states()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:53:44.989368Z","iopub.execute_input":"2024-08-07T15:53:44.989661Z","iopub.status.idle":"2024-08-07T15:53:44.997904Z","shell.execute_reply.started":"2024-08-07T15:53:44.989637Z","shell.execute_reply":"2024-08-07T15:53:44.997219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_model(decay_steps=10, warmup_steps=10):\n    base_model = tf.keras.applications.ResNet50(\n        weights=\"imagenet\", \n        include_top=False, \n        input_shape=(512, 512, 3)\n    )\n\n    x = base_model.output\n    x = tf.keras.layers.GlobalAveragePooling2D()(x)\n    x = tf.keras.layers.Dropout(0.2)(x)\n    \n    x_bowel = tf.keras.layers.Dense(32, activation='silu')(x)\n    x_extra = tf.keras.layers.Dense(32, activation='silu')(x)\n    x_liver = tf.keras.layers.Dense(32, activation='silu')(x)\n    x_kidney = tf.keras.layers.Dense(32, activation='silu')(x)\n    x_spleen = tf.keras.layers.Dense(32, activation='silu')(x)\n\n    # Define heads\n    out_bowel = tf.keras.layers.Dense(1, activation='sigmoid', name='bowel')(x_bowel)\n    out_extra = tf.keras.layers.Dense(1, activation='sigmoid', name='extra')(x_extra)\n    out_liver = tf.keras.layers.Dense(3, activation='softmax', name='liver')(x_liver)\n    out_kidney = tf.keras.layers.Dense(3, activation='softmax', name='kidney')(x_kidney)\n    out_spleen = tf.keras.layers.Dense(3, activation='softmax', name='spleen')(x_spleen)\n\n    model = tf.keras.Model(inputs=base_model.input, outputs=[out_bowel, out_extra, out_liver, out_kidney, out_spleen])\n\n    # Cosine Decay\n    cosine_decay = tf.keras.optimizers.schedules.CosineDecay(\n        initial_learning_rate=1e-4,\n        decay_steps=decay_steps,\n        alpha=0.0\n    )\n\n    # Compile the model\n    optimizer = tf.keras.optimizers.Adam(learning_rate=cosine_decay)\n    loss = [\n        tf.keras.losses.BinaryCrossentropy(),\n        tf.keras.losses.BinaryCrossentropy(),\n        tf.keras.losses.CategoricalCrossentropy(),\n        tf.keras.losses.CategoricalCrossentropy(),\n        tf.keras.losses.CategoricalCrossentropy()\n    ]\n    \n    metrics = [\n        [tf.keras.metrics.BinaryAccuracy(name=\"bowel_binary_accuracy\"), Metrics(name=\"bowel_metrics\")],\n        [tf.keras.metrics.BinaryAccuracy(name=\"extra_binary_accuracy\"), Metrics(name=\"extra_metrics\")],\n        [tf.keras.metrics.CategoricalAccuracy(name=\"liver_cat_accuracy\"), Metrics(name=\"liver_metrics\")],\n        [tf.keras.metrics.CategoricalAccuracy(name=\"kidney_cat_accuracy\"), Metrics(name=\"kidney_metrics\")],\n        [tf.keras.metrics.CategoricalAccuracy(name=\"spleen_cat_accuracy\"), Metrics(name=\"spleen_metrics\")],\n    ]\n    \n    print(\"[INFO] Compiling the model...\")\n    model.compile(\n        optimizer=optimizer,\n        loss=loss,\n        metrics=metrics\n    )\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:53:44.999172Z","iopub.execute_input":"2024-08-07T15:53:44.999637Z","iopub.status.idle":"2024-08-07T15:53:45.014314Z","shell.execute_reply.started":"2024-08-07T15:53:44.999613Z","shell.execute_reply":"2024-08-07T15:53:45.013441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Callbacks\nearly_stopping = tf.keras.callbacks.EarlyStopping(patience=5, restore_best_weights=True)\nreduce_lr = tf.keras.callbacks.ReduceLROnPlateau(factor=0.1, patience=2)\n\nmodel_checkpoint = tf.keras.callbacks.ModelCheckpoint('best_model.keras', save_best_only=True)","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:53:45.015249Z","iopub.execute_input":"2024-08-07T15:53:45.015479Z","iopub.status.idle":"2024-08-07T15:53:45.02705Z","shell.execute_reply.started":"2024-08-07T15:53:45.015459Z","shell.execute_reply":"2024-08-07T15:53:45.026279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = create_model()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:53:45.028091Z","iopub.execute_input":"2024-08-07T15:53:45.02835Z","iopub.status.idle":"2024-08-07T15:53:50.635794Z","shell.execute_reply.started":"2024-08-07T15:53:45.028328Z","shell.execute_reply":"2024-08-07T15:53:50.634888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:53:50.636803Z","iopub.execute_input":"2024-08-07T15:53:50.637106Z","iopub.status.idle":"2024-08-07T15:53:50.885658Z","shell.execute_reply.started":"2024-08-07T15:53:50.637081Z","shell.execute_reply":"2024-08-07T15:53:50.884834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.utils.plot_model(\n    model,\n    to_file='model.png'\n)","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:53:50.886864Z","iopub.execute_input":"2024-08-07T15:53:50.887269Z","iopub.status.idle":"2024-08-07T15:53:53.877932Z","shell.execute_reply.started":"2024-08-07T15:53:50.88724Z","shell.execute_reply":"2024-08-07T15:53:53.87685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels = training_labels.set_index('patient_id')\ntraining_labels.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:53:53.879066Z","iopub.execute_input":"2024-08-07T15:53:53.879342Z","iopub.status.idle":"2024-08-07T15:53:53.893018Z","shell.execute_reply.started":"2024-08-07T15:53:53.87932Z","shell.execute_reply":"2024-08-07T15:53:53.892091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Part C","metadata":{}},{"cell_type":"code","source":"#Using histogram to get the distribution of the labels and to check if there are outliers and wrong labels\ntraining_labels.hist(figsize=(20,12),bins=2)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:53:53.894108Z","iopub.execute_input":"2024-08-07T15:53:53.894377Z","iopub.status.idle":"2024-08-07T15:53:56.145993Z","shell.execute_reply.started":"2024-08-07T15:53:53.894354Z","shell.execute_reply":"2024-08-07T15:53:56.145078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"images = []\nlabels = []\nfor i in sorted(os.listdir(train_img)):\n    folder = os.listdir(os.path.join(train_img,i))[0]\n    file = os.listdir(os.path.join(train_img,i,folder))[0]\n    images.append(cv2.imread(os.path.join(train_img,i,folder,file), cv2.IMREAD_COLOR ))\n    labels.append(np.asarray(training_labels.loc[int(i)]))\nimages = np.asarray(images)\nlabels = np.asarray(labels)","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:53:56.147053Z","iopub.execute_input":"2024-08-07T15:53:56.147311Z","iopub.status.idle":"2024-08-07T15:53:58.685218Z","shell.execute_reply.started":"2024-08-07T15:53:56.147288Z","shell.execute_reply":"2024-08-07T15:53:58.684384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(images.shape)\nprint(labels.shape)","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:53:58.686311Z","iopub.execute_input":"2024-08-07T15:53:58.686602Z","iopub.status.idle":"2024-08-07T15:53:58.691425Z","shell.execute_reply.started":"2024-08-07T15:53:58.686578Z","shell.execute_reply":"2024-08-07T15:53:58.690513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Example of images","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(30,12))\nfor i in range(20):\n    plt.subplot(4,5,i+1)\n    plt.imshow(images[i])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:53:58.69235Z","iopub.execute_input":"2024-08-07T15:53:58.69258Z","iopub.status.idle":"2024-08-07T15:54:01.604307Z","shell.execute_reply.started":"2024-08-07T15:53:58.69256Z","shell.execute_reply":"2024-08-07T15:54:01.603406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Augumenting the training dataset\n# Create an instance of ImageDataGenerator\ndatagen = tf.keras.preprocessing.image.ImageDataGenerator(\n    rotation_range=10,   # Randomly rotate images by up to 20 degrees\n    width_shift_range=0.5,   # Randomly shift images horizontally by up to 5% of the width\n    height_shift_range=0.5,  # Randomly shift images vertically by up to 5% of the height\n    shear_range=0,   # Shear transformations\n    zoom_range=0.1,    # Randomly zoom in on images\n    horizontal_flip=True,   # Randomly flip images horizontally\n    fill_mode='nearest'     # How to fill in newly created pixels after rotation/shifts\n)\n\n# Fit the data generator on your training data\ndatagen.fit(images)\n\n# Generate augmented data\naugmented_data = datagen.flow(images, labels, batch_size=32)","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:54:01.605416Z","iopub.execute_input":"2024-08-07T15:54:01.605701Z","iopub.status.idle":"2024-08-07T15:54:02.247542Z","shell.execute_reply.started":"2024-08-07T15:54:01.605677Z","shell.execute_reply":"2024-08-07T15:54:02.246502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Splittig the training dataset into training set and validation set\nfrom sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:54:02.248939Z","iopub.execute_input":"2024-08-07T15:54:02.249352Z","iopub.status.idle":"2024-08-07T15:54:02.253821Z","shell.execute_reply.started":"2024-08-07T15:54:02.249318Z","shell.execute_reply":"2024-08-07T15:54:02.252964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_val, y_train, y_val = train_test_split(images, labels, test_size=0.25)\n\n#Printing the shapes of the split datasets\nprint(\"Shape of Training Images:\", X_train.shape)\nprint(\"Shape of Training Labels:\", y_train.shape)\nprint(\"Shape of Validation Images:\", X_val.shape)\nprint(\"Shape of Validation Labels:\", y_val.shape)\n","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:54:02.254929Z","iopub.execute_input":"2024-08-07T15:54:02.255287Z","iopub.status.idle":"2024-08-07T15:54:02.321625Z","shell.execute_reply.started":"2024-08-07T15:54:02.255254Z","shell.execute_reply":"2024-08-07T15:54:02.32067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert augmented data to an array\naugmented_images, augmented_labels = next(augmented_data)\n\n# Print the shapes of the augmented datasets\nprint(\"Shape of Augmented Images:\", augmented_images.shape)\nprint(\"Shape of Augmented Labels:\", augmented_labels.shape)","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:54:02.323011Z","iopub.execute_input":"2024-08-07T15:54:02.323679Z","iopub.status.idle":"2024-08-07T15:54:03.941127Z","shell.execute_reply.started":"2024-08-07T15:54:02.323639Z","shell.execute_reply":"2024-08-07T15:54:03.940173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(40,20))\nfor i in range(5):\n    plt.subplot(1,5,i+1)\n    plt.imshow(augmented_images[i])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:54:03.942383Z","iopub.execute_input":"2024-08-07T15:54:03.944363Z","iopub.status.idle":"2024-08-07T15:54:05.079801Z","shell.execute_reply.started":"2024-08-07T15:54:03.944334Z","shell.execute_reply":"2024-08-07T15:54:05.078886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(augmented_labels)","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:54:05.080976Z","iopub.execute_input":"2024-08-07T15:54:05.081332Z","iopub.status.idle":"2024-08-07T15:54:05.102489Z","shell.execute_reply.started":"2024-08-07T15:54:05.081303Z","shell.execute_reply":"2024-08-07T15:54:05.101612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Part D\n## Training the model with the training dataset","metadata":{}},{"cell_type":"code","source":"pd.DataFrame(y_train)","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:54:05.103627Z","iopub.execute_input":"2024-08-07T15:54:05.103963Z","iopub.status.idle":"2024-08-07T15:54:05.119926Z","shell.execute_reply.started":"2024-08-07T15:54:05.10393Z","shell.execute_reply":"2024-08-07T15:54:05.119058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(y_val)","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:54:05.121329Z","iopub.execute_input":"2024-08-07T15:54:05.12164Z","iopub.status.idle":"2024-08-07T15:54:05.137453Z","shell.execute_reply.started":"2024-08-07T15:54:05.121604Z","shell.execute_reply":"2024-08-07T15:54:05.13661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bowel_labels = labels[:,2]\nextravasation_labels = labels[:,4]\nkidney_labels = labels[:,4:7]\nliver_labels = labels[:,7:10]\nspleen_labels = labels[:,10:13]\nany_labels = labels[:,-1]","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:54:05.144393Z","iopub.execute_input":"2024-08-07T15:54:05.144886Z","iopub.status.idle":"2024-08-07T15:54:05.149229Z","shell.execute_reply.started":"2024-08-07T15:54:05.144862Z","shell.execute_reply":"2024-08-07T15:54:05.148381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bowel_val = y_val[:,2]\nextravasation_val = y_val[:,4]\nkidney_val = y_val[:,4:7]\nliver_val = y_val[:,7:10]\nspleen_val = y_val[:,10:13]\nany_val = y_val[:,-1]","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:54:05.150221Z","iopub.execute_input":"2024-08-07T15:54:05.150496Z","iopub.status.idle":"2024-08-07T15:54:05.158017Z","shell.execute_reply.started":"2024-08-07T15:54:05.150474Z","shell.execute_reply":"2024-08-07T15:54:05.157212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bowel_train = y_train[:,2]\nextravasation_train = y_train[:,4]\nkidney_train = y_train[:,4:7]\nliver_train = y_train[:,7:10]\nspleen_train = y_train[:,10:13]\nany_train = y_train[:,-1]","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:54:05.159004Z","iopub.execute_input":"2024-08-07T15:54:05.159327Z","iopub.status.idle":"2024-08-07T15:54:05.167917Z","shell.execute_reply.started":"2024-08-07T15:54:05.159304Z","shell.execute_reply":"2024-08-07T15:54:05.167088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bowel_test = augmented_labels[:,2]\nextravasation_test = augmented_labels[:,4]\nkidney_test = augmented_labels[:,4:7]\nliver_test = augmented_labels[:,7:10]\nspleen_test = augmented_labels[:,10:13]\nany_test = augmented_labels[:,-1]","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:54:05.168983Z","iopub.execute_input":"2024-08-07T15:54:05.169553Z","iopub.status.idle":"2024-08-07T15:54:05.177957Z","shell.execute_reply.started":"2024-08-07T15:54:05.169522Z","shell.execute_reply":"2024-08-07T15:54:05.177089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size = 20\nnum_epoch = 20\nhistory = model.fit(x = X_train,\n                    y =[bowel_train,extravasation_train,kidney_train,liver_train,spleen_train],\n                    validation_data = (X_val, [bowel_val, extravasation_val, kidney_val, liver_val, spleen_val]),\n                    batch_size=batch_size, \n                    epochs = num_epoch, \n                    verbose = 1,\n                    callbacks=[early_stopping, reduce_lr, model_checkpoint]\n                   )","metadata":{"papermill":{"duration":1176.379334,"end_time":"2022-02-01T14:07:30.902597","exception":false,"start_time":"2022-02-01T13:47:54.523263","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-08-07T15:54:05.179226Z","iopub.execute_input":"2024-08-07T15:54:05.179594Z","iopub.status.idle":"2024-08-07T15:56:59.780906Z","shell.execute_reply.started":"2024-08-07T15:54:05.179565Z","shell.execute_reply":"2024-08-07T15:56:59.780048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history.history.keys()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:56:59.782547Z","iopub.execute_input":"2024-08-07T15:56:59.782849Z","iopub.status.idle":"2024-08-07T15:56:59.788744Z","shell.execute_reply.started":"2024-08-07T15:56:59.782823Z","shell.execute_reply":"2024-08-07T15:56:59.787894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_acc = ['val_bowel_bowel_binary_accuracy', 'val_extra_extra_binary_accuracy', 'val_liver_liver_cat_accuracy', 'val_kidney_kidney_cat_accuracy', 'val_spleen_spleen_cat_accuracy']","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:56:59.790138Z","iopub.execute_input":"2024-08-07T15:56:59.790561Z","iopub.status.idle":"2024-08-07T15:56:59.800006Z","shell.execute_reply.started":"2024-08-07T15:56:59.790529Z","shell.execute_reply":"2024-08-07T15:56:59.799181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i,j in enumerate(val_acc):\n    print(j, np.asarray(history.history[val_acc[i]])[-1].round(2))\n#np.asarray(history.history['val_bowel_bowel_binary_accuracy'])[-1].round(2)","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:56:59.801164Z","iopub.execute_input":"2024-08-07T15:56:59.802472Z","iopub.status.idle":"2024-08-07T15:56:59.811527Z","shell.execute_reply.started":"2024-08-07T15:56:59.802447Z","shell.execute_reply":"2024-08-07T15:56:59.810295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in history.history.keys():\n    if i.endswith(\"accuracy\") and not i ==\"val_accuracy\":\n        plt.plot(history.history[i], label=i)\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Loss\")\nplt.legend(loc=(1.05,0.0))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:56:59.812505Z","iopub.execute_input":"2024-08-07T15:56:59.812787Z","iopub.status.idle":"2024-08-07T15:57:00.147201Z","shell.execute_reply.started":"2024-08-07T15:56:59.812763Z","shell.execute_reply":"2024-08-07T15:57:00.146312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(history.history[\"val_loss\"], label=\"val_loss\")\nplt.plot(history.history[\"loss\"], label=\"loss\")\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:57:00.1483Z","iopub.execute_input":"2024-08-07T15:57:00.148575Z","iopub.status.idle":"2024-08-07T15:57:00.323447Z","shell.execute_reply.started":"2024-08-07T15:57:00.148552Z","shell.execute_reply":"2024-08-07T15:57:00.32258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Testing the model using the Augmented data as test data","metadata":{}},{"cell_type":"code","source":"test_results = model.evaluate(\n    augmented_images, [bowel_test, extravasation_test, kidney_test, liver_test, spleen_test]\n) \n\n# Print the structure of test_results\nprint(\"Test Results Structure:\")\nprint(test_results)","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:57:00.324575Z","iopub.execute_input":"2024-08-07T15:57:00.325285Z","iopub.status.idle":"2024-08-07T15:57:16.570955Z","shell.execute_reply.started":"2024-08-07T15:57:00.325259Z","shell.execute_reply":"2024-08-07T15:57:16.570098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output_names = [\"bowel\", \"extra\", \"liver\", \"kidney\", \"spleen\"]","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:57:16.572116Z","iopub.execute_input":"2024-08-07T15:57:16.572422Z","iopub.status.idle":"2024-08-07T15:57:16.576731Z","shell.execute_reply.started":"2024-08-07T15:57:16.57239Z","shell.execute_reply":"2024-08-07T15:57:16.575857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_prob = model.predict(augmented_images)","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:57:16.578247Z","iopub.execute_input":"2024-08-07T15:57:16.57851Z","iopub.status.idle":"2024-08-07T15:57:20.830911Z","shell.execute_reply.started":"2024-08-07T15:57:16.578487Z","shell.execute_reply":"2024-08-07T15:57:20.83001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Model Output Names:\", model.output_names)  # Check names in model\nprint(\"y_pred_prob:\", y_pred_prob)\nprint(len(y_pred_prob))  # Print predictions and their shape","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:57:20.832265Z","iopub.execute_input":"2024-08-07T15:57:20.832529Z","iopub.status.idle":"2024-08-07T15:57:20.842299Z","shell.execute_reply.started":"2024-08-07T15:57:20.832506Z","shell.execute_reply":"2024-08-07T15:57:20.84135Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_true_dict = {\n    \"bowel\": bowel_test, \n    \"extra\": extravasation_test,\n    \"liver\": liver_test,\n    \"kidney\": kidney_test,\n    \"spleen\": spleen_test,\n}\n\n\n# Generate and plot confusion matrices for each output\nfor i, output_name in enumerate(output_names):\n    y_true = y_true_dict[output_name]\n    num_classes = 1 if i < 2 else 3  # Binary: 1 class, Multiclass: 3 classes\n    \n    # Ensure y_pred_prob is an array or similar structure\n    if not isinstance(y_pred_prob, (np.ndarray, list, tuple)):\n        y_pred_prob = np.array([y_pred_prob])\n        \n    # If predictions are a tuple of 5 elements\n    if len(y_pred_prob) == 5:\n        # Select the relevant element of the tuple based on the output name\n        y_pred_prob = y_pred_prob[i]\n        \n    # Make sure the predictions are 2D arrays with samples in the first dimension\n    if y_pred_prob.ndim == 1:\n        y_pred_prob = y_pred_prob.reshape(-1, 1)  # Reshape to 2D if necessary\n    \n    if num_classes == 1:\n        # Binary case (but with an extra dimension)\n        y_pred = (y_pred_prob > 0.5).astype(int).flatten()\n        labels = [\"Negative\", \"Positive\"]\n    else:\n        # Multiclass case \n        y_pred = np.argmax(y_pred_prob, axis=1)\n        labels = [f\"Class {i}\" for i in range(y_pred_prob.shape[1])]  \n\n    # Ensure lengths match before calculating the confusion matrix\n    min_len = min(len(y_true), len(y_pred))\n    y_true = y_true[:min_len]\n    y_pred = y_pred[:min_len]\n\n\n    # Ensure both y_true and y_pred are interpreted as multiclass if y_true.ndim > 1:\n    if y_true.ndim > 1:\n        y_true = np.argmax(y_true, axis=1)\n    if y_pred.ndim > 1:\n        y_pred = np.argmax(y_pred, axis=1)\n\n    cm = confusion_matrix(y_true, y_pred)\n    disp = ConfusionMatrixDisplay(confusion_matrix=cm)\n    disp.plot(cmap=plt.cm.Blues)\n\n    # Update ticks and labels directly on the Axes object\n    ax = disp.ax_ \n    ticks = np.arange(len(labels))\n    ax.set_xticks(ticks)\n    ax.set_yticks(ticks)\n    ax.set_xticklabels(labels)\n    ax.set_yticklabels(labels)\n    \n    plt.title(f\"Confusion Matrix - {output_name}\")\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:57:20.843513Z","iopub.execute_input":"2024-08-07T15:57:20.843775Z","iopub.status.idle":"2024-08-07T15:57:21.90976Z","shell.execute_reply.started":"2024-08-07T15:57:20.843752Z","shell.execute_reply":"2024-08-07T15:57:21.908648Z"},"trusted":true},"execution_count":null,"outputs":[]}]}