{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":52254,"databundleVersionId":8756537,"sourceType":"competition"},{"sourceId":6211844,"sourceType":"datasetVersion","datasetId":3567114}],"dockerImageVersionId":30699,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# <center>Medical image ...</center>\n\n","metadata":{"papermill":{"duration":0.025483,"end_time":"2022-02-01T10:21:43.84374","exception":false,"start_time":"2022-02-01T10:21:43.818257","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"## To commence this project, the neccesary libraries need to be installed.\n## We would be using the **TensorFlow** framework and the pretrained **EfficientNetB1**\n","metadata":{}},{"cell_type":"code","source":"# Imporitng libraries\nimport pandas as pd\nimport os\nimport random\nimport numpy as np\nimport tensorflow as tf\nimport cv2\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom IPython.display import clear_output\nfrom sklearn.metrics import confusion_matrix, ConfusionMatrixDisplay\nfrom sklearn.model_selection import StratifiedKFold\nfrom tensorflow.keras import backend as K\n\nrandom.seed(42)","metadata":{"id":"fUPKjZaaJHkM","papermill":{"duration":5.021373,"end_time":"2022-02-01T10:21:48.88737","exception":false,"start_time":"2022-02-01T10:21:43.865997","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-08-07T15:30:23.165189Z","iopub.execute_input":"2024-08-07T15:30:23.165457Z","iopub.status.idle":"2024-08-07T15:30:37.392606Z","shell.execute_reply.started":"2024-08-07T15:30:23.165432Z","shell.execute_reply":"2024-08-07T15:30:37.391733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Loading the training dataset\ntrain_img = \"/kaggle/input/rsna-atd-512x512-png-v2-dataset/train_images\"","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:30:37.394339Z","iopub.execute_input":"2024-08-07T15:30:37.394854Z","iopub.status.idle":"2024-08-07T15:30:37.399227Z","shell.execute_reply.started":"2024-08-07T15:30:37.394828Z","shell.execute_reply":"2024-08-07T15:30:37.398182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in os.listdir(train_img):\n    for j in os.listdir(os.path.join(train_img,i)):\n        print(len(os.listdir(os.path.join(train_img,i,j))))","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:30:37.400493Z","iopub.execute_input":"2024-08-07T15:30:37.400849Z","iopub.status.idle":"2024-08-07T15:30:42.127731Z","shell.execute_reply.started":"2024-08-07T15:30:37.400817Z","shell.execute_reply":"2024-08-07T15:30:42.126781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Part A\n## Import and explore the data","metadata":{"papermill":{"duration":0.02193,"end_time":"2022-02-01T10:21:48.93356","exception":false,"start_time":"2022-02-01T10:21:48.91163","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Making a list containing all unique classes in the training set\n\n# Reading the training labels\ntraining_labels = pd.read_csv(\"/kaggle/input/rsna-2023-abdominal-trauma-detection/train_2024.csv\")","metadata":{"id":"PBlH8ae3LbG0","papermill":{"duration":0.069238,"end_time":"2022-02-01T10:21:49.024476","exception":false,"start_time":"2022-02-01T10:21:48.955238","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-08-07T15:30:42.130448Z","iopub.execute_input":"2024-08-07T15:30:42.13091Z","iopub.status.idle":"2024-08-07T15:30:42.148843Z","shell.execute_reply.started":"2024-08-07T15:30:42.130876Z","shell.execute_reply":"2024-08-07T15:30:42.14801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Viewing the dataset in a structured format\ntraining_labels","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:30:42.14985Z","iopub.execute_input":"2024-08-07T15:30:42.15009Z","iopub.status.idle":"2024-08-07T15:30:42.1779Z","shell.execute_reply.started":"2024-08-07T15:30:42.150069Z","shell.execute_reply":"2024-08-07T15:30:42.177151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Part B\n## In the next few cells, we chose to explore the dataframe to check for missing values","metadata":{}},{"cell_type":"code","source":"training_labels.columns","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:30:42.178955Z","iopub.execute_input":"2024-08-07T15:30:42.179259Z","iopub.status.idle":"2024-08-07T15:30:42.185052Z","shell.execute_reply.started":"2024-08-07T15:30:42.179237Z","shell.execute_reply":"2024-08-07T15:30:42.184187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels['patient_id'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:30:42.186114Z","iopub.execute_input":"2024-08-07T15:30:42.186597Z","iopub.status.idle":"2024-08-07T15:30:42.202314Z","shell.execute_reply.started":"2024-08-07T15:30:42.186574Z","shell.execute_reply":"2024-08-07T15:30:42.201433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels['bowel_healthy'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:30:42.203232Z","iopub.execute_input":"2024-08-07T15:30:42.203485Z","iopub.status.idle":"2024-08-07T15:30:42.210118Z","shell.execute_reply.started":"2024-08-07T15:30:42.203464Z","shell.execute_reply":"2024-08-07T15:30:42.209202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels['bowel_injury'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:30:42.211274Z","iopub.execute_input":"2024-08-07T15:30:42.211659Z","iopub.status.idle":"2024-08-07T15:30:42.222034Z","shell.execute_reply.started":"2024-08-07T15:30:42.211636Z","shell.execute_reply":"2024-08-07T15:30:42.221172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels['extravasation_healthy'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:30:42.225461Z","iopub.execute_input":"2024-08-07T15:30:42.225991Z","iopub.status.idle":"2024-08-07T15:30:42.232759Z","shell.execute_reply.started":"2024-08-07T15:30:42.225966Z","shell.execute_reply":"2024-08-07T15:30:42.231979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels['extravasation_injury'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:30:42.233618Z","iopub.execute_input":"2024-08-07T15:30:42.234358Z","iopub.status.idle":"2024-08-07T15:30:42.244471Z","shell.execute_reply.started":"2024-08-07T15:30:42.234335Z","shell.execute_reply":"2024-08-07T15:30:42.243708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels['kidney_healthy'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:30:42.245482Z","iopub.execute_input":"2024-08-07T15:30:42.245786Z","iopub.status.idle":"2024-08-07T15:30:42.255643Z","shell.execute_reply.started":"2024-08-07T15:30:42.245756Z","shell.execute_reply":"2024-08-07T15:30:42.254775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels['kidney_low'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:30:42.256719Z","iopub.execute_input":"2024-08-07T15:30:42.256982Z","iopub.status.idle":"2024-08-07T15:30:42.266308Z","shell.execute_reply.started":"2024-08-07T15:30:42.25696Z","shell.execute_reply":"2024-08-07T15:30:42.265447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels['kidney_high'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:30:42.267234Z","iopub.execute_input":"2024-08-07T15:30:42.267494Z","iopub.status.idle":"2024-08-07T15:30:42.27822Z","shell.execute_reply.started":"2024-08-07T15:30:42.267472Z","shell.execute_reply":"2024-08-07T15:30:42.277466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels['liver_healthy'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:30:42.279315Z","iopub.execute_input":"2024-08-07T15:30:42.279615Z","iopub.status.idle":"2024-08-07T15:30:42.289872Z","shell.execute_reply.started":"2024-08-07T15:30:42.279591Z","shell.execute_reply":"2024-08-07T15:30:42.289017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels['liver_low'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:30:42.290878Z","iopub.execute_input":"2024-08-07T15:30:42.291109Z","iopub.status.idle":"2024-08-07T15:30:42.302614Z","shell.execute_reply.started":"2024-08-07T15:30:42.291087Z","shell.execute_reply":"2024-08-07T15:30:42.301724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels['liver_high'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:30:42.303731Z","iopub.execute_input":"2024-08-07T15:30:42.304048Z","iopub.status.idle":"2024-08-07T15:30:42.313796Z","shell.execute_reply.started":"2024-08-07T15:30:42.304019Z","shell.execute_reply":"2024-08-07T15:30:42.312992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels['spleen_healthy'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:30:42.314928Z","iopub.execute_input":"2024-08-07T15:30:42.315289Z","iopub.status.idle":"2024-08-07T15:30:42.326332Z","shell.execute_reply.started":"2024-08-07T15:30:42.315255Z","shell.execute_reply":"2024-08-07T15:30:42.325359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels['spleen_low'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:30:42.327302Z","iopub.execute_input":"2024-08-07T15:30:42.327549Z","iopub.status.idle":"2024-08-07T15:30:42.338125Z","shell.execute_reply.started":"2024-08-07T15:30:42.327528Z","shell.execute_reply":"2024-08-07T15:30:42.337287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels['spleen_high'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:30:42.339184Z","iopub.execute_input":"2024-08-07T15:30:42.339461Z","iopub.status.idle":"2024-08-07T15:30:42.350313Z","shell.execute_reply.started":"2024-08-07T15:30:42.339439Z","shell.execute_reply":"2024-08-07T15:30:42.349219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels['any_injury'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:30:42.351359Z","iopub.execute_input":"2024-08-07T15:30:42.351625Z","iopub.status.idle":"2024-08-07T15:30:42.361969Z","shell.execute_reply.started":"2024-08-07T15:30:42.351603Z","shell.execute_reply":"2024-08-07T15:30:42.361121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Metrics(tf.keras.metrics.Metric):\n    def __init__(self, name='metrics', **kwargs):\n        super(Metrics, self).__init__(name=name, **kwargs)\n        self.precision = tf.keras.metrics.Precision()\n        self.recall = tf.keras.metrics.Recall()\n\n    def update_state(self, y_true, y_pred, sample_weight=None):\n        self.precision.update_state(y_true, y_pred, sample_weight)\n        self.recall.update_state(y_true, y_pred, sample_weight)\n\n    def result(self):\n        return {\n            \"precision\": self.precision.result(),\n            \"recall\": self.recall.result(),\n            \"f1_score\": 2 * ((self.precision.result() * self.recall.result()) / (self.precision.result() + self.recall.result() + K.epsilon()))\n        }\n\n    def reset_states(self):\n        self.precision.reset_states()\n        self.recall.reset_states()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:30:42.363053Z","iopub.execute_input":"2024-08-07T15:30:42.363382Z","iopub.status.idle":"2024-08-07T15:30:42.371945Z","shell.execute_reply.started":"2024-08-07T15:30:42.363359Z","shell.execute_reply":"2024-08-07T15:30:42.37097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_model(decay_steps=10, warmup_steps=10):\n    base_model = tf.keras.applications.MobileNetV2(\n        weights=\"imagenet\", \n        include_top=False, \n        input_shape=(512, 512, 3)\n    )\n\n    x = base_model.output\n    x = tf.keras.layers.GlobalAveragePooling2D()(x)\n    x = tf.keras.layers.Dropout(0.2)(x)\n    \n    x_bowel = tf.keras.layers.Dense(32, activation='silu')(x)\n    x_extra = tf.keras.layers.Dense(32, activation='silu')(x)\n    x_liver = tf.keras.layers.Dense(32, activation='silu')(x)\n    x_kidney = tf.keras.layers.Dense(32, activation='silu')(x)\n    x_spleen = tf.keras.layers.Dense(32, activation='silu')(x)\n\n    # Define heads\n    out_bowel = tf.keras.layers.Dense(1, activation='sigmoid', name='bowel')(x_bowel)\n    out_extra = tf.keras.layers.Dense(1, activation='sigmoid', name='extra')(x_extra)\n    out_liver = tf.keras.layers.Dense(3, activation='softmax', name='liver')(x_liver)\n    out_kidney = tf.keras.layers.Dense(3, activation='softmax', name='kidney')(x_kidney)\n    out_spleen = tf.keras.layers.Dense(3, activation='softmax', name='spleen')(x_spleen)\n\n    model = tf.keras.Model(inputs=base_model.input, outputs=[out_bowel, out_extra, out_liver, out_kidney, out_spleen])\n\n    # Cosine Decay\n    cosine_decay = tf.keras.optimizers.schedules.CosineDecay(\n        initial_learning_rate=1e-4,\n        decay_steps=decay_steps,\n        alpha=0.0\n    )\n\n    # Compile the model\n    optimizer = tf.keras.optimizers.Adam(learning_rate=cosine_decay)\n    loss = [\n        tf.keras.losses.BinaryCrossentropy(),\n        tf.keras.losses.BinaryCrossentropy(),\n        tf.keras.losses.CategoricalCrossentropy(),\n        tf.keras.losses.CategoricalCrossentropy(),\n        tf.keras.losses.CategoricalCrossentropy()\n    ]\n    \n    metrics = [\n        [tf.keras.metrics.BinaryAccuracy(name=\"bowel_binary_accuracy\"), Metrics(name=\"bowel_metrics\")],\n        [tf.keras.metrics.BinaryAccuracy(name=\"extra_binary_accuracy\"), Metrics(name=\"extra_metrics\")],\n        [tf.keras.metrics.CategoricalAccuracy(name=\"liver_cat_accuracy\"), Metrics(name=\"liver_metrics\")],\n        [tf.keras.metrics.CategoricalAccuracy(name=\"kidney_cat_accuracy\"), Metrics(name=\"kidney_metrics\")],\n        [tf.keras.metrics.CategoricalAccuracy(name=\"spleen_cat_accuracy\"), Metrics(name=\"spleen_metrics\")],\n    ]\n    \n    print(\"[INFO] Compiling the model...\")\n    model.compile(\n        optimizer=optimizer,\n        loss=loss,\n        metrics=metrics\n    )\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:30:42.373099Z","iopub.execute_input":"2024-08-07T15:30:42.3734Z","iopub.status.idle":"2024-08-07T15:30:42.38825Z","shell.execute_reply.started":"2024-08-07T15:30:42.373374Z","shell.execute_reply":"2024-08-07T15:30:42.387317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Callbacks\nearly_stopping = tf.keras.callbacks.EarlyStopping(patience=5, restore_best_weights=True)\nreduce_lr = tf.keras.callbacks.ReduceLROnPlateau(factor=0.1, patience=2)\n\nmodel_checkpoint = tf.keras.callbacks.ModelCheckpoint('best_model.keras', save_best_only=True)","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:30:42.389324Z","iopub.execute_input":"2024-08-07T15:30:42.389826Z","iopub.status.idle":"2024-08-07T15:30:42.401437Z","shell.execute_reply.started":"2024-08-07T15:30:42.389802Z","shell.execute_reply":"2024-08-07T15:30:42.400691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = create_model()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:30:42.402525Z","iopub.execute_input":"2024-08-07T15:30:42.402799Z","iopub.status.idle":"2024-08-07T15:30:45.27381Z","shell.execute_reply.started":"2024-08-07T15:30:42.402763Z","shell.execute_reply":"2024-08-07T15:30:45.273015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:30:45.275405Z","iopub.execute_input":"2024-08-07T15:30:45.275766Z","iopub.status.idle":"2024-08-07T15:30:45.52007Z","shell.execute_reply.started":"2024-08-07T15:30:45.275735Z","shell.execute_reply":"2024-08-07T15:30:45.519197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.utils.plot_model(\n    model,\n    to_file='model.png'\n)","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:30:45.52117Z","iopub.execute_input":"2024-08-07T15:30:45.521463Z","iopub.status.idle":"2024-08-07T15:30:48.691763Z","shell.execute_reply.started":"2024-08-07T15:30:45.521439Z","shell.execute_reply":"2024-08-07T15:30:48.690466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels = training_labels.set_index('patient_id')\ntraining_labels.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:30:48.699077Z","iopub.execute_input":"2024-08-07T15:30:48.69982Z","iopub.status.idle":"2024-08-07T15:30:48.715497Z","shell.execute_reply.started":"2024-08-07T15:30:48.699787Z","shell.execute_reply":"2024-08-07T15:30:48.714615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Part C","metadata":{}},{"cell_type":"code","source":"#Using histogram to get the distribution of the labels and to check if there are outliers and wrong labels\ntraining_labels.hist(figsize=(20,12),bins=2)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:30:48.71679Z","iopub.execute_input":"2024-08-07T15:30:48.717061Z","iopub.status.idle":"2024-08-07T15:30:51.075237Z","shell.execute_reply.started":"2024-08-07T15:30:48.717039Z","shell.execute_reply":"2024-08-07T15:30:51.074301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"images = []\nlabels = []\nfor i in sorted(os.listdir(train_img)):\n    folder = os.listdir(os.path.join(train_img,i))[0]\n    file = os.listdir(os.path.join(train_img,i,folder))[0]\n    images.append(cv2.imread(os.path.join(train_img,i,folder,file), cv2.IMREAD_COLOR ))\n    labels.append(np.asarray(training_labels.loc[int(i)]))\nimages = np.asarray(images)\nlabels = np.asarray(labels)","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:30:51.076618Z","iopub.execute_input":"2024-08-07T15:30:51.076972Z","iopub.status.idle":"2024-08-07T15:30:53.900412Z","shell.execute_reply.started":"2024-08-07T15:30:51.076942Z","shell.execute_reply":"2024-08-07T15:30:53.899592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(images.shape)\nprint(labels.shape)","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:30:53.901488Z","iopub.execute_input":"2024-08-07T15:30:53.901757Z","iopub.status.idle":"2024-08-07T15:30:53.906798Z","shell.execute_reply.started":"2024-08-07T15:30:53.901734Z","shell.execute_reply":"2024-08-07T15:30:53.905869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Example of images","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(30,12))\nfor i in range(20):\n    plt.subplot(4,5,i+1)\n    plt.imshow(images[i])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:30:53.907875Z","iopub.execute_input":"2024-08-07T15:30:53.908179Z","iopub.status.idle":"2024-08-07T15:30:56.972996Z","shell.execute_reply.started":"2024-08-07T15:30:53.908127Z","shell.execute_reply":"2024-08-07T15:30:56.972129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Augumenting the training dataset\n# Create an instance of ImageDataGenerator\ndatagen = tf.keras.preprocessing.image.ImageDataGenerator(\n    rotation_range=10,   # Randomly rotate images by up to 20 degrees\n    width_shift_range=0.5,   # Randomly shift images horizontally by up to 5% of the width\n    height_shift_range=0.5,  # Randomly shift images vertically by up to 5% of the height\n    shear_range=0,   # Shear transformations\n    zoom_range=0.1,    # Randomly zoom in on images\n    horizontal_flip=True,   # Randomly flip images horizontally\n    fill_mode='nearest'     # How to fill in newly created pixels after rotation/shifts\n)\n\n# Fit the data generator on your training data\ndatagen.fit(images)\n\n# Generate augmented data\naugmented_data = datagen.flow(images, labels, batch_size=32)","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:30:56.974091Z","iopub.execute_input":"2024-08-07T15:30:56.974378Z","iopub.status.idle":"2024-08-07T15:30:57.623867Z","shell.execute_reply.started":"2024-08-07T15:30:56.974354Z","shell.execute_reply":"2024-08-07T15:30:57.62304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Splittig the training dataset into training set and validation set\nfrom sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:30:57.624994Z","iopub.execute_input":"2024-08-07T15:30:57.625304Z","iopub.status.idle":"2024-08-07T15:30:57.629704Z","shell.execute_reply.started":"2024-08-07T15:30:57.62528Z","shell.execute_reply":"2024-08-07T15:30:57.628684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_val, y_train, y_val = train_test_split(images, labels, test_size=0.25)\n\n#Printing the shapes of the split datasets\nprint(\"Shape of Training Images:\", X_train.shape)\nprint(\"Shape of Training Labels:\", y_train.shape)\nprint(\"Shape of Validation Images:\", X_val.shape)\nprint(\"Shape of Validation Labels:\", y_val.shape)\n","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:30:57.63088Z","iopub.execute_input":"2024-08-07T15:30:57.631162Z","iopub.status.idle":"2024-08-07T15:30:57.702682Z","shell.execute_reply.started":"2024-08-07T15:30:57.631121Z","shell.execute_reply":"2024-08-07T15:30:57.701682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert augmented data to an array\naugmented_images, augmented_labels = next(augmented_data)\n\n# Print the shapes of the augmented datasets\nprint(\"Shape of Augmented Images:\", augmented_images.shape)\nprint(\"Shape of Augmented Labels:\", augmented_labels.shape)","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:30:57.703808Z","iopub.execute_input":"2024-08-07T15:30:57.704074Z","iopub.status.idle":"2024-08-07T15:30:59.816747Z","shell.execute_reply.started":"2024-08-07T15:30:57.704051Z","shell.execute_reply":"2024-08-07T15:30:59.815813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(40,20))\nfor i in range(5):\n    plt.subplot(1,5,i+1)\n    plt.imshow(augmented_images[i])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:30:59.817936Z","iopub.execute_input":"2024-08-07T15:30:59.81825Z","iopub.status.idle":"2024-08-07T15:31:00.98124Z","shell.execute_reply.started":"2024-08-07T15:30:59.818224Z","shell.execute_reply":"2024-08-07T15:31:00.980346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(augmented_labels)","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:31:00.982409Z","iopub.execute_input":"2024-08-07T15:31:00.982798Z","iopub.status.idle":"2024-08-07T15:31:01.002217Z","shell.execute_reply.started":"2024-08-07T15:31:00.982763Z","shell.execute_reply":"2024-08-07T15:31:01.001285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Part D\n## Training the model with the training dataset","metadata":{}},{"cell_type":"code","source":"pd.DataFrame(y_train)","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:31:01.003325Z","iopub.execute_input":"2024-08-07T15:31:01.003614Z","iopub.status.idle":"2024-08-07T15:31:01.022115Z","shell.execute_reply.started":"2024-08-07T15:31:01.003592Z","shell.execute_reply":"2024-08-07T15:31:01.021193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(y_val)","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:31:01.023214Z","iopub.execute_input":"2024-08-07T15:31:01.023701Z","iopub.status.idle":"2024-08-07T15:31:01.040526Z","shell.execute_reply.started":"2024-08-07T15:31:01.02367Z","shell.execute_reply":"2024-08-07T15:31:01.039669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bowel_labels = labels[:,2]\nextravasation_labels = labels[:,4]\nkidney_labels = labels[:,4:7]\nliver_labels = labels[:,7:10]\nspleen_labels = labels[:,10:13]\nany_labels = labels[:,-1]","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:31:01.041467Z","iopub.execute_input":"2024-08-07T15:31:01.041739Z","iopub.status.idle":"2024-08-07T15:31:01.049614Z","shell.execute_reply.started":"2024-08-07T15:31:01.041716Z","shell.execute_reply":"2024-08-07T15:31:01.048679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bowel_val = y_val[:,2]\nextravasation_val = y_val[:,4]\nkidney_val = y_val[:,4:7]\nliver_val = y_val[:,7:10]\nspleen_val = y_val[:,10:13]\nany_val = y_val[:,-1]","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:31:01.050561Z","iopub.execute_input":"2024-08-07T15:31:01.050807Z","iopub.status.idle":"2024-08-07T15:31:01.060298Z","shell.execute_reply.started":"2024-08-07T15:31:01.050785Z","shell.execute_reply":"2024-08-07T15:31:01.059389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bowel_train = y_train[:,2]\nextravasation_train = y_train[:,4]\nkidney_train = y_train[:,4:7]\nliver_train = y_train[:,7:10]\nspleen_train = y_train[:,10:13]\nany_train = y_train[:,-1]","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:31:01.06137Z","iopub.execute_input":"2024-08-07T15:31:01.061943Z","iopub.status.idle":"2024-08-07T15:31:01.070757Z","shell.execute_reply.started":"2024-08-07T15:31:01.061914Z","shell.execute_reply":"2024-08-07T15:31:01.069872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bowel_test = augmented_labels[:,2]\nextravasation_test = augmented_labels[:,4]\nkidney_test = augmented_labels[:,4:7]\nliver_test = augmented_labels[:,7:10]\nspleen_test = augmented_labels[:,10:13]\nany_test = augmented_labels[:,-1]","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:31:01.071819Z","iopub.execute_input":"2024-08-07T15:31:01.072204Z","iopub.status.idle":"2024-08-07T15:31:01.081358Z","shell.execute_reply.started":"2024-08-07T15:31:01.072174Z","shell.execute_reply":"2024-08-07T15:31:01.080536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size = 20\nnum_epoch = 20\nhistory = model.fit(x = X_train,\n                    y =[bowel_train,extravasation_train,kidney_train,liver_train,spleen_train],\n                    validation_data = (X_val, [bowel_val, extravasation_val, kidney_val, liver_val, spleen_val]),\n                    batch_size=batch_size, \n                    epochs = num_epoch, \n                    verbose = 1,\n                    callbacks=[early_stopping, reduce_lr, model_checkpoint]\n                   )","metadata":{"papermill":{"duration":1176.379334,"end_time":"2022-02-01T14:07:30.902597","exception":false,"start_time":"2022-02-01T13:47:54.523263","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-08-07T15:31:01.082644Z","iopub.execute_input":"2024-08-07T15:31:01.082971Z","iopub.status.idle":"2024-08-07T15:33:35.521639Z","shell.execute_reply.started":"2024-08-07T15:31:01.082943Z","shell.execute_reply":"2024-08-07T15:33:35.520612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history.history.keys()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:33:35.526733Z","iopub.execute_input":"2024-08-07T15:33:35.527457Z","iopub.status.idle":"2024-08-07T15:33:35.53469Z","shell.execute_reply.started":"2024-08-07T15:33:35.52742Z","shell.execute_reply":"2024-08-07T15:33:35.533596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_acc = ['val_bowel_bowel_binary_accuracy', 'val_extra_extra_binary_accuracy', 'val_liver_liver_cat_accuracy', 'val_kidney_kidney_cat_accuracy', 'val_spleen_spleen_cat_accuracy']","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:33:35.536001Z","iopub.execute_input":"2024-08-07T15:33:35.536353Z","iopub.status.idle":"2024-08-07T15:33:35.546053Z","shell.execute_reply.started":"2024-08-07T15:33:35.53632Z","shell.execute_reply":"2024-08-07T15:33:35.543936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i,j in enumerate(val_acc):\n    print(j, np.asarray(history.history[val_acc[i]])[-1].round(2))\n#np.asarray(history.history['val_bowel_bowel_binary_accuracy'])[-1].round(2)","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:33:35.548074Z","iopub.execute_input":"2024-08-07T15:33:35.548474Z","iopub.status.idle":"2024-08-07T15:33:35.555079Z","shell.execute_reply.started":"2024-08-07T15:33:35.548435Z","shell.execute_reply":"2024-08-07T15:33:35.554095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in history.history.keys():\n    if i.endswith(\"accuracy\") and not i ==\"val_accuracy\":\n        plt.plot(history.history[i], label=i)\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Loss\")\nplt.legend(loc=(1.05,0.0))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:33:35.556226Z","iopub.execute_input":"2024-08-07T15:33:35.557032Z","iopub.status.idle":"2024-08-07T15:33:35.913335Z","shell.execute_reply.started":"2024-08-07T15:33:35.556996Z","shell.execute_reply":"2024-08-07T15:33:35.91235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(history.history[\"val_loss\"], label=\"val_loss\")\nplt.plot(history.history[\"loss\"], label=\"loss\")\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:33:35.914678Z","iopub.execute_input":"2024-08-07T15:33:35.915017Z","iopub.status.idle":"2024-08-07T15:33:36.171497Z","shell.execute_reply.started":"2024-08-07T15:33:35.914986Z","shell.execute_reply":"2024-08-07T15:33:36.170553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Testing the model using the Augmented data as test data","metadata":{}},{"cell_type":"code","source":"test_results = model.evaluate(\n    augmented_images, [bowel_test, extravasation_test, kidney_test, liver_test, spleen_test]\n) \n\n# Print the structure of test_results\nprint(\"Test Results Structure:\")\nprint(test_results)","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:33:36.172642Z","iopub.execute_input":"2024-08-07T15:33:36.172925Z","iopub.status.idle":"2024-08-07T15:33:50.728767Z","shell.execute_reply.started":"2024-08-07T15:33:36.172902Z","shell.execute_reply":"2024-08-07T15:33:50.727893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output_names = [\"bowel\", \"extra\", \"liver\", \"kidney\", \"spleen\"]","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:33:50.730059Z","iopub.execute_input":"2024-08-07T15:33:50.730702Z","iopub.status.idle":"2024-08-07T15:33:50.735216Z","shell.execute_reply.started":"2024-08-07T15:33:50.730666Z","shell.execute_reply":"2024-08-07T15:33:50.734106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_prob = model.predict(augmented_images)","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:33:50.736408Z","iopub.execute_input":"2024-08-07T15:33:50.736696Z","iopub.status.idle":"2024-08-07T15:33:54.331238Z","shell.execute_reply.started":"2024-08-07T15:33:50.736672Z","shell.execute_reply":"2024-08-07T15:33:54.330124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Model Output Names:\", model.output_names)  # Check names in model\nprint(\"y_pred_prob:\", y_pred_prob)\nprint(len(y_pred_prob))  # Print predictions and their shape","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:33:54.332944Z","iopub.execute_input":"2024-08-07T15:33:54.333276Z","iopub.status.idle":"2024-08-07T15:33:54.343094Z","shell.execute_reply.started":"2024-08-07T15:33:54.333248Z","shell.execute_reply":"2024-08-07T15:33:54.342091Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_true_dict = {\n    \"bowel\": bowel_test, \n    \"extra\": extravasation_test,\n    \"liver\": liver_test,\n    \"kidney\": kidney_test,\n    \"spleen\": spleen_test,\n}\n\n\n# Generate and plot confusion matrices for each output\nfor i, output_name in enumerate(output_names):\n    y_true = y_true_dict[output_name]\n    num_classes = 1 if i < 2 else 3  # Binary: 1 class, Multiclass: 3 classes\n    \n    # Ensure y_pred_prob is an array or similar structure\n    if not isinstance(y_pred_prob, (np.ndarray, list, tuple)):\n        y_pred_prob = np.array([y_pred_prob])\n        \n    # If predictions are a tuple of 5 elements\n    if len(y_pred_prob) == 5:\n        # Select the relevant element of the tuple based on the output name\n        y_pred_prob = y_pred_prob[i]\n        \n    # Make sure the predictions are 2D arrays with samples in the first dimension\n    if y_pred_prob.ndim == 1:\n        y_pred_prob = y_pred_prob.reshape(-1, 1)  # Reshape to 2D if necessary\n    \n    if num_classes == 1:\n        # Binary case (but with an extra dimension)\n        y_pred = (y_pred_prob > 0.5).astype(int).flatten()\n        labels = [\"Negative\", \"Positive\"]\n    else:\n        # Multiclass case \n        y_pred = np.argmax(y_pred_prob, axis=1)\n        labels = [f\"Class {i}\" for i in range(y_pred_prob.shape[1])]  \n\n    # Ensure lengths match before calculating the confusion matrix\n    min_len = min(len(y_true), len(y_pred))\n    y_true = y_true[:min_len]\n    y_pred = y_pred[:min_len]\n\n\n    # Ensure both y_true and y_pred are interpreted as multiclass if y_true.ndim > 1:\n    if y_true.ndim > 1:\n        y_true = np.argmax(y_true, axis=1)\n    if y_pred.ndim > 1:\n        y_pred = np.argmax(y_pred, axis=1)\n\n    cm = confusion_matrix(y_true, y_pred)\n    disp = ConfusionMatrixDisplay(confusion_matrix=cm)\n    disp.plot(cmap=plt.cm.Blues)\n\n    # Update ticks and labels directly on the Axes object\n    ax = disp.ax_ \n    ticks = np.arange(len(labels))\n    ax.set_xticks(ticks)\n    ax.set_yticks(ticks)\n    ax.set_xticklabels(labels)\n    ax.set_yticklabels(labels)\n    \n    plt.title(f\"Confusion Matrix - {output_name}\")\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-08-07T15:33:54.344872Z","iopub.execute_input":"2024-08-07T15:33:54.345307Z","iopub.status.idle":"2024-08-07T15:33:55.350515Z","shell.execute_reply.started":"2024-08-07T15:33:54.345272Z","shell.execute_reply":"2024-08-07T15:33:55.349545Z"},"trusted":true},"execution_count":null,"outputs":[]}]}