{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":52254,"databundleVersionId":9674523,"sourceType":"competition"},{"sourceId":6211844,"sourceType":"datasetVersion","datasetId":3567114}],"dockerImageVersionId":30554,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# <center>Multi-label classification of abdominal trauma from CT images</center>\n\n","metadata":{"papermill":{"duration":0.025483,"end_time":"2022-02-01T10:21:43.84374","exception":false,"start_time":"2022-02-01T10:21:43.818257","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"## To commence this project, the neccesary libraries need to be installed.\n## We would be using the **TensorFlow** framework and the pretrained **EfficientNetB1**\n","metadata":{}},{"cell_type":"code","source":"# Imporitng libraries\nimport pandas as pd\nimport os\nimport random\nimport numpy as np\nimport tensorflow as tf\nimport cv2\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom IPython.display import clear_output\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.model_selection import StratifiedKFold\n\n\nrandom.seed(42)","metadata":{"id":"fUPKjZaaJHkM","papermill":{"duration":5.021373,"end_time":"2022-02-01T10:21:48.88737","exception":false,"start_time":"2022-02-01T10:21:43.865997","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-11-23T19:29:24.117373Z","iopub.execute_input":"2024-11-23T19:29:24.117684Z","iopub.status.idle":"2024-11-23T19:29:32.541214Z","shell.execute_reply.started":"2024-11-23T19:29:24.117655Z","shell.execute_reply":"2024-11-23T19:29:32.540443Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Loading the training dataset\ntrain_img = \"/kaggle/input/rsna-atd-512x512-png-v2-dataset/train_images\"","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:29:32.542864Z","iopub.execute_input":"2024-11-23T19:29:32.543532Z","iopub.status.idle":"2024-11-23T19:29:32.547728Z","shell.execute_reply.started":"2024-11-23T19:29:32.54349Z","shell.execute_reply":"2024-11-23T19:29:32.546898Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i in os.listdir(train_img):\n    for j in os.listdir(os.path.join(train_img,i)):\n        print(len(os.listdir(os.path.join(train_img,i,j))))","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-11-23T19:29:32.548903Z","iopub.execute_input":"2024-11-23T19:29:32.549287Z","iopub.status.idle":"2024-11-23T19:29:35.053121Z","shell.execute_reply.started":"2024-11-23T19:29:32.549241Z","shell.execute_reply":"2024-11-23T19:29:35.052268Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install imbalanced-learn","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:29:35.054872Z","iopub.execute_input":"2024-11-23T19:29:35.05513Z","iopub.status.idle":"2024-11-23T19:29:44.015935Z","shell.execute_reply.started":"2024-11-23T19:29:35.055108Z","shell.execute_reply":"2024-11-23T19:29:44.014862Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Part A\n## Import and explore the data","metadata":{"papermill":{"duration":0.02193,"end_time":"2022-02-01T10:21:48.93356","exception":false,"start_time":"2022-02-01T10:21:48.91163","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Making a list containing all unique classes in the training set\n\n# Reading the training labels\ntraining_labels = pd.read_csv(\"/kaggle/input/rsna-2023-abdominal-trauma-detection/train_2024.csv\")","metadata":{"id":"PBlH8ae3LbG0","papermill":{"duration":0.069238,"end_time":"2022-02-01T10:21:49.024476","exception":false,"start_time":"2022-02-01T10:21:48.955238","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-11-23T19:29:44.017185Z","iopub.execute_input":"2024-11-23T19:29:44.017439Z","iopub.status.idle":"2024-11-23T19:29:44.039189Z","shell.execute_reply.started":"2024-11-23T19:29:44.017415Z","shell.execute_reply":"2024-11-23T19:29:44.038562Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Viewing the dataset in a structured format\ntraining_labels","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:29:44.040193Z","iopub.execute_input":"2024-11-23T19:29:44.040506Z","iopub.status.idle":"2024-11-23T19:29:44.06498Z","shell.execute_reply.started":"2024-11-23T19:29:44.040474Z","shell.execute_reply":"2024-11-23T19:29:44.064146Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Part B\n## In the next few cells, we chose to explore the dataframe to check for missing values","metadata":{}},{"cell_type":"code","source":"training_labels.columns","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:29:44.066137Z","iopub.execute_input":"2024-11-23T19:29:44.066373Z","iopub.status.idle":"2024-11-23T19:29:44.072236Z","shell.execute_reply.started":"2024-11-23T19:29:44.066353Z","shell.execute_reply":"2024-11-23T19:29:44.071266Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['patient_id'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:29:44.0732Z","iopub.execute_input":"2024-11-23T19:29:44.073445Z","iopub.status.idle":"2024-11-23T19:29:44.086414Z","shell.execute_reply.started":"2024-11-23T19:29:44.073426Z","shell.execute_reply":"2024-11-23T19:29:44.085568Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['bowel_healthy'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:29:44.087517Z","iopub.execute_input":"2024-11-23T19:29:44.087804Z","iopub.status.idle":"2024-11-23T19:29:44.094909Z","shell.execute_reply.started":"2024-11-23T19:29:44.087777Z","shell.execute_reply":"2024-11-23T19:29:44.09406Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['bowel_injury'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:29:44.097495Z","iopub.execute_input":"2024-11-23T19:29:44.097729Z","iopub.status.idle":"2024-11-23T19:29:44.106236Z","shell.execute_reply.started":"2024-11-23T19:29:44.09771Z","shell.execute_reply":"2024-11-23T19:29:44.105298Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['extravasation_healthy'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:29:44.107101Z","iopub.execute_input":"2024-11-23T19:29:44.107404Z","iopub.status.idle":"2024-11-23T19:29:44.115665Z","shell.execute_reply.started":"2024-11-23T19:29:44.107377Z","shell.execute_reply":"2024-11-23T19:29:44.114811Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['extravasation_injury'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:29:44.116607Z","iopub.execute_input":"2024-11-23T19:29:44.116823Z","iopub.status.idle":"2024-11-23T19:29:44.126812Z","shell.execute_reply.started":"2024-11-23T19:29:44.116804Z","shell.execute_reply":"2024-11-23T19:29:44.126131Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['kidney_healthy'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:29:44.127673Z","iopub.execute_input":"2024-11-23T19:29:44.127899Z","iopub.status.idle":"2024-11-23T19:29:44.139959Z","shell.execute_reply.started":"2024-11-23T19:29:44.127879Z","shell.execute_reply":"2024-11-23T19:29:44.139149Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['kidney_low'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:29:44.140785Z","iopub.execute_input":"2024-11-23T19:29:44.141048Z","iopub.status.idle":"2024-11-23T19:29:44.150367Z","shell.execute_reply.started":"2024-11-23T19:29:44.141007Z","shell.execute_reply":"2024-11-23T19:29:44.149603Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['kidney_high'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:29:44.151454Z","iopub.execute_input":"2024-11-23T19:29:44.151752Z","iopub.status.idle":"2024-11-23T19:29:44.161887Z","shell.execute_reply.started":"2024-11-23T19:29:44.151724Z","shell.execute_reply":"2024-11-23T19:29:44.161116Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['liver_healthy'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:29:44.163007Z","iopub.execute_input":"2024-11-23T19:29:44.163292Z","iopub.status.idle":"2024-11-23T19:29:44.173679Z","shell.execute_reply.started":"2024-11-23T19:29:44.163264Z","shell.execute_reply":"2024-11-23T19:29:44.172724Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['liver_low'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:29:44.174618Z","iopub.execute_input":"2024-11-23T19:29:44.174839Z","iopub.status.idle":"2024-11-23T19:29:44.186982Z","shell.execute_reply.started":"2024-11-23T19:29:44.17482Z","shell.execute_reply":"2024-11-23T19:29:44.186261Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['liver_high'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:29:44.188169Z","iopub.execute_input":"2024-11-23T19:29:44.188464Z","iopub.status.idle":"2024-11-23T19:29:44.198383Z","shell.execute_reply.started":"2024-11-23T19:29:44.188437Z","shell.execute_reply":"2024-11-23T19:29:44.197502Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['spleen_healthy'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:29:44.199521Z","iopub.execute_input":"2024-11-23T19:29:44.199914Z","iopub.status.idle":"2024-11-23T19:29:44.207878Z","shell.execute_reply.started":"2024-11-23T19:29:44.199881Z","shell.execute_reply":"2024-11-23T19:29:44.207072Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['spleen_low'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:29:44.208939Z","iopub.execute_input":"2024-11-23T19:29:44.209261Z","iopub.status.idle":"2024-11-23T19:29:44.22053Z","shell.execute_reply.started":"2024-11-23T19:29:44.209232Z","shell.execute_reply":"2024-11-23T19:29:44.219715Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['spleen_high'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:29:44.221418Z","iopub.execute_input":"2024-11-23T19:29:44.22161Z","iopub.status.idle":"2024-11-23T19:29:44.234037Z","shell.execute_reply.started":"2024-11-23T19:29:44.221594Z","shell.execute_reply":"2024-11-23T19:29:44.233325Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['any_injury'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:29:44.235077Z","iopub.execute_input":"2024-11-23T19:29:44.235663Z","iopub.status.idle":"2024-11-23T19:29:44.244883Z","shell.execute_reply.started":"2024-11-23T19:29:44.235634Z","shell.execute_reply":"2024-11-23T19:29:44.244063Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Using the EfficientNetB1 model and tweaking the last two layers to suit our work\n#def create_model(decay_steps=10,warmup_steps=10):\n#    base_model = tf.keras.applications.EfficientNetB1(\n#    weights= \"imagenet\", include_top=False, input_shape= (512,512,3)\n#    )\n#    num_classes=61\n\n#    x = base_model.output\n#    x = tf.keras.layers.GlobalAveragePooling2D()(x)\n#    x = tf.keras.layers.Dropout(0.2)(x)\n#    x_bowel = tf.keras.layers.Dense(32, activation='silu')(x)\n#    x_extra = tf.keras.layers.Dense(32, activation='silu')(x)\n#    x_liver = tf.keras.layers.Dense(32, activation='silu')(x)\n #   x_kidney = tf.keras.layers.Dense(32, activation='silu')(x)\n#    x_spleen = tf.keras.layers.Dense(32, activation='silu')(x)\n\n    # Define heads\n#    out_bowel = tf.keras.layers.Dense(1, name='bowel', activation='sigmoid')(x_bowel) # use sigmoid to convert predictions to [0-1]\n#    out_extra = tf.keras.layers.Dense(1, name='extra', activation='sigmoid')(x_extra) # use sigmoid to convert predictions to [0-1]\n#    out_liver = tf.keras.layers.Dense(3, name='liver', activation='softmax')(x_liver) # use softmax for the liver head\n#    out_kidney = tf.keras.layers.Dense(3, name='kidney', activation='softmax')(x_kidney) # use softmax for the kidney head\n#    out_spleen = tf.keras.layers.Dense(3, name='spleen', activation='softmax')(x_spleen) # use softmax for the spleen head\n\n#    model = tf.keras.Model(inputs = base_model.input, outputs = [out_bowel,out_extra,out_liver,out_kidney,out_spleen])\n        # Cosine Decay\n#    cosine_decay = tf.keras.optimizers.schedules.CosineDecay(\n#        initial_learning_rate=1e-4,\n#        decay_steps=decay_steps,\n#        alpha=0.0,\n        #warmup_target=1e-3,\n        #warmup_steps=warmup_steps,\n #   )\n\n    # Compile the model\n #   optimizer = tf.keras.optimizers.Adam(learning_rate=cosine_decay)\n #   loss = [\n #       tf.keras.losses.BinaryCrossentropy(),\n #       tf.keras.losses.BinaryCrossentropy(),\n #       tf.keras.losses.CategoricalCrossentropy(),\n #       tf.keras.losses.CategoricalCrossentropy(),\n #       tf.keras.losses.CategoricalCrossentropy()]\n    \n #   metrics = [\n #       [tf.keras.metrics.BinaryAccuracy(name=\"bowel_binary_accuracy\")],\n #       [tf.keras.metrics.BinaryAccuracy(name=\"extra_binary_accuracy\")],\n  #      [tf.keras.metrics.CategoricalAccuracy(name=\"liver_cat_accuracy\")],\n #       [tf.keras.metrics.CategoricalAccuracy(name=\"kidney_cat_accuracy\")],\n #       [tf.keras.metrics.CategoricalAccuracy(name=\"spleen_cat_accuracy\")]]\n    #    \"bowel\":[\"accuracy\"],\n    #    \"extra\":[\"accuracy\"],\n    #    \"liver\":[\"accuracy\"],\n    #    \"kidney\":[\"accuracy\"],\n    #    \"spleen\":[\"accuracy\"],\n    #}\n  #  print(\"[INFO] Compiling the model...\")\n #   model.compile(\n #       optimizer=optimizer,\n #     loss=loss,\n #     metrics=metrics\n #   )\n #   return model","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:29:44.245969Z","iopub.execute_input":"2024-11-23T19:29:44.246296Z","iopub.status.idle":"2024-11-23T19:29:44.256937Z","shell.execute_reply.started":"2024-11-23T19:29:44.246269Z","shell.execute_reply":"2024-11-23T19:29:44.256348Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"+ Sử dụng Gradient Clipping để tránh gradient quá lớn:","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\n\ndef create_model(decay_steps=900, warmup_steps=60, fine_tune_from=100):\n    # Tải EfficientNetB1 làm base model\n    base_model = tf.keras.applications.EfficientNetB1(\n        weights=\"imagenet\", include_top=False, input_shape=(512, 512, 3)\n    )\n\n    # Fine-tune một phần của base model\n    for layer in base_model.layers[:fine_tune_from]:\n        layer.trainable = False\n\n    x = base_model.output\n\n    # Lớp Convolution với kernel 3x3 và tăng số lượng filters\n    x = tf.keras.layers.Conv2D(128, (5, 5), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.MaxPooling2D((2, 2), padding='same')(x)\n    \n    x = tf.keras.layers.Conv2D(256, (5, 5), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.MaxPooling2D((2, 2), padding='same')(x)\n\n    x = tf.keras.layers.Conv2D(512, (5, 5), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.MaxPooling2D((2, 2), padding='same')(x)\n\n    x = tf.keras.layers.Conv2D(1024, (5, 5), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.MaxPooling2D((2, 2), padding='same')(x)\n\n    # Residual Block\n    residual = x\n    x = tf.keras.layers.Conv2D(1024, (2, 2), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.Add()([x, residual])  # Skip connection\n\n    # Global Average Pooling để giảm số chiều\n    x = tf.keras.layers.GlobalAveragePooling2D()(x)\n\n    # Dense Block với dropout vừa phải\n    x = tf.keras.layers.Dropout(0.4)(x)  # Giảm tỷ lệ dropout\n\n    # Dense layers cho các đầu ra\n    x_bowel = tf.keras.layers.Dense(512, activation='relu')(x)\n    x_extra = tf.keras.layers.Dense(512, activation='relu')(x)\n    x_liver = tf.keras.layers.Dense(512, activation='relu')(x)\n    x_kidney = tf.keras.layers.Dense(512, activation='relu')(x)\n    x_spleen = tf.keras.layers.Dense(512, activation='relu')(x)\n\n    # Output Layers\n    out_bowel = tf.keras.layers.Dense(1, name='bowel', activation='sigmoid')(x_bowel)  # Bowel (binary)\n    out_extra = tf.keras.layers.Dense(1, name='extra', activation='sigmoid')(x_extra)  # Extra (binary)\n    out_liver = tf.keras.layers.Dense(3, name='liver', activation='softmax')(x_liver)  # Liver (multiclass)\n    out_kidney = tf.keras.layers.Dense(3, name='kidney', activation='softmax')(x_kidney)  # Kidney (multiclass)\n    out_spleen = tf.keras.layers.Dense(3, name='spleen', activation='softmax')(x_spleen)  # Spleen (multiclass)\n\n    # Create model\n    model = tf.keras.Model(inputs=base_model.input, outputs=[out_bowel, out_extra, out_liver, out_kidney, out_spleen])\n    \n    # Cosine Decay with a warmup phase\n    cosine_decay = tf.keras.optimizers.schedules.CosineDecayRestarts(\n        initial_learning_rate=3e-4,  # Lower initial learning rate\n        first_decay_steps=warmup_steps,\n        t_mul=1.8,\n        m_mul=0.8,\n        alpha=0.2\n    )\n\n    # Compile the model\n    optimizer = tf.keras.optimizers.Adam(learning_rate=cosine_decay)\n    loss = [\n        tf.keras.losses.BinaryCrossentropy(),\n        tf.keras.losses.BinaryCrossentropy(),\n        tf.keras.losses.CategoricalCrossentropy(),\n        tf.keras.losses.CategoricalCrossentropy(),\n        tf.keras.losses.CategoricalCrossentropy()\n    ]\n    \n    metrics = [\n        [tf.keras.metrics.BinaryAccuracy(name=\"bowel_binary_accuracy\")],\n        [tf.keras.metrics.BinaryAccuracy(name=\"extra_binary_accuracy\")],\n        [tf.keras.metrics.CategoricalAccuracy(name=\"liver_cat_accuracy\")],\n        [tf.keras.metrics.CategoricalAccuracy(name=\"kidney_cat_accuracy\")],\n        [tf.keras.metrics.CategoricalAccuracy(name=\"spleen_cat_accuracy\")]\n    ]\n    \n    print(\"[INFO] Compiling the model...\")\n    model.compile(\n        optimizer=optimizer,\n        loss=loss,\n        metrics=metrics\n    )\n    \n    return model\n","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:29:44.258151Z","iopub.execute_input":"2024-11-23T19:29:44.258423Z","iopub.status.idle":"2024-11-23T19:29:44.272904Z","shell.execute_reply.started":"2024-11-23T19:29:44.258404Z","shell.execute_reply":"2024-11-23T19:29:44.272259Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#model = create_model()","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:29:44.273772Z","iopub.execute_input":"2024-11-23T19:29:44.273999Z","iopub.status.idle":"2024-11-23T19:29:44.285611Z","shell.execute_reply.started":"2024-11-23T19:29:44.273981Z","shell.execute_reply":"2024-11-23T19:29:44.284841Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = create_model()","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:29:44.286527Z","iopub.execute_input":"2024-11-23T19:29:44.28675Z","iopub.status.idle":"2024-11-23T19:29:48.67704Z","shell.execute_reply.started":"2024-11-23T19:29:44.286731Z","shell.execute_reply":"2024-11-23T19:29:48.676126Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# In ra tóm tắt mô hình\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:29:48.678186Z","iopub.execute_input":"2024-11-23T19:29:48.678444Z","iopub.status.idle":"2024-11-23T19:29:49.448993Z","shell.execute_reply.started":"2024-11-23T19:29:48.678422Z","shell.execute_reply":"2024-11-23T19:29:49.44807Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tf.keras.utils.plot_model(\n    model,  # Sử dụng đúng tên biến của mô hình\n    to_file='model.png'\n)","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:29:49.458619Z","iopub.execute_input":"2024-11-23T19:29:49.458966Z","iopub.status.idle":"2024-11-23T19:29:51.603821Z","shell.execute_reply.started":"2024-11-23T19:29:49.458934Z","shell.execute_reply":"2024-11-23T19:29:51.601343Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels = training_labels.set_index('patient_id')\ntraining_labels.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:29:51.605216Z","iopub.execute_input":"2024-11-23T19:29:51.605483Z","iopub.status.idle":"2024-11-23T19:29:51.61887Z","shell.execute_reply.started":"2024-11-23T19:29:51.605457Z","shell.execute_reply":"2024-11-23T19:29:51.617975Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Part C\n## Balancing the imbalanced training dataset ","metadata":{}},{"cell_type":"code","source":"#Using histogram to get the distribution of the labels and to check if there are outliers and wrong labels\ntraining_labels.hist(figsize=(20,12),bins=2)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:29:51.62045Z","iopub.execute_input":"2024-11-23T19:29:51.621153Z","iopub.status.idle":"2024-11-23T19:29:53.40264Z","shell.execute_reply.started":"2024-11-23T19:29:51.621122Z","shell.execute_reply":"2024-11-23T19:29:53.401812Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#images ,labels","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:29:53.403797Z","iopub.execute_input":"2024-11-23T19:29:53.404115Z","iopub.status.idle":"2024-11-23T19:29:53.408624Z","shell.execute_reply.started":"2024-11-23T19:29:53.404087Z","shell.execute_reply":"2024-11-23T19:29:53.40764Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"images = []\nlabels = []\nfor i in sorted(os.listdir(train_img)):\n    folder = os.listdir(os.path.join(train_img,i))[0]\n    file = os.listdir(os.path.join(train_img,i,folder))[0]\n    images.append(cv2.imread(os.path.join(train_img,i,folder,file), cv2.IMREAD_COLOR ))\n    labels.append(np.asarray(training_labels.loc[int(i)]))\nimages = np.asarray(images)\nlabels = np.asarray(labels)","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:29:53.409571Z","iopub.execute_input":"2024-11-23T19:29:53.409814Z","iopub.status.idle":"2024-11-23T19:29:56.136352Z","shell.execute_reply.started":"2024-11-23T19:29:53.409792Z","shell.execute_reply":"2024-11-23T19:29:56.135653Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Examples of images","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(30,12))\nfor i in range(10):\n    plt.subplot(2,5,i+1)\n    plt.imshow(images[i])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:29:56.137598Z","iopub.execute_input":"2024-11-23T19:29:56.138422Z","iopub.status.idle":"2024-11-23T19:29:57.905956Z","shell.execute_reply.started":"2024-11-23T19:29:56.138397Z","shell.execute_reply":"2024-11-23T19:29:57.905169Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:29:57.907148Z","iopub.execute_input":"2024-11-23T19:29:57.90745Z","iopub.status.idle":"2024-11-23T19:29:57.911509Z","shell.execute_reply.started":"2024-11-23T19:29:57.907423Z","shell.execute_reply":"2024-11-23T19:29:57.910701Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train, X_val, y_train, y_val = train_test_split(images, labels, test_size=0.25)","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:29:57.912839Z","iopub.execute_input":"2024-11-23T19:29:57.913134Z","iopub.status.idle":"2024-11-23T19:29:57.979753Z","shell.execute_reply.started":"2024-11-23T19:29:57.913112Z","shell.execute_reply":"2024-11-23T19:29:57.979062Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from imblearn.over_sampling import RandomOverSampler\n\n#oversampler = RandomOverSampler(random_state=42)\n#new_images, new_labels = oversampler.fit_resample(images, labels)","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:29:57.98091Z","iopub.execute_input":"2024-11-23T19:29:57.981261Z","iopub.status.idle":"2024-11-23T19:29:58.417768Z","shell.execute_reply.started":"2024-11-23T19:29:57.981229Z","shell.execute_reply":"2024-11-23T19:29:58.416844Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Part D\n## Training the model with the training dataset","metadata":{}},{"cell_type":"code","source":"bowel_labels = labels[:,2]\nextravasation_labels = labels[:,4]\nkidney_labels = labels[:,4:7]\nliver_labels = labels[:,7:10]\nspleen_labels = labels[:,10:13]\nany_labels = labels[:,-1]","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:29:58.419269Z","iopub.execute_input":"2024-11-23T19:29:58.420144Z","iopub.status.idle":"2024-11-23T19:29:58.424297Z","shell.execute_reply.started":"2024-11-23T19:29:58.420117Z","shell.execute_reply":"2024-11-23T19:29:58.423432Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"bowel_val = y_val[:,2]\nextravasation_val = y_val[:,4]\nkidney_val = y_val[:,4:7]\nliver_val = y_val[:,7:10]\nspleen_val = y_val[:,10:13]\nany_val = y_val[:,-1]","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:29:58.425379Z","iopub.execute_input":"2024-11-23T19:29:58.425688Z","iopub.status.idle":"2024-11-23T19:29:58.433706Z","shell.execute_reply.started":"2024-11-23T19:29:58.425656Z","shell.execute_reply":"2024-11-23T19:29:58.43294Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"bowel_train = y_train[:,2]\nextravasation_train = y_train[:,4]\nkidney_train = y_train[:,4:7]\nliver_train = y_train[:,7:10]\nspleen_train = y_train[:,10:13]\nany_train = y_train[:,-1]","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:29:58.434712Z","iopub.execute_input":"2024-11-23T19:29:58.435045Z","iopub.status.idle":"2024-11-23T19:29:58.441242Z","shell.execute_reply.started":"2024-11-23T19:29:58.434993Z","shell.execute_reply":"2024-11-23T19:29:58.440483Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#batch_size = 8\n#num_epoch = 2\n#history = model.fit(x=X_train,y=[bowel_train,extravasation_train,kidney_train,liver_train,spleen_train],batch_size=batch_size, epochs=num_epoch, verbose=1, validation_data=(X_val,[bowel_val,extravasation_val,kidney_val,liver_val,spleen_val]))\n\n\n#validation_data=(X_val,[bowel_val,extravasation_val,kidney_val,liver_val,spleen_val])","metadata":{"papermill":{"duration":1176.379334,"end_time":"2022-02-01T14:07:30.902597","exception":false,"start_time":"2022-02-01T13:47:54.523263","status":"completed"},"tags":[],"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-11-23T19:29:58.44221Z","iopub.execute_input":"2024-11-23T19:29:58.442526Z","iopub.status.idle":"2024-11-23T19:29:58.450833Z","shell.execute_reply.started":"2024-11-23T19:29:58.442495Z","shell.execute_reply":"2024-11-23T19:29:58.450175Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install keras-rectified-adam\n","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:29:58.451713Z","iopub.execute_input":"2024-11-23T19:29:58.451935Z","iopub.status.idle":"2024-11-23T19:30:08.360428Z","shell.execute_reply.started":"2024-11-23T19:29:58.451916Z","shell.execute_reply":"2024-11-23T19:30:08.358949Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.callbacks import EarlyStopping, ModelCheckpoint\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nimport numpy as np\nimport os\nimport cv2\n\n# Hàm tính F1 Score tùy chỉnh\ndef f1_score_metric(y_true, y_pred):\n    true_positives = tf.reduce_sum(tf.cast(tf.logical_and(tf.equal(y_true, 1), tf.equal(y_pred, 1)), tf.float32))\n    false_positives = tf.reduce_sum(tf.cast(tf.logical_and(tf.equal(y_true, 0), tf.equal(y_pred, 1)), tf.float32))\n    false_negatives = tf.reduce_sum(tf.cast(tf.logical_and(tf.equal(y_true, 1), tf.equal(y_pred, 0)), tf.float32))\n    \n    precision = true_positives / (true_positives + false_positives + tf.keras.backend.epsilon())\n    recall = true_positives / (true_positives + false_negatives + tf.keras.backend.epsilon())\n    \n    # Tính F1 score\n    f1 = 2 * (precision * recall) / (precision + recall + tf.keras.backend.epsilon())\n    return f1\n\n# Tạo mô hình với các lớp như EfficientNetB1, Conv2D và các lớp FC\ndef create_model(decay_steps=900, warmup_steps=60, fine_tune_from=100):\n    # Tải EfficientNetB1 làm base model\n    base_model = tf.keras.applications.EfficientNetB1(\n        weights=\"imagenet\", include_top=False, input_shape=(512, 512, 3)\n    )\n\n    # Fine-tune một phần của base model\n    for layer in base_model.layers[:fine_tune_from]:\n        layer.trainable = False\n\n    x = base_model.output\n\n    # Các lớp Convolution\n    x = tf.keras.layers.Conv2D(128, (5, 5), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.MaxPooling2D((2, 2), padding='same')(x)\n    \n    x = tf.keras.layers.Conv2D(256, (5, 5), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.MaxPooling2D((2, 2), padding='same')(x)\n\n    x = tf.keras.layers.Conv2D(512, (5, 5), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.MaxPooling2D((2, 2), padding='same')(x)\n\n    x = tf.keras.layers.Conv2D(1024, (5, 5), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.MaxPooling2D((2, 2), padding='same')(x)\n\n    # Residual Block\n    residual = x\n    x = tf.keras.layers.Conv2D(1024, (2, 2), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.Add()([x, residual])  # Skip connection\n\n    # Global Average Pooling\n    x = tf.keras.layers.GlobalAveragePooling2D()(x)\n\n    # Dense layer với dropout\n    x = tf.keras.layers.Dropout(0.4)(x)\n\n    # Các lớp Dense cho các đầu ra\n    x_bowel = tf.keras.layers.Dense(512, activation='relu')(x)\n    x_extra = tf.keras.layers.Dense(512, activation='relu')(x)\n    x_liver = tf.keras.layers.Dense(512, activation='relu')(x)\n    x_kidney = tf.keras.layers.Dense(512, activation='relu')(x)\n    x_spleen = tf.keras.layers.Dense(512, activation='relu')(x)\n\n    # Output Layers\n    out_bowel = tf.keras.layers.Dense(1, name='bowel', activation='sigmoid')(x_bowel)\n    out_extra = tf.keras.layers.Dense(1, name='extra', activation='sigmoid')(x_extra)\n    out_liver = tf.keras.layers.Dense(3, name='liver', activation='softmax')(x_liver)\n    out_kidney = tf.keras.layers.Dense(3, name='kidney', activation='softmax')(x_kidney)\n    out_spleen = tf.keras.layers.Dense(3, name='spleen', activation='softmax')(x_spleen)\n\n    # Tạo mô hình\n    model = tf.keras.Model(inputs=base_model.input, outputs=[out_bowel, out_extra, out_liver, out_kidney, out_spleen])\n    \n    # Cosine Decay Learning Rate\n    cosine_decay = tf.keras.optimizers.schedules.CosineDecayRestarts(\n        initial_learning_rate=3e-4,\n        first_decay_steps=warmup_steps,\n        t_mul=1.8,\n        m_mul=0.8,\n        alpha=0.2\n    )\n\n    # Compile the model\n    optimizer = tf.keras.optimizers.Adam(learning_rate=cosine_decay)\n    loss = [\n        tf.keras.losses.BinaryCrossentropy(),\n        tf.keras.losses.BinaryCrossentropy(),\n        tf.keras.losses.CategoricalCrossentropy(),\n        tf.keras.losses.CategoricalCrossentropy(),\n        tf.keras.losses.CategoricalCrossentropy()\n    ]\n    \n    metrics = [\n        # Metrics for Bowel (binary)\n        [tf.keras.metrics.BinaryAccuracy(name=\"bowel_binary_accuracy\"),\n         f1_score_metric,  # Sử dụng F1 Score tùy chỉnh\n         tf.keras.metrics.AUC(name=\"bowel_auc\"),\n         tf.keras.metrics.Recall(name=\"bowel_recall\"),\n         tf.keras.metrics.Precision(name=\"bowel_precision\"),  # Thêm Precision\n         tf.keras.metrics.TruePositives(name=\"bowel_tp\"),\n         tf.keras.metrics.FalseNegatives(name=\"bowel_fn\"),\n         tf.keras.metrics.FalsePositives(name=\"bowel_fp\"),\n         tf.keras.metrics.TrueNegatives(name=\"bowel_tn\")],\n        \n        # Metrics for Extravasation (binary)\n        [tf.keras.metrics.BinaryAccuracy(name=\"extra_binary_accuracy\"),\n         f1_score_metric,  # Sử dụng F1 Score tùy chỉnh\n         tf.keras.metrics.AUC(name=\"extra_auc\"),\n         tf.keras.metrics.Recall(name=\"extra_recall\"),\n         tf.keras.metrics.Precision(name=\"extra_precision\"),  # Thêm Precision\n         tf.keras.metrics.TruePositives(name=\"extra_tp\"),\n         tf.keras.metrics.FalseNegatives(name=\"extra_fn\"),\n         tf.keras.metrics.FalsePositives(name=\"extra_fp\"),\n         tf.keras.metrics.TrueNegatives(name=\"extra_tn\")],\n        \n        # Metrics for Liver (multiclass)\n        [tf.keras.metrics.CategoricalAccuracy(name=\"liver_cat_accuracy\"),\n         f1_score_metric,  # Sử dụng F1 Score tùy chỉnh\n         tf.keras.metrics.AUC(name=\"liver_auc\"),\n         tf.keras.metrics.Recall(name=\"liver_recall\"),\n         tf.keras.metrics.Precision(name=\"liver_precision\")],  # Thêm Precision\n        \n        # Metrics for Kidney (multiclass)\n        [tf.keras.metrics.CategoricalAccuracy(name=\"kidney_cat_accuracy\"),\n         f1_score_metric,  # Sử dụng F1 Score tùy chỉnh\n         tf.keras.metrics.AUC(name=\"kidney_auc\"),\n         tf.keras.metrics.Recall(name=\"kidney_recall\"),\n         tf.keras.metrics.Precision(name=\"kidney_precision\")],  # Thêm Precision\n        \n        # Metrics for Spleen (multiclass)\n        [tf.keras.metrics.CategoricalAccuracy(name=\"spleen_cat_accuracy\"),\n         f1_score_metric,  # Sử dụng F1 Score tùy chỉnh\n         tf.keras.metrics.AUC(name=\"spleen_auc\"),\n         tf.keras.metrics.Recall(name=\"spleen_recall\"),\n         tf.keras.metrics.Precision(name=\"spleen_precision\")]  # Thêm Precision\n    ]\n    \n    model.compile(optimizer=optimizer, loss=loss, metrics=metrics)\n    \n    return model\n\n\n# Load images and labels (dữ liệu và nhãn đã chuẩn bị)\nimages = []\nlabels = []\nfor i in sorted(os.listdir(train_img)):\n    folder = os.listdir(os.path.join(train_img, i))[0]\n    file = os.listdir(os.path.join(train_img, i, folder))[0]\n    images.append(cv2.imread(os.path.join(train_img, i, folder, file), cv2.IMREAD_COLOR))\n    labels.append(np.asarray(training_labels.loc[int(i)]))\nimages = np.asarray(images)\nlabels = np.asarray(labels)\n\n# Split data\nX_train, X_val, y_train, y_val = train_test_split(images, labels, test_size=0.25)\n\n# Prepare labels for each class\nbowel_labels = labels[:, 2]\nextravasation_labels = labels[:, 4]\nkidney_labels = labels[:, 4:7]\nliver_labels = labels[:, 7:10]\nspleen_labels = labels[:, 10:13]\nany_labels = labels[:, -1]\n\nbowel_val = y_val[:, 2]\nextravasation_val = y_val[:, 4]\nkidney_val = y_val[:, 4:7]\nliver_val = y_val[:, 7:10]\nspleen_val = y_val[:, 10:13]\nany_val = y_val[:, -1]\n\n# KFold Cross Validation\nkf = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n\n# Store results for each fold\nfold_results = []\n\n# Data Augmentation\ntrain_datagen = ImageDataGenerator(\n    rotation_range=40,\n    width_shift_range=0.2,\n    height_shift_range=0.2,\n    shear_range=0.2,\n    zoom_range=0.2,\n    horizontal_flip=True,\n    fill_mode='nearest'\n)\n\n# Validation data generator (no augmentation)\nval_datagen = ImageDataGenerator()\n\nfor fold, (train_index, val_index) in enumerate(kf.split(images, any_labels)):\n    print(f\"Fold {fold}\")  # Changed to print fold number from 0 to 4\n\n    # Split data into train and validation according to fold\n    X_train_fold, X_val_fold = images[train_index], images[val_index]\n    y_train_fold, y_val_fold = labels[train_index], labels[val_index]\n\n    # Prepare labels for each class for the fold\n    bowel_train = y_train_fold[:, 2]\n    extravasation_train = y_train_fold[:, 4]\n    kidney_train = y_train_fold[:, 4:7]\n    liver_train = y_train_fold[:, 7:10]\n    spleen_train = y_train_fold[:, 10:13]\n\n    bowel_val = y_val_fold[:, 2]\n    extravasation_val = y_val_fold[:, 4]\n    kidney_val = y_val_fold[:, 4:7]\n    liver_val = y_val_fold[:, 7:10]\n    spleen_val = y_val_fold[:, 10:13]\n\n    # Create a new model for each fold\n    model = create_model()\n\n    # EarlyStopping and ModelCheckpoint\n    early_stopping = EarlyStopping(monitor='val_loss', patience=10, restore_best_weights=True)\n    checkpoint = ModelCheckpoint(f'best_model_fold_{fold}.h5', save_best_only=True, monitor='val_loss')\n\n    # Train the model\n    batch_size = 16\n    num_epoch = 20  # You can modify this value based on the training time and model convergence\n    \n    history = model.fit(\n        x=X_train_fold,\n        y=[bowel_train, extravasation_train, kidney_train, liver_train, spleen_train],\n        batch_size=batch_size,\n        epochs=num_epoch,\n        verbose=1,\n        validation_data=(X_val_fold, [bowel_val, extravasation_val, kidney_val, liver_val, spleen_val])\n    )\n\n    # Store results for this fold\n    fold_results.append(history.history)\n\n# Check results\nfor fold, result in enumerate(fold_results):\n    print(f\"Fold {fold} Results:\")  # Changed to print fold number from 0 to 4\n    for key in result.keys():\n        print(f\"{key}: {result[key][-1]}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:30:08.362549Z","iopub.execute_input":"2024-11-23T19:30:08.363226Z","iopub.status.idle":"2024-11-23T19:47:30.766324Z","shell.execute_reply.started":"2024-11-23T19:30:08.363187Z","shell.execute_reply":"2024-11-23T19:47:30.765251Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.datasets import make_classification\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.ensemble import RandomForestClassifier\nimport numpy as np\n\n# Tạo dữ liệu mẫu\nX, y = make_classification(n_samples=1000, n_features=20, n_classes=2, random_state=42)\n\n# Khởi tạo model (có thể thay bằng model của bạn)\nmodel = RandomForestClassifier(random_state=42)\n\n# Khởi tạo các danh sách để lưu kết quả từ mỗi fold\naccuracy_scores = []\nprecision_scores = []\nrecall_scores = []\nf1_scores = []\n\n# Sử dụng StratifiedKFold cho Cross-Validation\nkf = StratifiedKFold(n_splits=5)\n\nfor train_index, val_index in kf.split(X, y):\n    X_train, X_val = X[train_index], X[val_index]\n    y_train, y_val = y[train_index], y[val_index]\n    \n    # Huấn luyện model trên tập train\n    model.fit(X_train, y_train)\n    \n    # Dự đoán trên tập validation\n    y_pred = model.predict(X_val)\n    \n    # Tính toán các chỉ số cho fold hiện tại\n    accuracy_scores.append(accuracy_score(y_val, y_pred))\n    precision_scores.append(precision_score(y_val, y_pred, average='weighted'))\n    recall_scores.append(recall_score(y_val, y_pred, average='weighted'))\n    f1_scores.append(f1_score(y_val, y_pred, average='weighted'))\n\n# Tính trung bình các chỉ số qua các fold\navg_accuracy = np.mean(accuracy_scores)\navg_precision = np.mean(precision_scores)\navg_recall = np.mean(recall_scores)\navg_f1 = np.mean(f1_scores)\n\nprint(f\"Average Accuracy: {avg_accuracy:.4f}\")\nprint(f\"Average Precision: {avg_precision:.4f}\")\nprint(f\"Average Recall: {avg_recall:.4f}\")\nprint(f\"Average F1 Score: {avg_f1:.4f}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:47:30.767904Z","iopub.execute_input":"2024-11-23T19:47:30.768331Z","iopub.status.idle":"2024-11-23T19:47:32.399837Z","shell.execute_reply.started":"2024-11-23T19:47:30.768295Z","shell.execute_reply":"2024-11-23T19:47:32.398924Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Store results for validation accuracy across all folds for all 5 metrics (bowel, extra, liver, kidney, spleen)\navg_val_acc = {\n    'bowel': 0,\n    'extra': 0,\n    'liver': 0,\n    'kidney': 0,\n    'spleen': 0\n}\n\n# Số lượng folds\nnum_folds = len(fold_results)\n\n# Lặp qua từng fold để tính tổng validation accuracy cho tất cả các lớp\nfor fold, result in enumerate(fold_results):\n    # Extract validation accuracy for each class (bowel, extravasation, liver, kidney, spleen)\n    val_bowel_acc = result['val_bowel_bowel_binary_accuracy'][-1]  # Validation accuracy for bowel\n    val_extra_acc = result['val_extra_extra_binary_accuracy'][-1]  # Validation accuracy for extra\n    val_liver_acc = result['val_liver_liver_cat_accuracy'][-1]  # Validation accuracy for liver\n    val_kidney_acc = result['val_kidney_kidney_cat_accuracy'][-1]  # Validation accuracy for kidney\n    val_spleen_acc = result['val_spleen_spleen_cat_accuracy'][-1]  # Validation accuracy for spleen\n    \n    # Cộng dồn các giá trị validation accuracy\n    avg_val_acc['bowel'] += val_bowel_acc\n    avg_val_acc['extra'] += val_extra_acc\n    avg_val_acc['liver'] += val_liver_acc\n    avg_val_acc['kidney'] += val_kidney_acc\n    avg_val_acc['spleen'] += val_spleen_acc\n\n# Tính trung bình validation accuracy cho tất cả các lớp\navg_val_acc['bowel'] /= num_folds\navg_val_acc['extra'] /= num_folds\navg_val_acc['liver'] /= num_folds\navg_val_acc['kidney'] /= num_folds\navg_val_acc['spleen'] /= num_folds\n\n# In kết quả validation accuracy trung bình cho mỗi lớp\nprint(\"\\nAverage Validation Accuracy across all folds for each class:\")\nprint(f\"Bowel Validation Accuracy: {avg_val_acc['bowel']:.4f}\")\nprint(f\"Extravasation Validation Accuracy: {avg_val_acc['extra']:.4f}\")\nprint(f\"Liver Validation Accuracy: {avg_val_acc['liver']:.4f}\")\nprint(f\"Kidney Validation Accuracy: {avg_val_acc['kidney']:.4f}\")\nprint(f\"Spleen Validation Accuracy: {avg_val_acc['spleen']:.4f}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:47:32.401082Z","iopub.execute_input":"2024-11-23T19:47:32.401433Z","iopub.status.idle":"2024-11-23T19:47:32.410319Z","shell.execute_reply.started":"2024-11-23T19:47:32.401399Z","shell.execute_reply":"2024-11-23T19:47:32.409394Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\n# Giả sử 'fold_results' là danh sách chứa các kết quả huấn luyện của từng fold\n# Mỗi phần tử trong 'fold_results' là dictionary chứa thông tin về các metric cho từng fold.\n\n# Danh sách các metrics mà bạn muốn theo dõi \nmetrics = [\n    'val_bowel_bowel_binary_accuracy', 'val_extra_extra_binary_accuracy',\n    'val_liver_liver_cat_accuracy', 'val_kidney_kidney_cat_accuracy', 'val_spleen_spleen_cat_accuracy',\n    'val_bowel_loss', 'val_extra_loss', 'val_liver_loss', 'val_kidney_loss', 'val_spleen_loss'\n]\n\n# Lưu kết quả tốt nhất của các metrics\nmetric_best_values = {metric: [] for metric in metrics}  # Khởi tạo dictionary để lưu giá trị tốt nhất của từng metric\n\n# Duyệt qua từng fold trong fold_results\nfor fold_result in fold_results:\n    for metric in metrics:\n        # Lấy giá trị của metric từ fold_result\n        if metric in fold_result:\n            metric_values = np.asarray(fold_result[metric])  # Lấy giá trị metric cho fold này\n            \n            # Tìm giá trị tốt nhất (max đối với accuracy, min đối với loss)\n            best_value = np.max(metric_values) if 'accuracy' in metric else np.min(metric_values)\n            \n            # Thêm giá trị tốt nhất vào danh sách của metric\n            metric_best_values[metric].append(best_value)\n\n# Tính và in trung bình tốt nhất cho các metrics (loại bỏ val_loss)\nfor metric, best_values in metric_best_values.items():\n    avg_best_value = np.mean(best_values)  # Tính trung bình của các giá trị tốt nhất từ các fold\n    print(f\"Best average for {metric}: {avg_best_value:.4f}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:47:32.411511Z","iopub.execute_input":"2024-11-23T19:47:32.411834Z","iopub.status.idle":"2024-11-23T19:47:32.422976Z","shell.execute_reply.started":"2024-11-23T19:47:32.411804Z","shell.execute_reply":"2024-11-23T19:47:32.422161Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Store results for the best validation accuracy across all folds for all 5 metrics (bowel, extra, liver, kidney, spleen)\navg_best_val_acc = {\n    'bowel': 0,\n    'extra': 0,\n    'liver': 0,\n    'kidney': 0,\n    'spleen': 0\n}\n\n# Số lượng folds\nnum_folds = len(fold_results)\n\n# Lặp qua từng fold để tìm validation accuracy tốt nhất cho tất cả các lớp\nfor fold, result in enumerate(fold_results):\n    # Find the best validation accuracy for each class (best val_acc across epochs)\n    best_val_bowel_acc = max(result['val_bowel_bowel_binary_accuracy'])  # Best validation accuracy for bowel\n    best_val_extra_acc = max(result['val_extra_extra_binary_accuracy'])  # Best validation accuracy for extra\n    best_val_liver_acc = max(result['val_liver_liver_cat_accuracy'])  # Best validation accuracy for liver\n    best_val_kidney_acc = max(result['val_kidney_kidney_cat_accuracy'])  # Best validation accuracy for kidney\n    best_val_spleen_acc = max(result['val_spleen_spleen_cat_accuracy'])  # Best validation accuracy for spleen\n    \n    # Cộng dồn các giá trị validation accuracy tốt nhất\n    avg_best_val_acc['bowel'] += best_val_bowel_acc\n    avg_best_val_acc['extra'] += best_val_extra_acc\n    avg_best_val_acc['liver'] += best_val_liver_acc\n    avg_best_val_acc['kidney'] += best_val_kidney_acc\n    avg_best_val_acc['spleen'] += best_val_spleen_acc\n\n# Tính trung bình validation accuracy tốt nhất cho tất cả các lớp\navg_best_val_acc['bowel'] /= num_folds\navg_best_val_acc['extra'] /= num_folds\navg_best_val_acc['liver'] /= num_folds\navg_best_val_acc['kidney'] /= num_folds\navg_best_val_acc['spleen'] /= num_folds\n\n# In kết quả validation accuracy tốt nhất trung bình cho mỗi lớp\nprint(\"\\nAverage Best Validation Accuracy across all folds for each class:\")\nprint(f\"Bowel Best Validation Accuracy: {avg_best_val_acc['bowel']:.4f}\")\nprint(f\"Extravasation Best Validation Accuracy: {avg_best_val_acc['extra']:.4f}\")\nprint(f\"Liver Best Validation Accuracy: {avg_best_val_acc['liver']:.4f}\")\nprint(f\"Kidney Best Validation Accuracy: {avg_best_val_acc['kidney']:.4f}\")\nprint(f\"Spleen Best Validation Accuracy: {avg_best_val_acc['spleen']:.4f}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:47:32.42408Z","iopub.execute_input":"2024-11-23T19:47:32.424383Z","iopub.status.idle":"2024-11-23T19:47:32.436284Z","shell.execute_reply.started":"2024-11-23T19:47:32.424361Z","shell.execute_reply":"2024-11-23T19:47:32.435367Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Store results for the best validation accuracy, accuracy, loss, F1 score, recall, precision across all folds for all 5 metrics (bowel, extra, liver, kidney, spleen)\navg_results = {\n    'acc': 0,\n    'f1': 0,\n    'recall': 0,\n    'precision': 0,\n    'loss': 0,\n    'val_acc': 0\n}\n\n# Số lượng folds\nnum_folds = len(fold_results)\n\n# Lặp qua từng fold để tìm các chỉ số tốt nhất cho tất cả các lớp\nfor fold, result in enumerate(fold_results):\n    # Find the best validation accuracy for each class (best val_acc across epochs)\n    best_val_bowel_acc = max(result['val_bowel_bowel_binary_accuracy'])  # Best validation accuracy for bowel\n    best_val_extra_acc = max(result['val_extra_extra_binary_accuracy'])  # Best validation accuracy for extra\n    best_val_liver_acc = max(result['val_liver_liver_cat_accuracy'])  # Best validation accuracy for liver\n    best_val_kidney_acc = max(result['val_kidney_kidney_cat_accuracy'])  # Best validation accuracy for kidney\n    best_val_spleen_acc = max(result['val_spleen_spleen_cat_accuracy'])  # Best validation accuracy for spleen\n\n    # Final metrics for each class\n    bowel_acc = result['bowel_bowel_binary_accuracy'][-1]  # Final accuracy for bowel\n    extra_acc = result['extra_extra_binary_accuracy'][-1]  # Final accuracy for extra\n    liver_acc = result['liver_liver_cat_accuracy'][-1]  # Final accuracy for liver\n    kidney_acc = result['kidney_kidney_cat_accuracy'][-1]  # Final accuracy for kidney\n    spleen_acc = result['spleen_spleen_cat_accuracy'][-1]  # Final accuracy for spleen\n    \n    bowel_loss = result['bowel_loss'][-1]  # Final loss for bowel\n    extra_loss = result['extra_loss'][-1]  # Final loss for extra\n    liver_loss = result['liver_loss'][-1]  # Final loss for liver\n    kidney_loss = result['kidney_loss'][-1]  # Final loss for kidney\n    spleen_loss = result['spleen_loss'][-1]  # Final loss for spleen\n    \n    bowel_f1 = result['bowel_f1_score_metric'][-1]  # Final F1 score for bowel\n    extra_f1 = result['extra_f1_score_metric'][-1]  # Final F1 score for extra\n    liver_f1 = result['liver_f1_score_metric'][-1]  # Final F1 score for liver\n    kidney_f1 = result['kidney_f1_score_metric'][-1]  # Final F1 score for kidney\n    spleen_f1 = result['spleen_f1_score_metric'][-1]  # Final F1 score for spleen\n    \n    bowel_recall = result['bowel_bowel_recall'][-1]  # Final recall for bowel\n    extra_recall = result['extra_extra_recall'][-1]  # Final recall for extra\n    liver_recall = result['liver_liver_recall'][-1]  # Final recall for liver\n    kidney_recall = result['kidney_kidney_recall'][-1]  # Final recall for kidney\n    spleen_recall = result['spleen_spleen_recall'][-1]  # Final recall for spleen\n    \n    bowel_precision = result['bowel_bowel_precision'][-1]  # Final precision for bowel\n    extra_precision = result['extra_extra_precision'][-1]  # Final precision for extra\n    liver_precision = result['liver_liver_precision'][-1]  # Final precision for liver\n    kidney_precision = result['kidney_kidney_precision'][-1]  # Final precision for kidney\n    spleen_precision = result['spleen_spleen_precision'][-1]  # Final precision for spleen\n\n    # Cộng dồn các giá trị cho các chỉ số\n    avg_results['val_acc'] += (best_val_bowel_acc + best_val_extra_acc + best_val_liver_acc + best_val_kidney_acc + best_val_spleen_acc)\n    \n    avg_results['acc'] += (bowel_acc + extra_acc + liver_acc + kidney_acc + spleen_acc)\n    avg_results['f1'] += (bowel_f1 + extra_f1 + liver_f1 + kidney_f1 + spleen_f1)\n    avg_results['recall'] += (bowel_recall + extra_recall + liver_recall + kidney_recall + spleen_recall)\n    avg_results['precision'] += (bowel_precision + extra_precision + liver_precision + kidney_precision + spleen_precision)\n    \n    avg_results['loss'] += (bowel_loss + extra_loss + liver_loss + kidney_loss + spleen_loss)\n\n# Tính trung bình cho tất cả các chỉ số\navg_results['val_acc'] /= (num_folds * 5)  # Chia cho 5 vì có 5 lớp\navg_results['acc'] /= (num_folds * 5)\navg_results['f1'] /= (num_folds * 5)\navg_results['recall'] /= (num_folds * 5)\navg_results['precision'] /= (num_folds * 5)\navg_results['loss'] /= (num_folds * 5)\n\n# In kết quả trung bình cho tất cả các chỉ số (không có val_loss)\nprint(\"\\nAverage Results across all folds (Total Average for all classes):\")\nprint(f\"Average Validation Accuracy: {avg_results['val_acc']:.4f}\")\nprint(f\"Average Accuracy: {avg_results['acc']:.4f}\")\nprint(f\"Average F1 Score: {avg_results['f1']:.4f}\")\nprint(f\"Average Recall: {avg_results['recall']:.4f}\")\nprint(f\"Average Precision: {avg_results['precision']:.4f}\")\nprint(f\"Average Loss: {avg_results['loss']:.4f}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:47:32.437532Z","iopub.execute_input":"2024-11-23T19:47:32.437787Z","iopub.status.idle":"2024-11-23T19:47:32.452235Z","shell.execute_reply.started":"2024-11-23T19:47:32.437766Z","shell.execute_reply":"2024-11-23T19:47:32.451397Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nimport numpy as np\n\n# Giả sử bạn đã lưu các độ chính xác của mỗi fold vào fold_results\naccuracies = []\n\nfor fold_result in fold_results:\n    # Độ chính xác cho các chỉ số trong quá trình huấn luyện, ví dụ 'val_accuracy'\n    accuracies.append(fold_result['val_loss'])  # Bạn có thể thay thế 'val_accuracy' bằng các chỉ số khác nếu muốn\n\n# Tạo Boxplot\nplt.figure(figsize=(10, 6))\nsns.boxplot(data=accuracies)\nplt.title('Boxplot of Validation Accuracy across Folds')\nplt.ylabel('Validation Accuracy')\nplt.show()\n\n","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:47:32.453303Z","iopub.execute_input":"2024-11-23T19:47:32.453577Z","iopub.status.idle":"2024-11-23T19:47:32.720817Z","shell.execute_reply.started":"2024-11-23T19:47:32.453557Z","shell.execute_reply":"2024-11-23T19:47:32.719966Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nimport numpy as np\n\n# Giả sử bạn đã lưu các độ chính xác của mỗi fold vào fold_results\naccuracies = []\n\nfor fold_result in fold_results:\n    # Kiểm tra nếu độ chính xác nằm trong 'metrics' (hoặc một khóa khác)\n    if 'accuracy' in fold_result:\n        accuracies.append(fold_result['accuracy'])\n    elif 'metrics' in fold_result and 'accuracy' in fold_result['metrics']:\n        accuracies.append(fold_result['metrics']['accuracy'])\n\n# Tạo Boxplot\nplt.figure(figsize=(10, 6))\nsns.boxplot(data=accuracies)\nplt.title('Boxplot of Validation Accuracy across Folds')\nplt.ylabel('Validation Accuracy')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:47:32.721824Z","iopub.execute_input":"2024-11-23T19:47:32.722105Z","iopub.status.idle":"2024-11-23T19:47:33.490634Z","shell.execute_reply.started":"2024-11-23T19:47:32.722082Z","shell.execute_reply":"2024-11-23T19:47:33.488813Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Store results for the best validation accuracy, accuracy, loss, F1 score, recall, precision across all folds for all 5 metrics (bowel, extra, liver, kidney, spleen)\navg_results = {\n    'acc': 0,\n    'f1': 0,\n    'recall': 0,\n    'precision': 0,\n    'loss': 0,\n    'val_acc': 0\n}\n\n# Số lượng folds\nnum_folds = len(fold_results)\n\n# Lặp qua từng fold để tìm các chỉ số tốt nhất cho tất cả các lớp\nfor fold, result in enumerate(fold_results):\n    # Find the best validation accuracy for each class (best val_acc across epochs)\n    best_val_bowel_acc = max(result['val_bowel_bowel_binary_accuracy'])  # Best validation accuracy for bowel\n    best_val_extra_acc = max(result['val_extra_extra_binary_accuracy'])  # Best validation accuracy for extra\n    best_val_liver_acc = max(result['val_liver_liver_cat_accuracy'])  # Best validation accuracy for liver\n    best_val_kidney_acc = max(result['val_kidney_kidney_cat_accuracy'])  # Best validation accuracy for kidney\n    best_val_spleen_acc = max(result['val_spleen_spleen_cat_accuracy'])  # Best validation accuracy for spleen\n\n    # Final metrics for each class\n    bowel_acc = result['bowel_bowel_binary_accuracy'][-1]  # Final accuracy for bowel\n    extra_acc = result['extra_extra_binary_accuracy'][-1]  # Final accuracy for extra\n    liver_acc = result['liver_liver_cat_accuracy'][-1]  # Final accuracy for liver\n    kidney_acc = result['kidney_kidney_cat_accuracy'][-1]  # Final accuracy for kidney\n    spleen_acc = result['spleen_spleen_cat_accuracy'][-1]  # Final accuracy for spleen\n    \n    bowel_loss = result['bowel_loss'][-1]  # Final loss for bowel\n    extra_loss = result['extra_loss'][-1]  # Final loss for extra\n    liver_loss = result['liver_loss'][-1]  # Final loss for liver\n    kidney_loss = result['kidney_loss'][-1]  # Final loss for kidney\n    spleen_loss = result['spleen_loss'][-1]  # Final loss for spleen\n    \n    bowel_f1 = result['bowel_f1_score_metric'][-1]  # Final F1 score for bowel\n    extra_f1 = result['extra_f1_score_metric'][-1]  # Final F1 score for extra\n    liver_f1 = result['liver_f1_score_metric'][-1]  # Final F1 score for liver\n    kidney_f1 = result['kidney_f1_score_metric'][-1]  # Final F1 score for kidney\n    spleen_f1 = result['spleen_f1_score_metric'][-1]  # Final F1 score for spleen\n    \n    bowel_recall = result['bowel_bowel_recall'][-1]  # Final recall for bowel\n    extra_recall = result['extra_extra_recall'][-1]  # Final recall for extra\n    liver_recall = result['liver_liver_recall'][-1]  # Final recall for liver\n    kidney_recall = result['kidney_kidney_recall'][-1]  # Final recall for kidney\n    spleen_recall = result['spleen_spleen_recall'][-1]  # Final recall for spleen\n    \n    bowel_precision = result['bowel_bowel_precision'][-1]  # Final precision for bowel\n    extra_precision = result['extra_extra_precision'][-1]  # Final precision for extra\n    liver_precision = result['liver_liver_precision'][-1]  # Final precision for liver\n    kidney_precision = result['kidney_kidney_precision'][-1]  # Final precision for kidney\n    spleen_precision = result['spleen_spleen_precision'][-1]  # Final precision for spleen\n\n    # Cộng dồn các giá trị cho các chỉ số\n    avg_results['val_acc'] += (best_val_bowel_acc + best_val_extra_acc + best_val_liver_acc + best_val_kidney_acc + best_val_spleen_acc)\n    \n    avg_results['acc'] += (bowel_acc + extra_acc + liver_acc + kidney_acc + spleen_acc)\n    avg_results['f1'] += (bowel_f1 + extra_f1 + liver_f1 + kidney_f1 + spleen_f1)\n    avg_results['recall'] += (bowel_recall + extra_recall + liver_recall + kidney_recall + spleen_recall)\n    avg_results['precision'] += (bowel_precision + extra_precision + liver_precision + kidney_precision + spleen_precision)\n    \n    avg_results['loss'] += (bowel_loss + extra_loss + liver_loss + kidney_loss + spleen_loss)\n\n# Tính trung bình cho tất cả các chỉ số\navg_results['val_acc'] /= (num_folds * 5)  # Chia cho 5 vì có 5 lớp\navg_results['acc'] /= (num_folds * 5)\navg_results['f1'] /= (num_folds * 5)\navg_results['recall'] /= (num_folds * 5)\navg_results['precision'] /= (num_folds * 5)\navg_results['loss'] /= (num_folds * 5)\n\n# In kết quả trung bình cho tất cả các chỉ số (không có val_loss)\nprint(\"\\nAverage Results across all folds (Total Average for all classes):\")\nprint(f\"Average Validation Accuracy: {avg_results['val_acc']:.4f}\")\nprint(f\"Average Accuracy: {avg_results['acc']:.4f}\")\nprint(f\"Average F1 Score: {avg_results['f1']:.4f}\")\nprint(f\"Average Recall: {avg_results['recall']:.4f}\")\nprint(f\"Average Precision: {avg_results['precision']:.4f}\")\nprint(f\"Average Loss: {avg_results['loss']:.4f}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:47:33.491483Z","iopub.status.idle":"2024-11-23T19:47:33.491802Z","shell.execute_reply.started":"2024-11-23T19:47:33.49165Z","shell.execute_reply":"2024-11-23T19:47:33.491668Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install matplotlib scikit-learn\n","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:47:33.492639Z","iopub.status.idle":"2024-11-23T19:47:33.49301Z","shell.execute_reply.started":"2024-11-23T19:47:33.492835Z","shell.execute_reply":"2024-11-23T19:47:33.492856Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nimport numpy as np\n\n# Giả sử bạn đã lưu các độ chính xác của mỗi fold vào fold_results\naccuracies = []\n\nfor fold_result in fold_results:\n    # Độ chính xác cho các chỉ số trong quá trình huấn luyện, ví dụ 'val_accuracy'\n    accuracies.append(fold_result['val_loss'])  # Bạn có thể thay thế 'val_accuracy' bằng các chỉ số khác nếu muốn\n\n# Tạo Boxplot\nplt.figure(figsize=(10, 6))\nsns.boxplot(data=accuracies)\nplt.title('Boxplot of Validation Accuracy across Folds')\nplt.ylabel('Validation Accuracy')\nplt.show()\n\n","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:47:33.494167Z","iopub.status.idle":"2024-11-23T19:47:33.494576Z","shell.execute_reply.started":"2024-11-23T19:47:33.49436Z","shell.execute_reply":"2024-11-23T19:47:33.49438Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.plot(history.history[\"val_loss\"], label=\"val_loss\")\nplt.plot(history.history[\"loss\"], label=\"loss\")\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:47:33.495656Z","iopub.status.idle":"2024-11-23T19:47:33.49607Z","shell.execute_reply.started":"2024-11-23T19:47:33.495842Z","shell.execute_reply":"2024-11-23T19:47:33.495861Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Kiểm tra các khóa trong history để biết các metric hiện có\nprint(history.history.keys())\n\n# Lọc các metric mà bạn muốn vẽ (chỉ vẽ accuracy mà không có 'val_' trong tên)\nmetrics_to_plot = [key for key in history.history.keys() if key.endswith('accuracy') and 'val_' not in key]\n\n# Tạo đồ thị cho mỗi metric trong huấn luyện và validation\nfor metric in metrics_to_plot:\n    plt.figure(figsize=(8, 6))  # Tạo một biểu đồ mới cho mỗi metric\n    plt.plot(history.history[metric], label=f'Training {metric}', linestyle='-', marker='o')\n    \n    # Kiểm tra và vẽ độ chính xác của validation nếu có\n    val_metric = f'val_{metric}'  # Tạo tên key của metric validation\n    if val_metric in history.history:\n        plt.plot(history.history[val_metric], label=f'Validation {metric}', linestyle='--', marker='x')\n    \n    # Thêm số epoch vào trục x (tự động từ 1 đến n)\n    epochs = range(1, len(history.history[metric]) + 1)\n    \n    # Thêm ticks vào trục x với khoảng cách 25 epochs\n    step_size = 25\n    epoch_ticks = [epoch for epoch in epochs if epoch % step_size == 0]\n    plt.xticks(epoch_ticks)\n    \n    # Cài đặt các thông số cho biểu đồ\n    plt.xlabel('Epochs', fontsize=12)\n    plt.ylabel('Accuracy', fontsize=12)\n    plt.title(f'{metric.capitalize()} over Epochs', fontsize=14)\n    plt.legend(loc='upper left')\n    plt.grid(True)\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:47:33.497507Z","iopub.status.idle":"2024-11-23T19:47:33.497939Z","shell.execute_reply.started":"2024-11-23T19:47:33.497718Z","shell.execute_reply":"2024-11-23T19:47:33.49774Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Kiểm tra các khóa trong history\nprint(history.history.keys())\n\n# Các metric bạn muốn vẽ (chỉ vẽ các accuracy mà không có \"val_\")\nmetrics_to_plot = [key for key in history.history.keys() if key.endswith('accuracy') and 'val_' not in key]\n\n# Tạo biểu đồ cho cả huấn luyện và validation\nfor metric in metrics_to_plot:\n    plt.figure(figsize=(8, 6))  # Tạo một biểu đồ mới cho mỗi metric\n    plt.plot(history.history[metric], label=f'Training {metric}', linestyle='-', marker='o')\n    \n    # Kiểm tra và vẽ độ chính xác của validation nếu có\n    val_metric = f'val_{metric}'\n    if val_metric in history.history:\n        plt.plot(history.history[val_metric], label=f'Validation {metric}', linestyle='--', marker='x')\n    \n    # Thêm số epoch vào trục x (tự động từ 0 đến n)\n    epochs = range(1, len(history.history[metric]) + 1)\n    \n    # Thêm ticks với khoảng cách 25\n    step_size = 25\n    epoch_ticks = [epoch for epoch in epochs if epoch % step_size == 0]\n    plt.xticks(epoch_ticks)\n    \n    # Cài đặt các thông số cho biểu đồ\n    plt.xlabel('Epochs', fontsize=12)\n    plt.ylabel('Accuracy', fontsize=12)\n    plt.title(f'{metric.capitalize()} over Epochs', fontsize=14)\n    plt.legend(loc='upper left')\n    plt.grid(True)\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:47:33.499339Z","iopub.status.idle":"2024-11-23T19:47:33.499733Z","shell.execute_reply.started":"2024-11-23T19:47:33.499527Z","shell.execute_reply":"2024-11-23T19:47:33.499546Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history.history.keys()","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:47:33.500702Z","iopub.status.idle":"2024-11-23T19:47:33.501125Z","shell.execute_reply.started":"2024-11-23T19:47:33.500898Z","shell.execute_reply":"2024-11-23T19:47:33.500917Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"val_acc = ['val_bowel_bowel_binary_accuracy', 'val_extra_extra_binary_accuracy', 'val_liver_liver_cat_accuracy', 'val_kidney_kidney_cat_accuracy', 'val_spleen_spleen_cat_accuracy']","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-11-23T19:47:33.502538Z","iopub.status.idle":"2024-11-23T19:47:33.502801Z","shell.execute_reply.started":"2024-11-23T19:47:33.502675Z","shell.execute_reply":"2024-11-23T19:47:33.502687Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Lặp qua các key trong history để vẽ accuracy\nfor i in history.history.keys():\n    if i.endswith(\"_accuracy\") and not i == \"val_accuracy\":\n        plt.plot(history.history[i], label=i)\n\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Accuracy\")\nplt.legend(loc=(1.05, 0.0))\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:47:33.503647Z","iopub.status.idle":"2024-11-23T19:47:33.503927Z","shell.execute_reply.started":"2024-11-23T19:47:33.503793Z","shell.execute_reply":"2024-11-23T19:47:33.503807Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i in history.history.keys():\n    if i.endswith(\"_loss\") and not i ==\"val_loss\":\n        plt.plot(history.history[i], label=i)\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Loss\")\nplt.legend(loc=(1.05,0.0))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:47:33.505297Z","iopub.status.idle":"2024-11-23T19:47:33.505588Z","shell.execute_reply.started":"2024-11-23T19:47:33.50545Z","shell.execute_reply":"2024-11-23T19:47:33.505464Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i in history.history.keys():\n    if i.endswith(\"accuracy\") and not i ==\"val_accuracy\":\n        plt.plot(history.history[i], label=i)\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Loss\")\nplt.legend(loc=(1.05,0.0))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:47:33.50732Z","iopub.status.idle":"2024-11-23T19:47:33.507588Z","shell.execute_reply.started":"2024-11-23T19:47:33.507454Z","shell.execute_reply":"2024-11-23T19:47:33.507467Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Lặp qua các key trong history để vẽ accuracy (loại bỏ val_accuracy)\nfor i in history.history.keys():\n    if i.endswith(\"accuracy\") and \"val_\" not in i:\n        plt.plot(history.history[i], label=i)\n\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Accuracy\")\nplt.legend(loc=(1.05, 0.0))\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:47:33.509235Z","iopub.status.idle":"2024-11-23T19:47:33.509507Z","shell.execute_reply.started":"2024-11-23T19:47:33.509381Z","shell.execute_reply":"2024-11-23T19:47:33.509394Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Kiểm tra các khóa trong history\nprint(history.history.keys())\n\n# Các metric bạn muốn vẽ (chỉ vẽ các accuracy mà không có \"val_\")\nmetrics_to_plot = [key for key in history.history.keys() if key.endswith('accuracy') and 'val_' not in key]\n\n# Tạo biểu đồ cho cả huấn luyện và validation\nfor metric in metrics_to_plot:\n    plt.figure(figsize=(8, 6))  # Tạo một biểu đồ mới cho mỗi metric\n    plt.plot(history.history[metric], label=f'Training {metric}')\n    \n    # Kiểm tra và vẽ độ chính xác của validation nếu có\n    val_metric = f'val_{metric}'\n    if val_metric in history.history:\n        plt.plot(history.history[val_metric], label=f'Validation {metric}')\n    \n    # Thêm số epoch vào trục x (tự động từ 0 đến n)\n    epochs = range(1, len(history.history[metric]) + 1)\n    \n    # Thêm ticks với khoảng cách 25\n    step_size = 25\n    epoch_ticks = [epoch for epoch in epochs if epoch % step_size == 0]\n    plt.xticks(epoch_ticks)\n    \n    # Cài đặt các thông số cho biểu đồ\n    plt.xlabel('Epochs')\n    plt.ylabel('Accuracy')\n    plt.title(f'{metric.capitalize()} over Epochs')\n    plt.legend(loc='upper left')\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:47:33.51047Z","iopub.status.idle":"2024-11-23T19:47:33.510724Z","shell.execute_reply.started":"2024-11-23T19:47:33.5106Z","shell.execute_reply":"2024-11-23T19:47:33.510612Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\n\n# Giả sử bạn có các giá trị accuracy và các metrics Ep cho từng epoch hoặc mô hình\nepochs = np.arange(1, 11)  # Ví dụ: 10 epochs\naccuracy = np.random.rand(10)  # Accuracy giả định (tạo ngẫu nhiên từ 0 đến 1)\nprecision = np.random.rand(10)  # Precision giả định (tạo ngẫu nhiên)\nrecall = np.random.rand(10)  # Recall giả định\nf1_score = np.random.rand(10)  # F1-Score giả định\n\n# Vẽ đồ thị đường cho Accuracy và các metrics Ep\nplt.figure(figsize=(10, 6))\n\n# Accuracy\nplt.plot(epochs, accuracy, label='Accuracy', color='blue', marker='o', linestyle='-', linewidth=2)\n\n# Precision\nplt.plot(epochs, precision, label='Precision', color='green', marker='s', linestyle='--', linewidth=2)\n\n# Recall\nplt.plot(epochs, recall, label='Recall', color='red', marker='^', linestyle='-.', linewidth=2)\n\n# F1-Score\nplt.plot(epochs, f1_score, label='F1-Score', color='purple', marker='x', linestyle=':', linewidth=2)\n\n# Thiết lập nhãn và tiêu đề\nplt.xlabel('Epochs', fontsize=14)\nplt.ylabel('Scores', fontsize=14)\nplt.title('Accuracy and EP (Precision, Recall, F1-Score) vs Epochs', fontsize=16)\nplt.legend(loc='upper left')\n\n# Hiển thị đồ thị\nplt.grid(True)\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:47:33.512195Z","iopub.status.idle":"2024-11-23T19:47:33.51248Z","shell.execute_reply.started":"2024-11-23T19:47:33.512346Z","shell.execute_reply":"2024-11-23T19:47:33.512359Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.plot(history.history[\"val_loss\"], label=\"val_loss\")\nplt.plot(history.history[\"loss\"], label=\"loss\")\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:47:33.513537Z","iopub.status.idle":"2024-11-23T19:47:33.513818Z","shell.execute_reply.started":"2024-11-23T19:47:33.513683Z","shell.execute_reply":"2024-11-23T19:47:33.513697Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Đọc dữ liệu từ file train.csv\ndf = pd.read_csv('/kaggle/input/rsna-atd-512x512-png-v2-dataset/train.csv')\n\n# Hiển thị các cột trong DataFrame\nprint(df.columns)\n\n# Giả sử nhãn thật là các cột từ \"bowel_healthy\" đến \"spleen_high\"\nbinary_columns = [\n    'bowel_healthy', 'bowel_injury', 'extravasation_healthy', 'extravasation_injury',\n    'kidney_healthy', 'kidney_low', 'kidney_high', 'liver_healthy', 'liver_low', 'liver_high',\n    'spleen_healthy', 'spleen_low', 'spleen_high'\n]\n\n# Lấy các nhãn thật từ DataFrame (các cột nhãn)\nall_targets_np = df[binary_columns].values\n\n# In ra 5 mẫu nhãn thật đầu tiên\nprint(all_targets_np[:5])  # In 5 mẫu đầu tiên\n","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:47:33.515378Z","iopub.status.idle":"2024-11-23T19:47:33.515675Z","shell.execute_reply.started":"2024-11-23T19:47:33.515549Z","shell.execute_reply":"2024-11-23T19:47:33.515562Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Đọc dữ liệu huấn luyện từ file CSV\ntrain_df = pd.read_csv('/kaggle/input/rsna-atd-512x512-png-v2-dataset/train.csv')\n\n# Kiểm tra cấu trúc dữ liệu\ntrain_df.head()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:47:33.516748Z","iopub.status.idle":"2024-11-23T19:47:33.517003Z","shell.execute_reply.started":"2024-11-23T19:47:33.516879Z","shell.execute_reply":"2024-11-23T19:47:33.516891Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Đọc dữ liệu từ file CSV\ntrain_df = pd.read_csv('/kaggle/input/rsna-atd-512x512-png-v2-dataset/train.csv')\n\n# Danh sách các cột nhị phân (binary_columns)\nbinary_columns = [\n    'bowel_healthy', 'bowel_injury', 'extravasation_healthy', 'extravasation_injury',\n    'kidney_healthy', 'kidney_low', 'kidney_high', 'liver_healthy', 'liver_low', 'liver_high',\n    'spleen_healthy', 'spleen_low', 'spleen_high'\n]\n\n# Giả sử bạn đã có nhãn dự đoán (y_pred) từ mô hình của bạn\n# Ví dụ: Nếu bạn có các dự đoán từ mô hình (thay 'y_pred' bằng kết quả dự đoán của bạn)\n# Ở đây, ta tạo ra các nhãn giả y_pred (thay thế bằng kết quả thực tế của bạn)\ny_true = train_df[binary_columns].values  # Nhãn thật\ny_pred = np.random.randint(0, 1, size=y_true.shape)  # Giả lập nhãn dự đoán (thay bằng giá trị thực tế từ mô hình)\n\n# Vẽ ma trận nhầm lẫn cho từng lớp\nfor idx, label in enumerate(binary_columns):\n    # Tạo ma trận nhầm lẫn cho từng lớp\n    cm = confusion_matrix(y_true[:, idx], y_pred[:, idx])\n    \n    # Vẽ ma trận nhầm lẫn\n    plt.figure(figsize=(6, 5))\n    sns.heatmap(cm, annot=True, fmt='d', cmap='Blues', xticklabels=['Pred Negative', 'Pred Positive'], yticklabels=['True Negative', 'True Positive'])\n    plt.title(f\"Confusion Matrix for {label}\")\n    plt.xlabel('Predicted')\n    plt.ylabel('True')\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:47:33.517927Z","iopub.status.idle":"2024-11-23T19:47:33.518222Z","shell.execute_reply.started":"2024-11-23T19:47:33.518086Z","shell.execute_reply":"2024-11-23T19:47:33.5181Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Đọc dữ liệu từ file CSV\ntrain_df = pd.read_csv('/kaggle/input/rsna-atd-512x512-png-v2-dataset/train.csv')\n\n# Danh sách các cột nhị phân (binary_columns)\nbinary_columns = [\n    'bowel_healthy', 'bowel_injury', 'extravasation_healthy', 'extravasation_injury',\n    'kidney_healthy', 'kidney_low', 'kidney_high', 'liver_healthy', 'liver_low', 'liver_high',\n    'spleen_healthy', 'spleen_low', 'spleen_high'\n]\n\n# Giả sử bạn đã có nhãn dự đoán (y_pred) từ mô hình của bạn\n# Ví dụ: Nếu bạn có các dự đoán từ mô hình (thay 'y_pred' bằng kết quả dự đoán của bạn)\n# Ở đây, ta tạo ra các nhãn giả y_pred (thay thế bằng giá trị thực tế của bạn)\ny_true = train_df[binary_columns].values  # Nhãn thật\ny_pred = np.random.randint(0, 10, size=y_true.shape)  # Giả lập nhãn dự đoán (thay bằng giá trị thực tế từ mô hình)\n\n# Tạo ma trận nhầm lẫn cho từng lớp\ncm_all = []\nfor idx, label in enumerate(binary_columns):\n    # Tính ma trận nhầm lẫn cho mỗi lớp\n    cm = confusion_matrix(y_true[:, idx], y_pred[:, idx])\n    cm_all.append(cm)\n\n# Vẽ ma trận nhầm lẫn cho tất cả các lớp\nfig, axes = plt.subplots(nrows=4, ncols=4, figsize=(15, 15))\naxes = axes.flatten()\n\nfor idx, cm in enumerate(cm_all):\n    ax = axes[idx]\n    sns.heatmap(cm, annot=True, fmt='d', cmap='Blues', xticklabels=['Predicted 0', 'Predicted 1'], \n                yticklabels=['True 0', 'True 1'], ax=ax)\n    ax.set_title(f\"Confusion Matrix for {binary_columns[idx]}\")\n    ax.set_xlabel('Predicted')\n    ax.set_ylabel('True')\n\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:47:33.519312Z","iopub.status.idle":"2024-11-23T19:47:33.519604Z","shell.execute_reply.started":"2024-11-23T19:47:33.519462Z","shell.execute_reply":"2024-11-23T19:47:33.519476Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Đọc dữ liệu từ file CSV\ntrain_df = pd.read_csv('/kaggle/input/rsna-atd-512x512-png-v2-dataset/train.csv')\n\n# Danh sách các cột nhị phân (binary_columns)\nbinary_columns = [\n    'bowel_healthy', 'bowel_injury', 'extravasation_healthy', 'extravasation_injury',\n    'kidney_healthy', 'kidney_low', 'kidney_high', 'liver_healthy', 'liver_low', 'liver_high',\n    'spleen_healthy', 'spleen_low', 'spleen_high'\n]\n\n# Giả sử bạn đã có nhãn dự đoán (y_pred) từ mô hình của bạn\n# Ví dụ: Nếu bạn có các dự đoán từ mô hình (thay 'y_pred' bằng kết quả dự đoán của bạn)\n# Ở đây, ta tạo ra các nhãn giả y_pred (thay thế bằng giá trị thực tế của bạn)\ny_true = train_df[binary_columns].values  # Nhãn thật\ny_pred = np.random.randint(0, 5, size=y_true.shape)  # Giả lập nhãn dự đoán (thay bằng giá trị thực tế từ mô hình)\n\n# Tạo ma trận nhầm lẫn cho từng lớp và vẽ\nfig, axes = plt.subplots(nrows=4, ncols=4, figsize=(15, 15))\naxes = axes.flatten()\n\n# Tạo ma trận nhầm lẫn cho từng lớp\nfor idx, label in enumerate(binary_columns):\n    cm = confusion_matrix(y_true[:, idx], y_pred[:, idx])\n    \n    # Vẽ ma trận nhầm lẫn cho từng lớp\n    ax = axes[idx]\n    sns.heatmap(cm, annot=True, fmt='d', cmap='Blues', \n                xticklabels=['Predicted 0', 'Predicted 1'], \n                yticklabels=['True 0', 'True 1'], ax=ax)\n    ax.set_title(f\"Confusion Matrix for {label}\")\n    ax.set_xlabel('Predicted')\n    ax.set_ylabel('True')\n\n# Điều chỉnh layout và hiển thị\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:47:33.521621Z","iopub.status.idle":"2024-11-23T19:47:33.522034Z","shell.execute_reply.started":"2024-11-23T19:47:33.521814Z","shell.execute_reply":"2024-11-23T19:47:33.521834Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Đảm bảo rằng tệp CSV hoặc nguồn dữ liệu của bạn đã được nạp vào DataFrame\ndf = pd.read_csv(\"/kaggle/input/rsna-2023-abdominal-trauma-detection/train_2024.csv\")  # Thay thế bằng đường dẫn đúng\n\n# Kiểm tra các giá trị thiếu trong DataFrame\nmissing_values = df.isnull().sum()\nprint(\"Missing Values:\")\nprint(missing_values)\n\n# Danh sách các cột nhị phân cần chuyển đổi sang kiểu boolean\nbinary_columns = [\n    'bowel_healthy', 'bowel_injury', 'extravasation_healthy', 'extravasation_injury',\n    'kidney_healthy', 'kidney_low', 'kidney_high', 'liver_healthy', 'liver_low', 'liver_high',\n    'spleen_healthy', 'spleen_low', 'spleen_high'\n]\n\n# Chuyển đổi các cột nhị phân thành kiểu boolean\ndf[binary_columns] = df[binary_columns].astype(bool)\n\n# Giải quyết các vấn đề về chất lượng dữ liệu (nếu có, bạn có thể thêm các bước xử lý dữ liệu ở đây)\n\n# Hiển thị DataFrame sau khi đã xử lý\nprint(\"\\nPreprocessed DataFrame:\")\nprint(df.head())  # In ra 5 dòng đầu tiên của DataFrame đã xử lý để kiểm tra\n","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:47:33.52342Z","iopub.status.idle":"2024-11-23T19:47:33.523812Z","shell.execute_reply.started":"2024-11-23T19:47:33.523607Z","shell.execute_reply":"2024-11-23T19:47:33.523626Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Đọc tệp CSV chứa nhãn\ndf = pd.read_csv('/kaggle/input/rsna-atd-512x512-png-v2-dataset/train.csv')\n\n# Loại bỏ khoảng trắng thừa trong tên cột (nếu có)\ndf.columns = df.columns.str.strip()\n\n# Tạo danh sách các cột nhãn cho các cơ quan bạn muốn hiển thị\norgan_columns = ['kidney', 'liver', 'spleen']  # Chỉ giữ lại các cơ quan bạn muốn hiển thị\n\n# Vẽ ma trận nhầm lẫn cho các cơ quan\nfor organ in organ_columns:\n    # Đảm bảo tên cột chính xác sau khi loại bỏ khoảng trắng\n    y_true = df[f'{organ}_healthy']  # Nhãn thực tế (ví dụ: kidney_healthy)\n    y_pred = df[f'{organ}_low']  # Hoặc chọn nhãn khác như kidney_low (tùy vào mục đích của bạn)\n\n    # Tạo ma trận nhầm lẫn với 5 lớp (0 đến 4)\n    cm = confusion_matrix(y_true, y_pred, labels=[False, True, 2, 3, 4,5,6,7,8,9])  # Chỉ định 5 lớp (tùy theo dữ liệu của bạn)\n\n    # Vẽ ma trận nhầm lẫn\n    plt.figure(figsize=(10, 8))  # Thay đổi kích thước đồ thị cho phù hợp với ma trận\n    sns.heatmap(cm, annot=True, fmt='d', cmap='Blues', xticklabels=[0, 1, 2, 3, 4 , 5, 6, 7, 8, 9, 10], yticklabels=[0, 1, 2, 3, 4 , 5, 6, 7, 8, 9, 10])\n\n    # Thiết lập tiêu đề và nhãn\n    plt.title(f'Confusion Matrix for {organ.capitalize()}')\n    plt.xlabel('Predicted')\n    plt.ylabel('Actual')\n\n    # Hiển thị ma trận nhầm lẫn\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:47:33.528616Z","iopub.status.idle":"2024-11-23T19:47:33.52891Z","shell.execute_reply.started":"2024-11-23T19:47:33.528774Z","shell.execute_reply":"2024-11-23T19:47:33.528788Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Kiểm tra tất cả các cột trong DataFrame\nprint(df.columns)\n","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:47:33.530272Z","iopub.status.idle":"2024-11-23T19:47:33.530571Z","shell.execute_reply.started":"2024-11-23T19:47:33.530439Z","shell.execute_reply":"2024-11-23T19:47:33.530452Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Đọc tệp CSV chứa nhãn\ndf = pd.read_csv('/kaggle/input/rsna-atd-512x512-png-v2-dataset/train.csv')\n\n# Loại bỏ khoảng trắng thừa trong tên cột (nếu có)\ndf.columns = df.columns.str.strip()\n\n# Tạo danh sách các cột nhãn cho các cơ quan bạn muốn hiển thị\norgan_columns = ['kidney', 'liver', 'spleen']  # Các cơ quan cần hiển thị\n\n# Vẽ ma trận nhầm lẫn cho các cơ quan\nfor organ in organ_columns:\n    # Đảm bảo tên cột chính xác sau khi loại bỏ khoảng trắng\n    y_true = df[f'{organ}_healthy']  # Nhãn thực tế (ví dụ: kidney_healthy)\n    y_pred = df[f'{organ}_low']  # Hoặc chọn nhãn khác như kidney_low (tùy vào mục đích của bạn)\n\n    # Tạo ma trận nhầm lẫn với 10 lớp (0 đến 9)\n    cm = confusion_matrix(y_true, y_pred, labels=[False, True, 2, 3, 4, 5, 6, 7, 8, 9])  # Giả sử dữ liệu có lớp 0-9\n\n    # Vẽ ma trận nhầm lẫn\n    plt.figure(figsize=(10, 8))  # Thay đổi kích thước đồ thị cho phù hợp với ma trận\n    sns.heatmap(cm, annot=True, fmt='d', cmap='Blues', xticklabels=[0, 1, 2, 3, 4 , 5, 6, 7, 8, 9], yticklabels=[0, 1, 2, 3, 4 , 5, 6, 7, 8, 9])\n\n    # Thiết lập tiêu đề và nhãn\n    plt.title(f'Confusion Matrix for {organ.capitalize()}')\n    plt.xlabel('Predicted')\n    plt.ylabel('Actual')\n\n    # Hiển thị ma trận nhầm lẫn\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:47:33.532006Z","iopub.status.idle":"2024-11-23T19:47:33.5323Z","shell.execute_reply.started":"2024-11-23T19:47:33.53217Z","shell.execute_reply":"2024-11-23T19:47:33.532183Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check for missing values\nmissing_values = df.isnull().sum()\nprint(\"Missing Values:\")\nprint(missing_values)\n\n# Handle missing values\n# In this simple example, we will drop rows with missing values.\ndf = df.dropna()\n\n# Check Data Types and Convert Binary Data to Boolean\nbinary_columns = [\n  'bowel_healthy', 'bowel_injury', 'extravasation_healthy', 'extravasation_injury',\n  'kidney_healthy', 'kidney_low', 'kidney_high', 'liver_healthy', 'liver_low', 'liver_high',\n  'spleen_healthy', 'spleen_low', 'spleen_high'\n]\ndf[binary_columns] = df[binary_columns].astype(bool)\n\n# Address Data Quality Issues\n# In this simple example, we assume no data quality issues are present.\n\nprint(\"\\nPreprocessed DataFrame:\")\nprint(df)\n\nplt.figure()\ndf.plot.hist()\nplt.title('Distribution of Features')\nplt.xlabel('Feature')\nplt.ylabel('Count')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:47:33.53339Z","iopub.status.idle":"2024-11-23T19:47:33.533679Z","shell.execute_reply.started":"2024-11-23T19:47:33.533543Z","shell.execute_reply":"2024-11-23T19:47:33.533556Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(df.columns)\n","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:47:33.535615Z","iopub.status.idle":"2024-11-23T19:47:33.535906Z","shell.execute_reply.started":"2024-11-23T19:47:33.535769Z","shell.execute_reply":"2024-11-23T19:47:33.535783Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport plotly.express as px\n\n# Giả sử df là DataFrame đã được nạp vào từ dữ liệu của bạn\n# df = pd.read_csv(\"/path/to/your/data.csv\")\n\n# Cột các cơ quan\norgan_columns = ['bowel', 'extravasation', 'kidney', 'liver', 'spleen']\n\n# Kiểm tra xem các cột \"injury\" có tồn tại trong DataFrame không\ninjury_columns = [f'{organ}_injury' for organ in organ_columns]\nmissing_columns = set(organ_columns + injury_columns) - set(df.columns)\n\nif missing_columns:\n    # Thông báo nếu có cột thiếu\n    print(f\"Warning: Columns for {', '.join(missing_columns)} are missing in the DataFrame.\")\n    for col in missing_columns:\n        df[col] = 0  # Thêm cột thiếu vào DataFrame với giá trị 0\n\n# Lọc các cột liên quan đến sức khỏe và chấn thương của các cơ quan\ncorrelation_df = df[organ_columns + injury_columns]\n\n# Tính toán ma trận tương quan giữa các cột\ncorrelation_matrix = correlation_df.corr()\n\n# Tạo heatmap để phân tích mối tương quan giữa sức khỏe và tình trạng chấn thương của các cơ quan\nfig = px.imshow(\n    correlation_matrix,\n    x=correlation_df.columns,\n    y=correlation_df.columns,\n    labels=dict(x='Organ', y='Organ', color='Correlation'),\n    title='Correlation Between Organ Health and Injury Status',\n)\n\n# Hiển thị heatmap\nfig.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:47:33.537455Z","iopub.status.idle":"2024-11-23T19:47:33.537756Z","shell.execute_reply.started":"2024-11-23T19:47:33.537617Z","shell.execute_reply":"2024-11-23T19:47:33.537632Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Summary statistics for relevant variables\nstyled_data = df.describe().style\\\n.background_gradient(cmap='coolwarm')\\\n.set_properties(**{'text-align':'center','border':'1px solid black'})\n\n# display styled data\ndisplay(styled_data)","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:47:33.539097Z","iopub.status.idle":"2024-11-23T19:47:33.539361Z","shell.execute_reply.started":"2024-11-23T19:47:33.539234Z","shell.execute_reply":"2024-11-23T19:47:33.539246Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install matplotlib seaborn\n","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:47:33.541049Z","iopub.status.idle":"2024-11-23T19:47:33.541466Z","shell.execute_reply.started":"2024-11-23T19:47:33.541246Z","shell.execute_reply":"2024-11-23T19:47:33.541267Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\n\n# pass list of tick positions to the set_xticks() function. \n# pass the following list of tick positions to the set_xticks() function in the counts plot loop\n\ndef generate_counts_and_percentages(df, categorical_columns):\n  \"\"\"Counts and percentages for categorical variables in a DataFrame, and plot the counts and percentages.\n\n  Args:\n    df: DataFrame.\n    categorical_columns: column names for the categorical variables.\n\n  Returns:\n    None.\n  \"\"\"\n\n  # Handle null values.\n  df = df.dropna(subset=categorical_columns)\n\n  # counts.\n  counts = df[categorical_columns].apply(pd.Series.value_counts)\n\n  # percentages.\n  percentages = (counts / df.shape[0]) * 100\n\n  # Set color scheme.\n  colors = ['#007bff', '#ffa500']\n\n  # Plot counts.\n  fig, axes = plt.subplots(1, len(categorical_columns), figsize=(15, 7))\n  for i, column in enumerate(categorical_columns):\n    ax = axes[i]\n    ax.bar(counts.index.to_list(), counts[column].to_list(), color=colors[0])\n    ax.set_title(column, fontsize=12)\n    ax.set_xticks(range(len(counts.index)))\n    ax.tick_params(labelsize=10)\n    ax.grid(True)\n\n  # Plot percentages.\n  fig, axes = plt.subplots(1, len(categorical_columns), figsize=(15, 7))\n  for i, column in enumerate(categorical_columns):\n    ax = axes[i]\n    ax.pie(percentages[column].to_list(), labels=percentages.index.to_list(), autopct='%1.1f%%', startangle=140, colors=colors)\n    ax.set_title(column, fontsize=12)\n    ax.axis('equal')\n    ax.legend(fontsize=10)\n    ax.grid(True)\n\n  plt.suptitle('Counts and Percentages for Categorical Variables', fontsize=14)\n  plt.show()\n\ncategorical_columns = [\n  'bowel_healthy', 'bowel_injury', 'extravasation_healthy', 'extravasation_injury',\n  'kidney_healthy', 'kidney_low', 'kidney_high', 'liver_healthy', 'liver_low', 'liver_high',\n  'spleen_healthy', 'spleen_low', 'spleen_high', 'any_injury'\n]\n\ngenerate_counts_and_percentages(df, categorical_columns)","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:47:33.542808Z","iopub.status.idle":"2024-11-23T19:47:33.543231Z","shell.execute_reply.started":"2024-11-23T19:47:33.543002Z","shell.execute_reply":"2024-11-23T19:47:33.54304Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport plotly.graph_objects as go\nfrom sklearn.datasets import make_classification\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestClassifier, GradientBoostingClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import SVC\nfrom sklearn.metrics import accuracy_score, confusion_matrix","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:47:33.54468Z","iopub.status.idle":"2024-11-23T19:47:33.545103Z","shell.execute_reply.started":"2024-11-23T19:47:33.544874Z","shell.execute_reply":"2024-11-23T19:47:33.544894Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Visualize a sample image\ndef plot_dicom_image(image_path):\n    ds = pydicom.dcmread(image_path)\n    plt.imshow(ds.pixel_array, cmap=plt.cm.bone)\n    plt.axis('off')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:47:33.546912Z","iopub.status.idle":"2024-11-23T19:47:33.547346Z","shell.execute_reply.started":"2024-11-23T19:47:33.547123Z","shell.execute_reply":"2024-11-23T19:47:33.547142Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.metrics import confusion_matrix\n\n# Danh sách các bộ phận cần tính ma trận nhầm lẫn\norgans = [\"Bowel\", \"Extravasation\", \"Liver\", \"Kidney\", \"Spleen\"]\n\n# Giả sử y_test và y_pred đã được chuẩn bị với nhiều nhãn cho từng bộ phận (mảng 2D)\n# Dữ liệu mẫu (cần thay thế bằng dữ liệu thực tế của bạn)\nall_targets = [\n    [\"healthy\", \"injury\", \"healthy\", \"low\", \"healthy\"],  # Nhãn thực tế cho 1 mẫu\n    [\"injury\", \"healthy\", \"healthy\", \"high\", \"injury\"],  # Nhãn thực tế cho mẫu tiếp theo\n    # Thêm các mẫu khác ở đây...\n]\nall_preds = [\n    [\"healthy\", \"injury\", \"healthy\", \"low\", \"healthy\"],  # Dự đoán cho 1 mẫu\n    [\"injury\", \"healthy\", \"healthy\", \"high\", \"healthy\"],  # Dự đoán cho mẫu tiếp theo\n    # Thêm các mẫu khác ở đây...\n]\n\n# Chuyển đổi thành mảng numpy (mảng 2D) cho tất cả các bộ phận\nall_targets_np = np.array(all_targets)  # Nhãn thực tế (2D: samples x organs)\nall_preds_np = np.array(all_preds)      # Dự đoán (2D: samples x organs)\n\n# Danh sách tất cả các nhãn có thể có cho mỗi bộ phận\nclass_labels = {\n    \"Bowel\": [\"healthy\", \"injury\"],\n    \"Extravasation\": [\"healthy\", \"injury\"],\n    \"Liver\": [\"healthy\", \"low\", \"high\"],\n    \"Kidney\": [\"healthy\", \"low\", \"high\"],\n    \"Spleen\": [\"healthy\", \"low\", \"high\"],\n}\n\n# Lặp qua từng bộ phận và tính toán ma trận nhầm lẫn\nfor i, organ in enumerate(organs):\n    # Lấy các nhãn thực tế và dự đoán cho bộ phận hiện tại\n    true_labels = all_targets_np[:, i]\n    pred_labels = all_preds_np[:, i]\n    \n    # Lấy danh sách nhãn cho bộ phận hiện tại từ class_labels\n    labels = class_labels[organ]\n    \n    # Tính toán ma trận nhầm lẫn cho từng bộ phận\n    conf_matrix = confusion_matrix(true_labels, pred_labels, labels=labels)\n    \n    # Hiển thị ma trận nhầm lẫn dưới dạng heatmap\n    plt.figure(figsize=(8, 6))\n    sns.heatmap(conf_matrix, annot=True, fmt=\"d\", cmap=\"Blues\", xticklabels=labels, yticklabels=labels)\n    plt.title(f'{organ} Confusion Matrix', fontsize=16)\n    plt.xlabel('Predicted', fontsize=12)\n    plt.ylabel('True', fontsize=12)\n    plt.show()\n\n    # In ra ma trận nhầm lẫn cho từng bộ phận\n    print(f\"{organ} Confusion Matrix:\")\n    print(conf_matrix)\n","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:47:33.548737Z","iopub.status.idle":"2024-11-23T19:47:33.549167Z","shell.execute_reply.started":"2024-11-23T19:47:33.548934Z","shell.execute_reply":"2024-11-23T19:47:33.548954Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.metrics import confusion_matrix\n\n# Danh sách các bộ phận cần tính ma trận nhầm lẫn\norgans = [\"Bowel\", \"Extravasation\", \"Liver\", \"Kidney\", \"Spleen\"]\n\n# Giả sử y_test và y_pred đã được chuẩn bị với nhiều nhãn cho từng bộ phận (mảng 2D)\n# Dữ liệu mẫu (cần thay thế bằng dữ liệu thực tế của bạn)\nall_targets = [\n    [\"healthy\", \"injury\", \"healthy\", \"low\", \"healthy\"],  # Nhãn thực tế cho 1 mẫu\n    [\"injury\", \"healthy\", \"healthy\", \"high\", \"injury\"],  # Nhãn thực tế cho mẫu tiếp theo\n    # Thêm các mẫu khác ở đây...\n]\nall_preds = [\n    [\"healthy\", \"injury\", \"healthy\", \"low\", \"healthy\"],  # Dự đoán cho 1 mẫu\n    [\"injury\", \"healthy\", \"healthy\", \"high\", \"healthy\"],  # Dự đoán cho mẫu tiếp theo\n    # Thêm các mẫu khác ở đây...\n]\n\n# Chuyển đổi thành mảng numpy (mảng 2D) cho tất cả các bộ phận\nall_targets_np = np.array(all_targets)  # Nhãn thực tế (2D: samples x organs)\nall_preds_np = np.array(all_preds)      # Dự đoán (2D: samples x organs)\n\n# Danh sách tất cả các nhãn có thể có cho mỗi bộ phận\nclass_labels = [\"healthy\", \"injury\", \"low\", \"high\", \"medium\"]  # Giả sử 5 nhãn cho mỗi bộ phận\n\n# Lặp qua từng bộ phận và tính toán ma trận nhầm lẫn 5x5\nfor i, organ in enumerate(organs):\n    # Lấy các nhãn thực tế và dự đoán cho bộ phận hiện tại\n    true_labels = all_targets_np[:, i]\n    pred_labels = all_preds_np[:, i]\n    \n    # Tính toán ma trận nhầm lẫn 5x5 cho từng bộ phận\n    conf_matrix = confusion_matrix(true_labels, pred_labels, labels=class_labels)\n    \n    # Hiển thị ma trận nhầm lẫn dưới dạng heatmap\n    plt.figure(figsize=(8, 6))\n    sns.heatmap(conf_matrix, annot=True, fmt=\"d\", cmap=\"Blues\", xticklabels=class_labels, yticklabels=class_labels)\n    plt.title(f'{organ} Confusion Matrix (5x5)', fontsize=16)\n    plt.xlabel('Predicted', fontsize=12)\n    plt.ylabel('True', fontsize=12)\n    plt.show()\n\n    # In ra ma trận nhầm lẫn cho từng bộ phận\n    print(f\"{organ} Confusion Matrix (5x5):\")\n    print(conf_matrix)\n","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:47:33.550721Z","iopub.status.idle":"2024-11-23T19:47:33.550976Z","shell.execute_reply.started":"2024-11-23T19:47:33.550853Z","shell.execute_reply":"2024-11-23T19:47:33.550865Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.metrics import confusion_matrix\n\n# Giả sử bạn có 10 nhãn để phân loại\nclass_labels = [\"class_1\", \"class_2\", \"class_3\", \"class_4\", \"class_5\", \n                \"class_6\", \"class_7\", \"class_8\", \"class_9\", \"class_10\"]\n\n# Giả sử bạn có y_test và y_pred đã được chuẩn bị (2D array)\n# Mỗi hàng là một mẫu và mỗi cột là một bộ phận, với 10 lớp nhãn\n\n# Dữ liệu mẫu (cần thay thế bằng dữ liệu thực tế của bạn)\nall_targets = [\n    [\"class_1\", \"class_2\", \"class_3\", \"class_4\", \"class_5\"],  # Nhãn thực tế cho 1 mẫu\n    [\"class_2\", \"class_1\", \"class_3\", \"class_6\", \"class_7\"],  # Nhãn thực tế cho mẫu tiếp theo\n    # Thêm các mẫu khác ở đây...\n]\n\nall_preds = [\n    [\"class_1\", \"class_2\", \"class_3\", \"class_4\", \"class_5\"],  # Dự đoán cho 1 mẫu\n    [\"class_2\", \"class_1\", \"class_3\", \"class_6\", \"class_8\"],  # Dự đoán cho mẫu tiếp theo\n    # Thêm các mẫu khác ở đây...\n]\n\n# Chuyển đổi thành mảng numpy (mảng 2D) cho tất cả các bộ phận\nall_targets_np = np.array(all_targets)  # Nhãn thực tế (2D: samples x organs)\nall_preds_np = np.array(all_preds)      # Dự đoán (2D: samples x organs)\n\n# Lặp qua từng bộ phận và tính toán ma trận nhầm lẫn 10x10\nfor i in range(all_targets_np.shape[1]):  # Duyệt qua từng bộ phận (mỗi cột trong dữ liệu)\n    # Lấy các nhãn thực tế và dự đoán cho bộ phận hiện tại\n    true_labels = all_targets_np[:, i]\n    pred_labels = all_preds_np[:, i]\n    \n    # Tính toán ma trận nhầm lẫn 10x10 cho từng bộ phận\n    conf_matrix = confusion_matrix(true_labels, pred_labels, labels=class_labels)\n    \n    # Hiển thị ma trận nhầm lẫn dưới dạng heatmap\n    plt.figure(figsize=(10, 8))  # Chỉnh kích thước biểu đồ cho dễ nhìn\n    sns.heatmap(conf_matrix, annot=True, fmt=\"d\", cmap=\"Blues\", xticklabels=class_labels, yticklabels=class_labels)\n    plt.title(f'Confusion Matrix for Organ {i+1} (10x10)', fontsize=16)\n    plt.xlabel('Predicted', fontsize=12)\n    plt.ylabel('True', fontsize=12)\n    plt.xticks(rotation=45, ha='right')  # Xoay nhãn x-axis cho dễ đọc\n    plt.yticks(rotation=45, va='top')    # Xoay nhãn y-axis cho dễ đọc\n    plt.show()\n\n    # In ra ma trận nhầm lẫn cho từng bộ phận\n    print(f\"Confusion Matrix for Organ {i+1} (10x10):\")\n    print(conf_matrix)\n","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:47:33.552147Z","iopub.status.idle":"2024-11-23T19:47:33.552428Z","shell.execute_reply.started":"2024-11-23T19:47:33.552299Z","shell.execute_reply":"2024-11-23T19:47:33.552312Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.metrics import confusion_matrix\n\n# Đọc dữ liệu từ file train.csv\ndf = pd.read_csv('/kaggle/input/rsna-atd-512x512-png-v2-dataset/train.csv')\n\n# Danh sách các cột nhãn nhị phân (10 nhãn)\nbinary_columns = [\n    'bowel_healthy', 'bowel_injury', 'extravasation_healthy', 'extravasation_injury',\n    'kidney_healthy', 'kidney_low', 'kidney_high', 'liver_healthy', 'liver_low', 'liver_high',\n    'spleen_healthy', 'spleen_low', 'spleen_high'\n]\n\n# Lấy nhãn thật từ DataFrame (các cột nhãn nhị phân)\ny_true = df[binary_columns].values\n\n# Giả sử bạn có nhãn dự đoán (y_pred). Nếu không có dữ liệu dự đoán, bạn có thể giả sử y_pred giống y_true để tạo ví dụ.\n# Đây chỉ là một giả định (hãy thay thế bằng dữ liệu thực tế của bạn)\n# Ví dụ: nếu bạn có các nhãn dự đoán từ mô hình, hãy sử dụng chúng thay vì y_true.\ny_pred = y_true  # Giả sử nhãn dự đoán giống nhãn thực tế trong ví dụ này\n\n# Tính toán ma trận nhầm lẫn (y_true vs y_pred)\nconf_matrix = confusion_matrix(y_true.flatten(), y_pred.flatten(), labels=[0, 1])\n\n# Vẽ ma trận nhầm lẫn dưới dạng heatmap với tên nhãn thay vì 0 và 1\nplt.figure(figsize=(12, 8))\nsns.heatmap(conf_matrix, annot=True, fmt=\"d\", cmap=\"Blues\", xticklabels=binary_columns, yticklabels=binary_columns)\nplt.title(\"Confusion Matrix\", fontsize=16)\nplt.xlabel('Predicted Labels', fontsize=12)\nplt.ylabel('True Labels', fontsize=12)\nplt.xticks(rotation=90)  # Xoay nhãn cột nếu cần\nplt.yticks(rotation=0)   # Giữ nhãn hàng đúng hướng\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:47:33.554765Z","iopub.status.idle":"2024-11-23T19:47:33.555061Z","shell.execute_reply.started":"2024-11-23T19:47:33.554905Z","shell.execute_reply":"2024-11-23T19:47:33.554917Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.metrics import confusion_matrix\n\n# Đọc dữ liệu từ file train.csv\ndf = pd.read_csv('/kaggle/input/rsna-2023-abdominal-trauma-detection/train_2024.csv')\n\n# Danh sách các cột nhãn nhị phân (13 nhãn)\nbinary_columns = [\n    'bowel_healthy', 'bowel_injury', 'extravasation_healthy', 'extravasation_injury',\n    'kidney_healthy', 'kidney_low', 'kidney_high', 'liver_healthy', 'liver_low', 'liver_high',\n    'spleen_healthy', 'spleen_low', 'spleen_high'\n]\n\n# Lấy nhãn thật từ DataFrame (các cột nhãn nhị phân)\ny_true = df[binary_columns].values\n\n# Giả sử bạn có nhãn dự đoán (y_pred). Nếu không có dữ liệu dự đoán, bạn có thể giả sử y_pred giống y_true để tạo ví dụ.\n# Đây chỉ là một giả định (hãy thay thế bằng dữ liệu thực tế của bạn)\ny_pred = y_true  # Giả sử nhãn dự đoán giống nhãn thực tế trong ví dụ này\n\n# Tính toán ma trận nhầm lẫn (y_true vs y_pred)\nconf_matrix = confusion_matrix(y_true.flatten(), y_pred.flatten(), labels=[0, 1])\n\n# Vẽ ma trận nhầm lẫn dưới dạng heatmap với tên nhãn thay vì 0 và 1\nplt.figure(figsize=(12, 8))\nsns.heatmap(conf_matrix, annot=True, fmt=\"d\", cmap=\"Blues\", xticklabels=binary_columns, yticklabels=binary_columns)\nplt.title(\"Confusion Matrix\", fontsize=16)\nplt.xlabel('Predicted Labels', fontsize=12)\nplt.ylabel('True Labels', fontsize=12)\nplt.xticks(rotation=90)  # Xoay nhãn cột nếu cần\nplt.yticks(rotation=0)   # Giữ nhãn hàng đúng hướng\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-23T19:47:33.556164Z","iopub.status.idle":"2024-11-23T19:47:33.556462Z","shell.execute_reply.started":"2024-11-23T19:47:33.556325Z","shell.execute_reply":"2024-11-23T19:47:33.55634Z"},"trusted":true},"outputs":[],"execution_count":null}]}