{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":52254,"databundleVersionId":9674523,"sourceType":"competition"},{"sourceId":6211844,"sourceType":"datasetVersion","datasetId":3567114}],"dockerImageVersionId":30554,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# <center>Multi-label classification of abdominal trauma from CT images</center>\n\n","metadata":{"papermill":{"duration":0.025483,"end_time":"2022-02-01T10:21:43.84374","exception":false,"start_time":"2022-02-01T10:21:43.818257","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"## To commence this project, the neccesary libraries need to be installed.\n## We would be using the **TensorFlow** framework and the pretrained **EfficientNetB1**\n","metadata":{}},{"cell_type":"code","source":"# Imporitng libraries\nimport pandas as pd\nimport os\nimport random\nimport numpy as np\nimport tensorflow as tf\nimport cv2\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom IPython.display import clear_output\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.model_selection import StratifiedKFold\n\n\nrandom.seed(42)","metadata":{"id":"fUPKjZaaJHkM","papermill":{"duration":5.021373,"end_time":"2022-02-01T10:21:48.88737","exception":false,"start_time":"2022-02-01T10:21:43.865997","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-11-14T23:44:06.832808Z","iopub.execute_input":"2024-11-14T23:44:06.833099Z","iopub.status.idle":"2024-11-14T23:44:16.231087Z","shell.execute_reply.started":"2024-11-14T23:44:06.833072Z","shell.execute_reply":"2024-11-14T23:44:16.230234Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Loading the training dataset\ntrain_img = \"/kaggle/input/rsna-atd-512x512-png-v2-dataset/train_images\"","metadata":{"execution":{"iopub.status.busy":"2024-11-14T23:44:16.233014Z","iopub.execute_input":"2024-11-14T23:44:16.233629Z","iopub.status.idle":"2024-11-14T23:44:16.238145Z","shell.execute_reply.started":"2024-11-14T23:44:16.233601Z","shell.execute_reply":"2024-11-14T23:44:16.236896Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i in os.listdir(train_img):\n    for j in os.listdir(os.path.join(train_img,i)):\n        print(len(os.listdir(os.path.join(train_img,i,j))))","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-11-14T23:44:16.239514Z","iopub.execute_input":"2024-11-14T23:44:16.240338Z","iopub.status.idle":"2024-11-14T23:44:23.571578Z","shell.execute_reply.started":"2024-11-14T23:44:16.240293Z","shell.execute_reply":"2024-11-14T23:44:23.570529Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install imbalanced-learn","metadata":{"execution":{"iopub.status.busy":"2024-11-14T23:44:23.574419Z","iopub.execute_input":"2024-11-14T23:44:23.574746Z","iopub.status.idle":"2024-11-14T23:44:35.888809Z","shell.execute_reply.started":"2024-11-14T23:44:23.574717Z","shell.execute_reply":"2024-11-14T23:44:35.88774Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Part A\n## Import and explore the data","metadata":{"papermill":{"duration":0.02193,"end_time":"2022-02-01T10:21:48.93356","exception":false,"start_time":"2022-02-01T10:21:48.91163","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Making a list containing all unique classes in the training set\n\n# Reading the training labels\ntraining_labels = pd.read_csv(\"/kaggle/input/rsna-2023-abdominal-trauma-detection/train_2024.csv\")","metadata":{"id":"PBlH8ae3LbG0","papermill":{"duration":0.069238,"end_time":"2022-02-01T10:21:49.024476","exception":false,"start_time":"2022-02-01T10:21:48.955238","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-11-14T23:44:35.890345Z","iopub.execute_input":"2024-11-14T23:44:35.890744Z","iopub.status.idle":"2024-11-14T23:44:35.912274Z","shell.execute_reply.started":"2024-11-14T23:44:35.890705Z","shell.execute_reply":"2024-11-14T23:44:35.911577Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Viewing the dataset in a structured format\ntraining_labels","metadata":{"execution":{"iopub.status.busy":"2024-11-14T23:44:35.913249Z","iopub.execute_input":"2024-11-14T23:44:35.913512Z","iopub.status.idle":"2024-11-14T23:44:35.941504Z","shell.execute_reply.started":"2024-11-14T23:44:35.913488Z","shell.execute_reply":"2024-11-14T23:44:35.940642Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Part B\n## In the next few cells, we chose to explore the dataframe to check for missing values","metadata":{}},{"cell_type":"code","source":"training_labels.columns","metadata":{"execution":{"iopub.status.busy":"2024-11-14T23:44:35.942571Z","iopub.execute_input":"2024-11-14T23:44:35.942926Z","iopub.status.idle":"2024-11-14T23:44:35.949354Z","shell.execute_reply.started":"2024-11-14T23:44:35.942891Z","shell.execute_reply":"2024-11-14T23:44:35.948418Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['patient_id'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T23:44:35.950757Z","iopub.execute_input":"2024-11-14T23:44:35.95115Z","iopub.status.idle":"2024-11-14T23:44:35.964177Z","shell.execute_reply.started":"2024-11-14T23:44:35.951117Z","shell.execute_reply":"2024-11-14T23:44:35.963369Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['bowel_healthy'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T23:44:35.9652Z","iopub.execute_input":"2024-11-14T23:44:35.965492Z","iopub.status.idle":"2024-11-14T23:44:35.972978Z","shell.execute_reply.started":"2024-11-14T23:44:35.965467Z","shell.execute_reply":"2024-11-14T23:44:35.972001Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['bowel_injury'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T23:44:35.977412Z","iopub.execute_input":"2024-11-14T23:44:35.977672Z","iopub.status.idle":"2024-11-14T23:44:35.985461Z","shell.execute_reply.started":"2024-11-14T23:44:35.977643Z","shell.execute_reply":"2024-11-14T23:44:35.984605Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['extravasation_healthy'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T23:44:35.986647Z","iopub.execute_input":"2024-11-14T23:44:35.987581Z","iopub.status.idle":"2024-11-14T23:44:35.995478Z","shell.execute_reply.started":"2024-11-14T23:44:35.987549Z","shell.execute_reply":"2024-11-14T23:44:35.994623Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['extravasation_injury'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T23:44:35.99665Z","iopub.execute_input":"2024-11-14T23:44:35.997008Z","iopub.status.idle":"2024-11-14T23:44:36.006563Z","shell.execute_reply.started":"2024-11-14T23:44:35.996976Z","shell.execute_reply":"2024-11-14T23:44:36.005706Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['kidney_healthy'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T23:44:36.007928Z","iopub.execute_input":"2024-11-14T23:44:36.008164Z","iopub.status.idle":"2024-11-14T23:44:36.016268Z","shell.execute_reply.started":"2024-11-14T23:44:36.008143Z","shell.execute_reply":"2024-11-14T23:44:36.015246Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['kidney_low'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T23:44:36.017366Z","iopub.execute_input":"2024-11-14T23:44:36.017629Z","iopub.status.idle":"2024-11-14T23:44:36.02701Z","shell.execute_reply.started":"2024-11-14T23:44:36.017606Z","shell.execute_reply":"2024-11-14T23:44:36.026114Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['kidney_high'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T23:44:36.02831Z","iopub.execute_input":"2024-11-14T23:44:36.02898Z","iopub.status.idle":"2024-11-14T23:44:36.038185Z","shell.execute_reply.started":"2024-11-14T23:44:36.028947Z","shell.execute_reply":"2024-11-14T23:44:36.037221Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['liver_healthy'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T23:44:36.039397Z","iopub.execute_input":"2024-11-14T23:44:36.039671Z","iopub.status.idle":"2024-11-14T23:44:36.048414Z","shell.execute_reply.started":"2024-11-14T23:44:36.039647Z","shell.execute_reply":"2024-11-14T23:44:36.047512Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['liver_low'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T23:44:36.049566Z","iopub.execute_input":"2024-11-14T23:44:36.049895Z","iopub.status.idle":"2024-11-14T23:44:36.058692Z","shell.execute_reply.started":"2024-11-14T23:44:36.049864Z","shell.execute_reply":"2024-11-14T23:44:36.057802Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['liver_high'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T23:44:36.059803Z","iopub.execute_input":"2024-11-14T23:44:36.060071Z","iopub.status.idle":"2024-11-14T23:44:36.069495Z","shell.execute_reply.started":"2024-11-14T23:44:36.060037Z","shell.execute_reply":"2024-11-14T23:44:36.068712Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['spleen_healthy'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T23:44:36.070694Z","iopub.execute_input":"2024-11-14T23:44:36.071038Z","iopub.status.idle":"2024-11-14T23:44:36.078143Z","shell.execute_reply.started":"2024-11-14T23:44:36.071013Z","shell.execute_reply":"2024-11-14T23:44:36.077326Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['spleen_low'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T23:44:36.079358Z","iopub.execute_input":"2024-11-14T23:44:36.079733Z","iopub.status.idle":"2024-11-14T23:44:36.087199Z","shell.execute_reply.started":"2024-11-14T23:44:36.079669Z","shell.execute_reply":"2024-11-14T23:44:36.086338Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['spleen_high'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T23:44:36.088368Z","iopub.execute_input":"2024-11-14T23:44:36.088642Z","iopub.status.idle":"2024-11-14T23:44:36.096493Z","shell.execute_reply.started":"2024-11-14T23:44:36.088613Z","shell.execute_reply":"2024-11-14T23:44:36.095694Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['any_injury'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T23:44:36.097613Z","iopub.execute_input":"2024-11-14T23:44:36.09792Z","iopub.status.idle":"2024-11-14T23:44:36.106559Z","shell.execute_reply.started":"2024-11-14T23:44:36.097897Z","shell.execute_reply":"2024-11-14T23:44:36.105658Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Using the EfficientNetB1 model and tweaking the last two layers to suit our work\n#def create_model(decay_steps=10,warmup_steps=10):\n#    base_model = tf.keras.applications.EfficientNetB1(\n#    weights= \"imagenet\", include_top=False, input_shape= (512,512,3)\n#    )\n#    num_classes=61\n\n#    x = base_model.output\n#    x = tf.keras.layers.GlobalAveragePooling2D()(x)\n#    x = tf.keras.layers.Dropout(0.2)(x)\n#    x_bowel = tf.keras.layers.Dense(32, activation='silu')(x)\n#    x_extra = tf.keras.layers.Dense(32, activation='silu')(x)\n#    x_liver = tf.keras.layers.Dense(32, activation='silu')(x)\n #   x_kidney = tf.keras.layers.Dense(32, activation='silu')(x)\n#    x_spleen = tf.keras.layers.Dense(32, activation='silu')(x)\n\n    # Define heads\n#    out_bowel = tf.keras.layers.Dense(1, name='bowel', activation='sigmoid')(x_bowel) # use sigmoid to convert predictions to [0-1]\n#    out_extra = tf.keras.layers.Dense(1, name='extra', activation='sigmoid')(x_extra) # use sigmoid to convert predictions to [0-1]\n#    out_liver = tf.keras.layers.Dense(3, name='liver', activation='softmax')(x_liver) # use softmax for the liver head\n#    out_kidney = tf.keras.layers.Dense(3, name='kidney', activation='softmax')(x_kidney) # use softmax for the kidney head\n#    out_spleen = tf.keras.layers.Dense(3, name='spleen', activation='softmax')(x_spleen) # use softmax for the spleen head\n\n#    model = tf.keras.Model(inputs = base_model.input, outputs = [out_bowel,out_extra,out_liver,out_kidney,out_spleen])\n        # Cosine Decay\n#    cosine_decay = tf.keras.optimizers.schedules.CosineDecay(\n#        initial_learning_rate=1e-4,\n#        decay_steps=decay_steps,\n#        alpha=0.0,\n        #warmup_target=1e-3,\n        #warmup_steps=warmup_steps,\n #   )\n\n    # Compile the model\n #   optimizer = tf.keras.optimizers.Adam(learning_rate=cosine_decay)\n #   loss = [\n #       tf.keras.losses.BinaryCrossentropy(),\n #       tf.keras.losses.BinaryCrossentropy(),\n #       tf.keras.losses.CategoricalCrossentropy(),\n #       tf.keras.losses.CategoricalCrossentropy(),\n #       tf.keras.losses.CategoricalCrossentropy()]\n    \n #   metrics = [\n #       [tf.keras.metrics.BinaryAccuracy(name=\"bowel_binary_accuracy\")],\n #       [tf.keras.metrics.BinaryAccuracy(name=\"extra_binary_accuracy\")],\n  #      [tf.keras.metrics.CategoricalAccuracy(name=\"liver_cat_accuracy\")],\n #       [tf.keras.metrics.CategoricalAccuracy(name=\"kidney_cat_accuracy\")],\n #       [tf.keras.metrics.CategoricalAccuracy(name=\"spleen_cat_accuracy\")]]\n    #    \"bowel\":[\"accuracy\"],\n    #    \"extra\":[\"accuracy\"],\n    #    \"liver\":[\"accuracy\"],\n    #    \"kidney\":[\"accuracy\"],\n    #    \"spleen\":[\"accuracy\"],\n    #}\n  #  print(\"[INFO] Compiling the model...\")\n #   model.compile(\n #       optimizer=optimizer,\n #     loss=loss,\n #     metrics=metrics\n #   )\n #   return model","metadata":{"execution":{"iopub.status.busy":"2024-11-14T23:44:36.108Z","iopub.execute_input":"2024-11-14T23:44:36.108353Z","iopub.status.idle":"2024-11-14T23:44:36.114928Z","shell.execute_reply.started":"2024-11-14T23:44:36.108324Z","shell.execute_reply":"2024-11-14T23:44:36.114024Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"+ Sử dụng Gradient Clipping để tránh gradient quá lớn:","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\n\ndef create_model(decay_steps=900, warmup_steps=60, fine_tune_from=100):\n    # Tải EfficientNetB1 làm base model\n    base_model = tf.keras.applications.EfficientNetB1(\n        weights=\"imagenet\", include_top=False, input_shape=(512, 512, 3)\n    )\n\n    # Fine-tune một phần của base model\n    for layer in base_model.layers[:fine_tune_from]:\n        layer.trainable = False\n\n    x = base_model.output\n\n    # Lớp Convolution với kernel 3x3 và tăng số lượng filters\n    x = tf.keras.layers.Conv2D(128, (5, 5), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.MaxPooling2D((2, 2), padding='same')(x)\n    \n    x = tf.keras.layers.Conv2D(256, (5, 5), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.MaxPooling2D((2, 2), padding='same')(x)\n\n    x = tf.keras.layers.Conv2D(512, (5, 5), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.MaxPooling2D((2, 2), padding='same')(x)\n\n    x = tf.keras.layers.Conv2D(1024, (5, 5), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.MaxPooling2D((2, 2), padding='same')(x)\n\n    # Residual Block\n    residual = x\n    x = tf.keras.layers.Conv2D(1024, (2, 2), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.Add()([x, residual])  # Skip connection\n\n    # Global Average Pooling để giảm số chiều\n    x = tf.keras.layers.GlobalAveragePooling2D()(x)\n\n    # Dense Block với dropout vừa phải\n    x = tf.keras.layers.Dropout(0.4)(x)  # Giảm tỷ lệ dropout\n\n    # Dense layers cho các đầu ra\n    x_bowel = tf.keras.layers.Dense(512, activation='relu')(x)\n    x_extra = tf.keras.layers.Dense(512, activation='relu')(x)\n    x_liver = tf.keras.layers.Dense(512, activation='relu')(x)\n    x_kidney = tf.keras.layers.Dense(512, activation='relu')(x)\n    x_spleen = tf.keras.layers.Dense(512, activation='relu')(x)\n\n    # Output Layers\n    out_bowel = tf.keras.layers.Dense(1, name='bowel', activation='sigmoid')(x_bowel)  # Bowel (binary)\n    out_extra = tf.keras.layers.Dense(1, name='extra', activation='sigmoid')(x_extra)  # Extra (binary)\n    out_liver = tf.keras.layers.Dense(3, name='liver', activation='softmax')(x_liver)  # Liver (multiclass)\n    out_kidney = tf.keras.layers.Dense(3, name='kidney', activation='softmax')(x_kidney)  # Kidney (multiclass)\n    out_spleen = tf.keras.layers.Dense(3, name='spleen', activation='softmax')(x_spleen)  # Spleen (multiclass)\n\n    # Create model\n    model = tf.keras.Model(inputs=base_model.input, outputs=[out_bowel, out_extra, out_liver, out_kidney, out_spleen])\n    \n    # Cosine Decay with a warmup phase\n    cosine_decay = tf.keras.optimizers.schedules.CosineDecayRestarts(\n        initial_learning_rate=3e-4,  # Lower initial learning rate\n        first_decay_steps=warmup_steps,\n        t_mul=1.8,\n        m_mul=0.8,\n        alpha=0.2\n    )\n\n    # Compile the model\n    optimizer = tf.keras.optimizers.Adam(learning_rate=cosine_decay)\n    loss = [\n        tf.keras.losses.BinaryCrossentropy(),\n        tf.keras.losses.BinaryCrossentropy(),\n        tf.keras.losses.CategoricalCrossentropy(),\n        tf.keras.losses.CategoricalCrossentropy(),\n        tf.keras.losses.CategoricalCrossentropy()\n    ]\n    \n    metrics = [\n        [tf.keras.metrics.BinaryAccuracy(name=\"bowel_binary_accuracy\")],\n        [tf.keras.metrics.BinaryAccuracy(name=\"extra_binary_accuracy\")],\n        [tf.keras.metrics.CategoricalAccuracy(name=\"liver_cat_accuracy\")],\n        [tf.keras.metrics.CategoricalAccuracy(name=\"kidney_cat_accuracy\")],\n        [tf.keras.metrics.CategoricalAccuracy(name=\"spleen_cat_accuracy\")]\n    ]\n    \n    print(\"[INFO] Compiling the model...\")\n    model.compile(\n        optimizer=optimizer,\n        loss=loss,\n        metrics=metrics\n    )\n    \n    return model\n","metadata":{"execution":{"iopub.status.busy":"2024-11-14T23:44:36.116323Z","iopub.execute_input":"2024-11-14T23:44:36.116582Z","iopub.status.idle":"2024-11-14T23:44:36.13757Z","shell.execute_reply.started":"2024-11-14T23:44:36.11656Z","shell.execute_reply":"2024-11-14T23:44:36.136717Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#model = create_model()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T23:44:36.138777Z","iopub.execute_input":"2024-11-14T23:44:36.139417Z","iopub.status.idle":"2024-11-14T23:44:36.147533Z","shell.execute_reply.started":"2024-11-14T23:44:36.139383Z","shell.execute_reply":"2024-11-14T23:44:36.14685Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = create_model()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T23:44:36.148597Z","iopub.execute_input":"2024-11-14T23:44:36.148893Z","iopub.status.idle":"2024-11-14T23:44:41.037629Z","shell.execute_reply.started":"2024-11-14T23:44:36.148856Z","shell.execute_reply":"2024-11-14T23:44:41.036691Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# In ra tóm tắt mô hình\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T23:44:41.039018Z","iopub.execute_input":"2024-11-14T23:44:41.03941Z","iopub.status.idle":"2024-11-14T23:44:41.86707Z","shell.execute_reply.started":"2024-11-14T23:44:41.039369Z","shell.execute_reply":"2024-11-14T23:44:41.864174Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tf.keras.utils.plot_model(\n    model,  # Sử dụng đúng tên biến của mô hình\n    to_file='model.png'\n)","metadata":{"execution":{"iopub.status.busy":"2024-11-14T23:44:41.875298Z","iopub.execute_input":"2024-11-14T23:44:41.875588Z","iopub.status.idle":"2024-11-14T23:44:44.420821Z","shell.execute_reply.started":"2024-11-14T23:44:41.875563Z","shell.execute_reply":"2024-11-14T23:44:44.419238Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels = training_labels.set_index('patient_id')\ntraining_labels.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-11-14T23:44:44.421991Z","iopub.execute_input":"2024-11-14T23:44:44.422294Z","iopub.status.idle":"2024-11-14T23:44:44.440082Z","shell.execute_reply.started":"2024-11-14T23:44:44.422266Z","shell.execute_reply":"2024-11-14T23:44:44.438783Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Part C\n## Balancing the imbalanced training dataset ","metadata":{}},{"cell_type":"code","source":"#Using histogram to get the distribution of the labels and to check if there are outliers and wrong labels\ntraining_labels.hist(figsize=(20,12),bins=2)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T23:44:44.441592Z","iopub.execute_input":"2024-11-14T23:44:44.443088Z","iopub.status.idle":"2024-11-14T23:44:46.858627Z","shell.execute_reply.started":"2024-11-14T23:44:44.443Z","shell.execute_reply":"2024-11-14T23:44:46.857722Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#images ,labels","metadata":{"execution":{"iopub.status.busy":"2024-11-14T23:44:46.85985Z","iopub.execute_input":"2024-11-14T23:44:46.860166Z","iopub.status.idle":"2024-11-14T23:44:46.864253Z","shell.execute_reply.started":"2024-11-14T23:44:46.86014Z","shell.execute_reply":"2024-11-14T23:44:46.863354Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"images = []\nlabels = []\nfor i in sorted(os.listdir(train_img)):\n    folder = os.listdir(os.path.join(train_img,i))[0]\n    file = os.listdir(os.path.join(train_img,i,folder))[0]\n    images.append(cv2.imread(os.path.join(train_img,i,folder,file), cv2.IMREAD_COLOR ))\n    labels.append(np.asarray(training_labels.loc[int(i)]))\nimages = np.asarray(images)\nlabels = np.asarray(labels)","metadata":{"execution":{"iopub.status.busy":"2024-11-14T23:44:46.865601Z","iopub.execute_input":"2024-11-14T23:44:46.865913Z","iopub.status.idle":"2024-11-14T23:44:50.239873Z","shell.execute_reply.started":"2024-11-14T23:44:46.86589Z","shell.execute_reply":"2024-11-14T23:44:50.23898Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Examples of images","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(30,12))\nfor i in range(10):\n    plt.subplot(2,5,i+1)\n    plt.imshow(images[i])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T23:44:50.241122Z","iopub.execute_input":"2024-11-14T23:44:50.241454Z","iopub.status.idle":"2024-11-14T23:44:52.445193Z","shell.execute_reply.started":"2024-11-14T23:44:50.241426Z","shell.execute_reply":"2024-11-14T23:44:52.444336Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2024-11-14T23:44:52.446497Z","iopub.execute_input":"2024-11-14T23:44:52.44684Z","iopub.status.idle":"2024-11-14T23:44:52.451311Z","shell.execute_reply.started":"2024-11-14T23:44:52.44681Z","shell.execute_reply":"2024-11-14T23:44:52.450362Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train, X_val, y_train, y_val = train_test_split(images, labels, test_size=0.25)","metadata":{"execution":{"iopub.status.busy":"2024-11-14T23:44:52.452532Z","iopub.execute_input":"2024-11-14T23:44:52.452841Z","iopub.status.idle":"2024-11-14T23:44:52.516375Z","shell.execute_reply.started":"2024-11-14T23:44:52.452816Z","shell.execute_reply":"2024-11-14T23:44:52.515542Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from imblearn.over_sampling import RandomOverSampler\n\n#oversampler = RandomOverSampler(random_state=42)\n#new_images, new_labels = oversampler.fit_resample(images, labels)","metadata":{"execution":{"iopub.status.busy":"2024-11-14T23:44:52.517551Z","iopub.execute_input":"2024-11-14T23:44:52.517851Z","iopub.status.idle":"2024-11-14T23:44:52.980917Z","shell.execute_reply.started":"2024-11-14T23:44:52.517825Z","shell.execute_reply":"2024-11-14T23:44:52.979934Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Part D\n## Training the model with the training dataset","metadata":{}},{"cell_type":"code","source":"bowel_labels = labels[:,2]\nextravasation_labels = labels[:,4]\nkidney_labels = labels[:,4:7]\nliver_labels = labels[:,7:10]\nspleen_labels = labels[:,10:13]\nany_labels = labels[:,-1]","metadata":{"execution":{"iopub.status.busy":"2024-11-14T23:44:52.982099Z","iopub.execute_input":"2024-11-14T23:44:52.983089Z","iopub.status.idle":"2024-11-14T23:44:52.988092Z","shell.execute_reply.started":"2024-11-14T23:44:52.983061Z","shell.execute_reply":"2024-11-14T23:44:52.987247Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"bowel_val = y_val[:,2]\nextravasation_val = y_val[:,4]\nkidney_val = y_val[:,4:7]\nliver_val = y_val[:,7:10]\nspleen_val = y_val[:,10:13]\nany_val = y_val[:,-1]","metadata":{"execution":{"iopub.status.busy":"2024-11-14T23:44:52.989256Z","iopub.execute_input":"2024-11-14T23:44:52.989579Z","iopub.status.idle":"2024-11-14T23:44:52.998491Z","shell.execute_reply.started":"2024-11-14T23:44:52.989546Z","shell.execute_reply":"2024-11-14T23:44:52.997607Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"bowel_train = y_train[:,2]\nextravasation_train = y_train[:,4]\nkidney_train = y_train[:,4:7]\nliver_train = y_train[:,7:10]\nspleen_train = y_train[:,10:13]\nany_train = y_train[:,-1]","metadata":{"execution":{"iopub.status.busy":"2024-11-14T23:44:52.999566Z","iopub.execute_input":"2024-11-14T23:44:52.999847Z","iopub.status.idle":"2024-11-14T23:44:53.006817Z","shell.execute_reply.started":"2024-11-14T23:44:52.999823Z","shell.execute_reply":"2024-11-14T23:44:53.005938Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#batch_size = 8\n#num_epoch = 2\n#history = model.fit(x=X_train,y=[bowel_train,extravasation_train,kidney_train,liver_train,spleen_train],batch_size=batch_size, epochs=num_epoch, verbose=1, validation_data=(X_val,[bowel_val,extravasation_val,kidney_val,liver_val,spleen_val]))\n\n\n#validation_data=(X_val,[bowel_val,extravasation_val,kidney_val,liver_val,spleen_val])","metadata":{"papermill":{"duration":1176.379334,"end_time":"2022-02-01T14:07:30.902597","exception":false,"start_time":"2022-02-01T13:47:54.523263","status":"completed"},"tags":[],"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-11-14T23:44:53.00794Z","iopub.execute_input":"2024-11-14T23:44:53.008215Z","iopub.status.idle":"2024-11-14T23:44:53.015313Z","shell.execute_reply.started":"2024-11-14T23:44:53.008192Z","shell.execute_reply":"2024-11-14T23:44:53.014464Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install keras-rectified-adam\n","metadata":{"execution":{"iopub.status.busy":"2024-11-14T23:44:53.016465Z","iopub.execute_input":"2024-11-14T23:44:53.0168Z","iopub.status.idle":"2024-11-14T23:45:06.802385Z","shell.execute_reply.started":"2024-11-14T23:44:53.016776Z","shell.execute_reply":"2024-11-14T23:45:06.80133Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.callbacks import EarlyStopping, ModelCheckpoint\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nimport numpy as np\nimport os\nimport cv2\n\n# Hàm tính F1 Score tùy chỉnh\ndef f1_score_metric(y_true, y_pred):\n    true_positives = tf.reduce_sum(tf.cast(tf.logical_and(tf.equal(y_true, 1), tf.equal(y_pred, 1)), tf.float32))\n    false_positives = tf.reduce_sum(tf.cast(tf.logical_and(tf.equal(y_true, 0), tf.equal(y_pred, 1)), tf.float32))\n    false_negatives = tf.reduce_sum(tf.cast(tf.logical_and(tf.equal(y_true, 1), tf.equal(y_pred, 0)), tf.float32))\n    \n    precision = true_positives / (true_positives + false_positives + tf.keras.backend.epsilon())\n    recall = true_positives / (true_positives + false_negatives + tf.keras.backend.epsilon())\n    \n    # Tính F1 score\n    f1 = 2 * (precision * recall) / (precision + recall + tf.keras.backend.epsilon())\n    return f1\n\n# Tạo mô hình với các lớp như EfficientNetB1, Conv2D và các lớp FC\ndef create_model(decay_steps=900, warmup_steps=60, fine_tune_from=100):\n    # Tải EfficientNetB1 làm base model\n    base_model = tf.keras.applications.EfficientNetB1(\n        weights=\"imagenet\", include_top=False, input_shape=(512, 512, 3)\n    )\n\n    # Fine-tune một phần của base model\n    for layer in base_model.layers[:fine_tune_from]:\n        layer.trainable = False\n\n    x = base_model.output\n\n    # Các lớp Convolution\n    x = tf.keras.layers.Conv2D(128, (5, 5), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.MaxPooling2D((2, 2), padding='same')(x)\n    \n    x = tf.keras.layers.Conv2D(256, (5, 5), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.MaxPooling2D((2, 2), padding='same')(x)\n\n    x = tf.keras.layers.Conv2D(512, (5, 5), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.MaxPooling2D((2, 2), padding='same')(x)\n\n    x = tf.keras.layers.Conv2D(1024, (5, 5), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.MaxPooling2D((2, 2), padding='same')(x)\n\n    # Residual Block\n    residual = x\n    x = tf.keras.layers.Conv2D(1024, (2, 2), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.Add()([x, residual])  # Skip connection\n\n    # Global Average Pooling\n    x = tf.keras.layers.GlobalAveragePooling2D()(x)\n\n    # Dense layer với dropout\n    x = tf.keras.layers.Dropout(0.4)(x)\n\n    # Các lớp Dense cho các đầu ra\n    x_bowel = tf.keras.layers.Dense(512, activation='relu')(x)\n    x_extra = tf.keras.layers.Dense(512, activation='relu')(x)\n    x_liver = tf.keras.layers.Dense(512, activation='relu')(x)\n    x_kidney = tf.keras.layers.Dense(512, activation='relu')(x)\n    x_spleen = tf.keras.layers.Dense(512, activation='relu')(x)\n\n    # Output Layers\n    out_bowel = tf.keras.layers.Dense(1, name='bowel', activation='sigmoid')(x_bowel)\n    out_extra = tf.keras.layers.Dense(1, name='extra', activation='sigmoid')(x_extra)\n    out_liver = tf.keras.layers.Dense(3, name='liver', activation='softmax')(x_liver)\n    out_kidney = tf.keras.layers.Dense(3, name='kidney', activation='softmax')(x_kidney)\n    out_spleen = tf.keras.layers.Dense(3, name='spleen', activation='softmax')(x_spleen)\n\n    # Tạo mô hình\n    model = tf.keras.Model(inputs=base_model.input, outputs=[out_bowel, out_extra, out_liver, out_kidney, out_spleen])\n    \n    # Cosine Decay Learning Rate\n    cosine_decay = tf.keras.optimizers.schedules.CosineDecayRestarts(\n        initial_learning_rate=3e-4,\n        first_decay_steps=warmup_steps,\n        t_mul=1.8,\n        m_mul=0.8,\n        alpha=0.2\n    )\n\n    # Compile the model\n    optimizer = tf.keras.optimizers.Adam(learning_rate=cosine_decay)\n    loss = [\n        tf.keras.losses.BinaryCrossentropy(),\n        tf.keras.losses.BinaryCrossentropy(),\n        tf.keras.losses.CategoricalCrossentropy(),\n        tf.keras.losses.CategoricalCrossentropy(),\n        tf.keras.losses.CategoricalCrossentropy()\n    ]\n    \n    metrics = [\n        # Metrics for Bowel (binary)\n        [tf.keras.metrics.BinaryAccuracy(name=\"bowel_binary_accuracy\"),\n         f1_score_metric,  # Sử dụng F1 Score tùy chỉnh\n         tf.keras.metrics.AUC(name=\"bowel_auc\"),\n         tf.keras.metrics.Recall(name=\"bowel_recall\"),\n         tf.keras.metrics.Precision(name=\"bowel_precision\"),  # Thêm Precision\n         tf.keras.metrics.TruePositives(name=\"bowel_tp\"),\n         tf.keras.metrics.FalseNegatives(name=\"bowel_fn\"),\n         tf.keras.metrics.FalsePositives(name=\"bowel_fp\"),\n         tf.keras.metrics.TrueNegatives(name=\"bowel_tn\")],\n        \n        # Metrics for Extravasation (binary)\n        [tf.keras.metrics.BinaryAccuracy(name=\"extra_binary_accuracy\"),\n         f1_score_metric,  # Sử dụng F1 Score tùy chỉnh\n         tf.keras.metrics.AUC(name=\"extra_auc\"),\n         tf.keras.metrics.Recall(name=\"extra_recall\"),\n         tf.keras.metrics.Precision(name=\"extra_precision\"),  # Thêm Precision\n         tf.keras.metrics.TruePositives(name=\"extra_tp\"),\n         tf.keras.metrics.FalseNegatives(name=\"extra_fn\"),\n         tf.keras.metrics.FalsePositives(name=\"extra_fp\"),\n         tf.keras.metrics.TrueNegatives(name=\"extra_tn\")],\n        \n        # Metrics for Liver (multiclass)\n        [tf.keras.metrics.CategoricalAccuracy(name=\"liver_cat_accuracy\"),\n         f1_score_metric,  # Sử dụng F1 Score tùy chỉnh\n         tf.keras.metrics.AUC(name=\"liver_auc\"),\n         tf.keras.metrics.Recall(name=\"liver_recall\"),\n         tf.keras.metrics.Precision(name=\"liver_precision\")],  # Thêm Precision\n        \n        # Metrics for Kidney (multiclass)\n        [tf.keras.metrics.CategoricalAccuracy(name=\"kidney_cat_accuracy\"),\n         f1_score_metric,  # Sử dụng F1 Score tùy chỉnh\n         tf.keras.metrics.AUC(name=\"kidney_auc\"),\n         tf.keras.metrics.Recall(name=\"kidney_recall\"),\n         tf.keras.metrics.Precision(name=\"kidney_precision\")],  # Thêm Precision\n        \n        # Metrics for Spleen (multiclass)\n        [tf.keras.metrics.CategoricalAccuracy(name=\"spleen_cat_accuracy\"),\n         f1_score_metric,  # Sử dụng F1 Score tùy chỉnh\n         tf.keras.metrics.AUC(name=\"spleen_auc\"),\n         tf.keras.metrics.Recall(name=\"spleen_recall\"),\n         tf.keras.metrics.Precision(name=\"spleen_precision\")]  # Thêm Precision\n    ]\n    \n    model.compile(optimizer=optimizer, loss=loss, metrics=metrics)\n    \n    return model\n\n\n# Load images and labels (dữ liệu và nhãn đã chuẩn bị)\nimages = []\nlabels = []\nfor i in sorted(os.listdir(train_img)):\n    folder = os.listdir(os.path.join(train_img, i))[0]\n    file = os.listdir(os.path.join(train_img, i, folder))[0]\n    images.append(cv2.imread(os.path.join(train_img, i, folder, file), cv2.IMREAD_COLOR))\n    labels.append(np.asarray(training_labels.loc[int(i)]))\nimages = np.asarray(images)\nlabels = np.asarray(labels)\n\n# Split data\nX_train, X_val, y_train, y_val = train_test_split(images, labels, test_size=0.25)\n\n# Prepare labels for each class\nbowel_labels = labels[:, 2]\nextravasation_labels = labels[:, 4]\nkidney_labels = labels[:, 4:7]\nliver_labels = labels[:, 7:10]\nspleen_labels = labels[:, 10:13]\nany_labels = labels[:, -1]\n\nbowel_val = y_val[:, 2]\nextravasation_val = y_val[:, 4]\nkidney_val = y_val[:, 4:7]\nliver_val = y_val[:, 7:10]\nspleen_val = y_val[:, 10:13]\nany_val = y_val[:, -1]\n\n# KFold Cross Validation\nkf = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n\n# Store results for each fold\nfold_results = []\n\n# Data Augmentation\ntrain_datagen = ImageDataGenerator(\n    rotation_range=40,\n    width_shift_range=0.2,\n    height_shift_range=0.2,\n    shear_range=0.2,\n    zoom_range=0.2,\n    horizontal_flip=True,\n    fill_mode='nearest'\n)\n\n# Validation data generator (no augmentation)\nval_datagen = ImageDataGenerator()\n\nfor fold, (train_index, val_index) in enumerate(kf.split(images, any_labels)):\n    print(f\"Fold {fold}\")  # Changed to print fold number from 0 to 4\n\n    # Split data into train and validation according to fold\n    X_train_fold, X_val_fold = images[train_index], images[val_index]\n    y_train_fold, y_val_fold = labels[train_index], labels[val_index]\n\n    # Prepare labels for each class for the fold\n    bowel_train = y_train_fold[:, 2]\n    extravasation_train = y_train_fold[:, 4]\n    kidney_train = y_train_fold[:, 4:7]\n    liver_train = y_train_fold[:, 7:10]\n    spleen_train = y_train_fold[:, 10:13]\n\n    bowel_val = y_val_fold[:, 2]\n    extravasation_val = y_val_fold[:, 4]\n    kidney_val = y_val_fold[:, 4:7]\n    liver_val = y_val_fold[:, 7:10]\n    spleen_val = y_val_fold[:, 10:13]\n\n    # Create a new model for each fold\n    model = create_model()\n\n    # EarlyStopping and ModelCheckpoint\n    early_stopping = EarlyStopping(monitor='val_loss', patience=10, restore_best_weights=True)\n    checkpoint = ModelCheckpoint(f'best_model_fold_{fold}.h5', save_best_only=True, monitor='val_loss')\n\n    # Train the model\n    batch_size = 16\n    num_epoch = 200  # You can modify this value based on the training time and model convergence\n    \n    history = model.fit(\n        x=X_train_fold,\n        y=[bowel_train, extravasation_train, kidney_train, liver_train, spleen_train],\n        batch_size=batch_size,\n        epochs=num_epoch,\n        verbose=1,\n        validation_data=(X_val_fold, [bowel_val, extravasation_val, kidney_val, liver_val, spleen_val])\n    )\n\n    # Store results for this fold\n    fold_results.append(history.history)\n\n# Check results\nfor fold, result in enumerate(fold_results):\n    print(f\"Fold {fold} Results:\")  # Changed to print fold number from 0 to 4\n    for key in result.keys():\n        print(f\"{key}: {result[key][-1]}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-11-14T23:45:06.803999Z","iopub.execute_input":"2024-11-14T23:45:06.804342Z","iopub.status.idle":"2024-11-15T01:49:01.027584Z","shell.execute_reply.started":"2024-11-14T23:45:06.804312Z","shell.execute_reply":"2024-11-15T01:49:01.026631Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.datasets import make_classification\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.ensemble import RandomForestClassifier\nimport numpy as np\n\n# Tạo dữ liệu mẫu\nX, y = make_classification(n_samples=1000, n_features=20, n_classes=2, random_state=42)\n\n# Khởi tạo model (có thể thay bằng model của bạn)\nmodel = RandomForestClassifier(random_state=42)\n\n# Khởi tạo các danh sách để lưu kết quả từ mỗi fold\naccuracy_scores = []\nprecision_scores = []\nrecall_scores = []\nf1_scores = []\n\n# Sử dụng StratifiedKFold cho Cross-Validation\nkf = StratifiedKFold(n_splits=5)\n\nfor train_index, val_index in kf.split(X, y):\n    X_train, X_val = X[train_index], X[val_index]\n    y_train, y_val = y[train_index], y[val_index]\n    \n    # Huấn luyện model trên tập train\n    model.fit(X_train, y_train)\n    \n    # Dự đoán trên tập validation\n    y_pred = model.predict(X_val)\n    \n    # Tính toán các chỉ số cho fold hiện tại\n    accuracy_scores.append(accuracy_score(y_val, y_pred))\n    precision_scores.append(precision_score(y_val, y_pred, average='weighted'))\n    recall_scores.append(recall_score(y_val, y_pred, average='weighted'))\n    f1_scores.append(f1_score(y_val, y_pred, average='weighted'))\n\n# Tính trung bình các chỉ số qua các fold\navg_accuracy = np.mean(accuracy_scores)\navg_precision = np.mean(precision_scores)\navg_recall = np.mean(recall_scores)\navg_f1 = np.mean(f1_scores)\n\nprint(f\"Average Accuracy: {avg_accuracy:.4f}\")\nprint(f\"Average Precision: {avg_precision:.4f}\")\nprint(f\"Average Recall: {avg_recall:.4f}\")\nprint(f\"Average F1 Score: {avg_f1:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T01:49:01.02939Z","iopub.execute_input":"2024-11-15T01:49:01.03025Z","iopub.status.idle":"2024-11-15T01:49:02.935237Z","shell.execute_reply.started":"2024-11-15T01:49:01.030213Z","shell.execute_reply":"2024-11-15T01:49:02.934227Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Store results for validation accuracy across all folds for all 5 metrics (bowel, extra, liver, kidney, spleen)\navg_val_acc = {\n    'bowel': 0,\n    'extra': 0,\n    'liver': 0,\n    'kidney': 0,\n    'spleen': 0\n}\n\n# Số lượng folds\nnum_folds = len(fold_results)\n\n# Lặp qua từng fold để tính tổng validation accuracy cho tất cả các lớp\nfor fold, result in enumerate(fold_results):\n    # Extract validation accuracy for each class (bowel, extravasation, liver, kidney, spleen)\n    val_bowel_acc = result['val_bowel_bowel_binary_accuracy'][-1]  # Validation accuracy for bowel\n    val_extra_acc = result['val_extra_extra_binary_accuracy'][-1]  # Validation accuracy for extra\n    val_liver_acc = result['val_liver_liver_cat_accuracy'][-1]  # Validation accuracy for liver\n    val_kidney_acc = result['val_kidney_kidney_cat_accuracy'][-1]  # Validation accuracy for kidney\n    val_spleen_acc = result['val_spleen_spleen_cat_accuracy'][-1]  # Validation accuracy for spleen\n    \n    # Cộng dồn các giá trị validation accuracy\n    avg_val_acc['bowel'] += val_bowel_acc\n    avg_val_acc['extra'] += val_extra_acc\n    avg_val_acc['liver'] += val_liver_acc\n    avg_val_acc['kidney'] += val_kidney_acc\n    avg_val_acc['spleen'] += val_spleen_acc\n\n# Tính trung bình validation accuracy cho tất cả các lớp\navg_val_acc['bowel'] /= num_folds\navg_val_acc['extra'] /= num_folds\navg_val_acc['liver'] /= num_folds\navg_val_acc['kidney'] /= num_folds\navg_val_acc['spleen'] /= num_folds\n\n# In kết quả validation accuracy trung bình cho mỗi lớp\nprint(\"\\nAverage Validation Accuracy across all folds for each class:\")\nprint(f\"Bowel Validation Accuracy: {avg_val_acc['bowel']:.4f}\")\nprint(f\"Extravasation Validation Accuracy: {avg_val_acc['extra']:.4f}\")\nprint(f\"Liver Validation Accuracy: {avg_val_acc['liver']:.4f}\")\nprint(f\"Kidney Validation Accuracy: {avg_val_acc['kidney']:.4f}\")\nprint(f\"Spleen Validation Accuracy: {avg_val_acc['spleen']:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T01:49:02.936491Z","iopub.execute_input":"2024-11-15T01:49:02.936837Z","iopub.status.idle":"2024-11-15T01:49:02.94671Z","shell.execute_reply.started":"2024-11-15T01:49:02.936809Z","shell.execute_reply":"2024-11-15T01:49:02.945735Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\n# Giả sử 'fold_results' là danh sách chứa các kết quả huấn luyện của từng fold\n# Mỗi phần tử trong 'fold_results' là dictionary chứa thông tin về các metric cho từng fold.\n\n# Danh sách các metrics mà bạn muốn theo dõi \nmetrics = [\n    'val_bowel_bowel_binary_accuracy', 'val_extra_extra_binary_accuracy',\n    'val_liver_liver_cat_accuracy', 'val_kidney_kidney_cat_accuracy', 'val_spleen_spleen_cat_accuracy',\n    'val_bowel_loss', 'val_extra_loss', 'val_liver_loss', 'val_kidney_loss', 'val_spleen_loss'\n]\n\n# Lưu kết quả tốt nhất của các metrics\nmetric_best_values = {metric: [] for metric in metrics}  # Khởi tạo dictionary để lưu giá trị tốt nhất của từng metric\n\n# Duyệt qua từng fold trong fold_results\nfor fold_result in fold_results:\n    for metric in metrics:\n        # Lấy giá trị của metric từ fold_result\n        if metric in fold_result:\n            metric_values = np.asarray(fold_result[metric])  # Lấy giá trị metric cho fold này\n            \n            # Tìm giá trị tốt nhất (max đối với accuracy, min đối với loss)\n            best_value = np.max(metric_values) if 'accuracy' in metric else np.min(metric_values)\n            \n            # Thêm giá trị tốt nhất vào danh sách của metric\n            metric_best_values[metric].append(best_value)\n\n# Tính và in trung bình tốt nhất cho các metrics (loại bỏ val_loss)\nfor metric, best_values in metric_best_values.items():\n    avg_best_value = np.mean(best_values)  # Tính trung bình của các giá trị tốt nhất từ các fold\n    print(f\"Best average for {metric}: {avg_best_value:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T01:49:02.94801Z","iopub.execute_input":"2024-11-15T01:49:02.948256Z","iopub.status.idle":"2024-11-15T01:49:02.962096Z","shell.execute_reply.started":"2024-11-15T01:49:02.948235Z","shell.execute_reply":"2024-11-15T01:49:02.961064Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Store results for the best validation accuracy across all folds for all 5 metrics (bowel, extra, liver, kidney, spleen)\navg_best_val_acc = {\n    'bowel': 0,\n    'extra': 0,\n    'liver': 0,\n    'kidney': 0,\n    'spleen': 0\n}\n\n# Số lượng folds\nnum_folds = len(fold_results)\n\n# Lặp qua từng fold để tìm validation accuracy tốt nhất cho tất cả các lớp\nfor fold, result in enumerate(fold_results):\n    # Find the best validation accuracy for each class (best val_acc across epochs)\n    best_val_bowel_acc = max(result['val_bowel_bowel_binary_accuracy'])  # Best validation accuracy for bowel\n    best_val_extra_acc = max(result['val_extra_extra_binary_accuracy'])  # Best validation accuracy for extra\n    best_val_liver_acc = max(result['val_liver_liver_cat_accuracy'])  # Best validation accuracy for liver\n    best_val_kidney_acc = max(result['val_kidney_kidney_cat_accuracy'])  # Best validation accuracy for kidney\n    best_val_spleen_acc = max(result['val_spleen_spleen_cat_accuracy'])  # Best validation accuracy for spleen\n    \n    # Cộng dồn các giá trị validation accuracy tốt nhất\n    avg_best_val_acc['bowel'] += best_val_bowel_acc\n    avg_best_val_acc['extra'] += best_val_extra_acc\n    avg_best_val_acc['liver'] += best_val_liver_acc\n    avg_best_val_acc['kidney'] += best_val_kidney_acc\n    avg_best_val_acc['spleen'] += best_val_spleen_acc\n\n# Tính trung bình validation accuracy tốt nhất cho tất cả các lớp\navg_best_val_acc['bowel'] /= num_folds\navg_best_val_acc['extra'] /= num_folds\navg_best_val_acc['liver'] /= num_folds\navg_best_val_acc['kidney'] /= num_folds\navg_best_val_acc['spleen'] /= num_folds\n\n# In kết quả validation accuracy tốt nhất trung bình cho mỗi lớp\nprint(\"\\nAverage Best Validation Accuracy across all folds for each class:\")\nprint(f\"Bowel Best Validation Accuracy: {avg_best_val_acc['bowel']:.4f}\")\nprint(f\"Extravasation Best Validation Accuracy: {avg_best_val_acc['extra']:.4f}\")\nprint(f\"Liver Best Validation Accuracy: {avg_best_val_acc['liver']:.4f}\")\nprint(f\"Kidney Best Validation Accuracy: {avg_best_val_acc['kidney']:.4f}\")\nprint(f\"Spleen Best Validation Accuracy: {avg_best_val_acc['spleen']:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T01:49:02.964322Z","iopub.execute_input":"2024-11-15T01:49:02.96465Z","iopub.status.idle":"2024-11-15T01:49:02.975671Z","shell.execute_reply.started":"2024-11-15T01:49:02.964624Z","shell.execute_reply":"2024-11-15T01:49:02.974729Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Store results for the best validation accuracy, accuracy, loss, F1 score, recall, precision across all folds for all 5 metrics (bowel, extra, liver, kidney, spleen)\navg_results = {\n    'acc': 0,\n    'f1': 0,\n    'recall': 0,\n    'precision': 0,\n    'loss': 0,\n    'val_acc': 0\n}\n\n# Số lượng folds\nnum_folds = len(fold_results)\n\n# Lặp qua từng fold để tìm các chỉ số tốt nhất cho tất cả các lớp\nfor fold, result in enumerate(fold_results):\n    # Find the best validation accuracy for each class (best val_acc across epochs)\n    best_val_bowel_acc = max(result['val_bowel_bowel_binary_accuracy'])  # Best validation accuracy for bowel\n    best_val_extra_acc = max(result['val_extra_extra_binary_accuracy'])  # Best validation accuracy for extra\n    best_val_liver_acc = max(result['val_liver_liver_cat_accuracy'])  # Best validation accuracy for liver\n    best_val_kidney_acc = max(result['val_kidney_kidney_cat_accuracy'])  # Best validation accuracy for kidney\n    best_val_spleen_acc = max(result['val_spleen_spleen_cat_accuracy'])  # Best validation accuracy for spleen\n\n    # Final metrics for each class\n    bowel_acc = result['bowel_bowel_binary_accuracy'][-1]  # Final accuracy for bowel\n    extra_acc = result['extra_extra_binary_accuracy'][-1]  # Final accuracy for extra\n    liver_acc = result['liver_liver_cat_accuracy'][-1]  # Final accuracy for liver\n    kidney_acc = result['kidney_kidney_cat_accuracy'][-1]  # Final accuracy for kidney\n    spleen_acc = result['spleen_spleen_cat_accuracy'][-1]  # Final accuracy for spleen\n    \n    bowel_loss = result['bowel_loss'][-1]  # Final loss for bowel\n    extra_loss = result['extra_loss'][-1]  # Final loss for extra\n    liver_loss = result['liver_loss'][-1]  # Final loss for liver\n    kidney_loss = result['kidney_loss'][-1]  # Final loss for kidney\n    spleen_loss = result['spleen_loss'][-1]  # Final loss for spleen\n    \n    bowel_f1 = result['bowel_f1_score_metric'][-1]  # Final F1 score for bowel\n    extra_f1 = result['extra_f1_score_metric'][-1]  # Final F1 score for extra\n    liver_f1 = result['liver_f1_score_metric'][-1]  # Final F1 score for liver\n    kidney_f1 = result['kidney_f1_score_metric'][-1]  # Final F1 score for kidney\n    spleen_f1 = result['spleen_f1_score_metric'][-1]  # Final F1 score for spleen\n    \n    bowel_recall = result['bowel_bowel_recall'][-1]  # Final recall for bowel\n    extra_recall = result['extra_extra_recall'][-1]  # Final recall for extra\n    liver_recall = result['liver_liver_recall'][-1]  # Final recall for liver\n    kidney_recall = result['kidney_kidney_recall'][-1]  # Final recall for kidney\n    spleen_recall = result['spleen_spleen_recall'][-1]  # Final recall for spleen\n    \n    bowel_precision = result['bowel_bowel_precision'][-1]  # Final precision for bowel\n    extra_precision = result['extra_extra_precision'][-1]  # Final precision for extra\n    liver_precision = result['liver_liver_precision'][-1]  # Final precision for liver\n    kidney_precision = result['kidney_kidney_precision'][-1]  # Final precision for kidney\n    spleen_precision = result['spleen_spleen_precision'][-1]  # Final precision for spleen\n\n    # Cộng dồn các giá trị cho các chỉ số\n    avg_results['val_acc'] += (best_val_bowel_acc + best_val_extra_acc + best_val_liver_acc + best_val_kidney_acc + best_val_spleen_acc)\n    \n    avg_results['acc'] += (bowel_acc + extra_acc + liver_acc + kidney_acc + spleen_acc)\n    avg_results['f1'] += (bowel_f1 + extra_f1 + liver_f1 + kidney_f1 + spleen_f1)\n    avg_results['recall'] += (bowel_recall + extra_recall + liver_recall + kidney_recall + spleen_recall)\n    avg_results['precision'] += (bowel_precision + extra_precision + liver_precision + kidney_precision + spleen_precision)\n    \n    avg_results['loss'] += (bowel_loss + extra_loss + liver_loss + kidney_loss + spleen_loss)\n\n# Tính trung bình cho tất cả các chỉ số\navg_results['val_acc'] /= (num_folds * 5)  # Chia cho 5 vì có 5 lớp\navg_results['acc'] /= (num_folds * 5)\navg_results['f1'] /= (num_folds * 5)\navg_results['recall'] /= (num_folds * 5)\navg_results['precision'] /= (num_folds * 5)\navg_results['loss'] /= (num_folds * 5)\n\n# In kết quả trung bình cho tất cả các chỉ số (không có val_loss)\nprint(\"\\nAverage Results across all folds (Total Average for all classes):\")\nprint(f\"Average Validation Accuracy: {avg_results['val_acc']:.4f}\")\nprint(f\"Average Accuracy: {avg_results['acc']:.4f}\")\nprint(f\"Average F1 Score: {avg_results['f1']:.4f}\")\nprint(f\"Average Recall: {avg_results['recall']:.4f}\")\nprint(f\"Average Precision: {avg_results['precision']:.4f}\")\nprint(f\"Average Loss: {avg_results['loss']:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T01:49:02.976782Z","iopub.execute_input":"2024-11-15T01:49:02.977052Z","iopub.status.idle":"2024-11-15T01:49:02.996553Z","shell.execute_reply.started":"2024-11-15T01:49:02.977029Z","shell.execute_reply":"2024-11-15T01:49:02.995733Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nimport numpy as np\n\n# Giả sử bạn đã lưu các độ chính xác của mỗi fold vào fold_results\naccuracies = []\n\nfor fold_result in fold_results:\n    # Độ chính xác cho các chỉ số trong quá trình huấn luyện, ví dụ 'val_accuracy'\n    accuracies.append(fold_result['val_loss'])  # Bạn có thể thay thế 'val_accuracy' bằng các chỉ số khác nếu muốn\n\n# Tạo Boxplot\nplt.figure(figsize=(10, 6))\nsns.boxplot(data=accuracies)\nplt.title('Boxplot of Validation Accuracy across Folds')\nplt.ylabel('Validation Accuracy')\nplt.show()\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T01:49:02.997808Z","iopub.execute_input":"2024-11-15T01:49:02.998134Z","iopub.status.idle":"2024-11-15T01:49:03.325623Z","shell.execute_reply.started":"2024-11-15T01:49:02.998103Z","shell.execute_reply":"2024-11-15T01:49:03.324684Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Store results for the best validation accuracy, accuracy, loss, F1 score, recall, precision across all folds for all 5 metrics (bowel, extra, liver, kidney, spleen)\navg_results = {\n    'acc': 0,\n    'f1': 0,\n    'recall': 0,\n    'precision': 0,\n    'loss': 0,\n    'val_acc': 0\n}\n\n# Số lượng folds\nnum_folds = len(fold_results)\n\n# Lặp qua từng fold để tìm các chỉ số tốt nhất cho tất cả các lớp\nfor fold, result in enumerate(fold_results):\n    # Find the best validation accuracy for each class (best val_acc across epochs)\n    best_val_bowel_acc = max(result['val_bowel_bowel_binary_accuracy'])  # Best validation accuracy for bowel\n    best_val_extra_acc = max(result['val_extra_extra_binary_accuracy'])  # Best validation accuracy for extra\n    best_val_liver_acc = max(result['val_liver_liver_cat_accuracy'])  # Best validation accuracy for liver\n    best_val_kidney_acc = max(result['val_kidney_kidney_cat_accuracy'])  # Best validation accuracy for kidney\n    best_val_spleen_acc = max(result['val_spleen_spleen_cat_accuracy'])  # Best validation accuracy for spleen\n\n    # Final metrics for each class\n    bowel_acc = result['bowel_bowel_binary_accuracy'][-1]  # Final accuracy for bowel\n    extra_acc = result['extra_extra_binary_accuracy'][-1]  # Final accuracy for extra\n    liver_acc = result['liver_liver_cat_accuracy'][-1]  # Final accuracy for liver\n    kidney_acc = result['kidney_kidney_cat_accuracy'][-1]  # Final accuracy for kidney\n    spleen_acc = result['spleen_spleen_cat_accuracy'][-1]  # Final accuracy for spleen\n    \n    bowel_loss = result['bowel_loss'][-1]  # Final loss for bowel\n    extra_loss = result['extra_loss'][-1]  # Final loss for extra\n    liver_loss = result['liver_loss'][-1]  # Final loss for liver\n    kidney_loss = result['kidney_loss'][-1]  # Final loss for kidney\n    spleen_loss = result['spleen_loss'][-1]  # Final loss for spleen\n    \n    bowel_f1 = result['bowel_f1_score_metric'][-1]  # Final F1 score for bowel\n    extra_f1 = result['extra_f1_score_metric'][-1]  # Final F1 score for extra\n    liver_f1 = result['liver_f1_score_metric'][-1]  # Final F1 score for liver\n    kidney_f1 = result['kidney_f1_score_metric'][-1]  # Final F1 score for kidney\n    spleen_f1 = result['spleen_f1_score_metric'][-1]  # Final F1 score for spleen\n    \n    bowel_recall = result['bowel_bowel_recall'][-1]  # Final recall for bowel\n    extra_recall = result['extra_extra_recall'][-1]  # Final recall for extra\n    liver_recall = result['liver_liver_recall'][-1]  # Final recall for liver\n    kidney_recall = result['kidney_kidney_recall'][-1]  # Final recall for kidney\n    spleen_recall = result['spleen_spleen_recall'][-1]  # Final recall for spleen\n    \n    bowel_precision = result['bowel_bowel_precision'][-1]  # Final precision for bowel\n    extra_precision = result['extra_extra_precision'][-1]  # Final precision for extra\n    liver_precision = result['liver_liver_precision'][-1]  # Final precision for liver\n    kidney_precision = result['kidney_kidney_precision'][-1]  # Final precision for kidney\n    spleen_precision = result['spleen_spleen_precision'][-1]  # Final precision for spleen\n\n    # Cộng dồn các giá trị cho các chỉ số\n    avg_results['val_acc'] += (best_val_bowel_acc + best_val_extra_acc + best_val_liver_acc + best_val_kidney_acc + best_val_spleen_acc)\n    \n    avg_results['acc'] += (bowel_acc + extra_acc + liver_acc + kidney_acc + spleen_acc)\n    avg_results['f1'] += (bowel_f1 + extra_f1 + liver_f1 + kidney_f1 + spleen_f1)\n    avg_results['recall'] += (bowel_recall + extra_recall + liver_recall + kidney_recall + spleen_recall)\n    avg_results['precision'] += (bowel_precision + extra_precision + liver_precision + kidney_precision + spleen_precision)\n    \n    avg_results['loss'] += (bowel_loss + extra_loss + liver_loss + kidney_loss + spleen_loss)\n\n# Tính trung bình cho tất cả các chỉ số\navg_results['val_acc'] /= (num_folds * 5)  # Chia cho 5 vì có 5 lớp\navg_results['acc'] /= (num_folds * 5)\navg_results['f1'] /= (num_folds * 5)\navg_results['recall'] /= (num_folds * 5)\navg_results['precision'] /= (num_folds * 5)\navg_results['loss'] /= (num_folds * 5)\n\n# In kết quả trung bình cho tất cả các chỉ số (không có val_loss)\nprint(\"\\nAverage Results across all folds (Total Average for all classes):\")\nprint(f\"Average Validation Accuracy: {avg_results['val_acc']:.4f}\")\nprint(f\"Average Accuracy: {avg_results['acc']:.4f}\")\nprint(f\"Average F1 Score: {avg_results['f1']:.4f}\")\nprint(f\"Average Recall: {avg_results['recall']:.4f}\")\nprint(f\"Average Precision: {avg_results['precision']:.4f}\")\nprint(f\"Average Loss: {avg_results['loss']:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T01:49:03.328698Z","iopub.execute_input":"2024-11-15T01:49:03.329013Z","iopub.status.idle":"2024-11-15T01:49:03.347198Z","shell.execute_reply.started":"2024-11-15T01:49:03.328986Z","shell.execute_reply":"2024-11-15T01:49:03.346168Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install matplotlib scikit-learn\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T01:49:03.348481Z","iopub.execute_input":"2024-11-15T01:49:03.348834Z","iopub.status.idle":"2024-11-15T01:49:15.015154Z","shell.execute_reply.started":"2024-11-15T01:49:03.348802Z","shell.execute_reply":"2024-11-15T01:49:15.013901Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nimport numpy as np\n\n# Giả sử bạn đã lưu các độ chính xác của mỗi fold vào fold_results\naccuracies = []\n\nfor fold_result in fold_results:\n    # Độ chính xác cho các chỉ số trong quá trình huấn luyện, ví dụ 'val_accuracy'\n    accuracies.append(fold_result['val_loss'])  # Bạn có thể thay thế 'val_accuracy' bằng các chỉ số khác nếu muốn\n\n# Tạo Boxplot\nplt.figure(figsize=(10, 6))\nsns.boxplot(data=accuracies)\nplt.title('Boxplot of Validation Accuracy across Folds')\nplt.ylabel('Validation Accuracy')\nplt.show()\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T01:49:15.016847Z","iopub.execute_input":"2024-11-15T01:49:15.017237Z","iopub.status.idle":"2024-11-15T01:49:15.340994Z","shell.execute_reply.started":"2024-11-15T01:49:15.017203Z","shell.execute_reply":"2024-11-15T01:49:15.340054Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.plot(history.history[\"val_loss\"], label=\"val_loss\")\nplt.plot(history.history[\"loss\"], label=\"loss\")\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T01:59:15.706747Z","iopub.execute_input":"2024-11-15T01:59:15.707439Z","iopub.status.idle":"2024-11-15T01:59:15.978568Z","shell.execute_reply.started":"2024-11-15T01:59:15.707405Z","shell.execute_reply":"2024-11-15T01:59:15.977719Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Kiểm tra các khóa trong history để biết các metric hiện có\nprint(history.history.keys())\n\n# Lọc các metric mà bạn muốn vẽ (chỉ vẽ accuracy mà không có 'val_' trong tên)\nmetrics_to_plot = [key for key in history.history.keys() if key.endswith('accuracy') and 'val_' not in key]\n\n# Tạo đồ thị cho mỗi metric trong huấn luyện và validation\nfor metric in metrics_to_plot:\n    plt.figure(figsize=(8, 6))  # Tạo một biểu đồ mới cho mỗi metric\n    plt.plot(history.history[metric], label=f'Training {metric}', linestyle='-', marker='o')\n    \n    # Kiểm tra và vẽ độ chính xác của validation nếu có\n    val_metric = f'val_{metric}'  # Tạo tên key của metric validation\n    if val_metric in history.history:\n        plt.plot(history.history[val_metric], label=f'Validation {metric}', linestyle='--', marker='x')\n    \n    # Thêm số epoch vào trục x (tự động từ 1 đến n)\n    epochs = range(1, len(history.history[metric]) + 1)\n    \n    # Thêm ticks vào trục x với khoảng cách 25 epochs\n    step_size = 25\n    epoch_ticks = [epoch for epoch in epochs if epoch % step_size == 0]\n    plt.xticks(epoch_ticks)\n    \n    # Cài đặt các thông số cho biểu đồ\n    plt.xlabel('Epochs', fontsize=12)\n    plt.ylabel('Accuracy', fontsize=12)\n    plt.title(f'{metric.capitalize()} over Epochs', fontsize=14)\n    plt.legend(loc='upper left')\n    plt.grid(True)\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T01:59:29.367644Z","iopub.execute_input":"2024-11-15T01:59:29.368588Z","iopub.status.idle":"2024-11-15T01:59:30.924276Z","shell.execute_reply.started":"2024-11-15T01:59:29.368554Z","shell.execute_reply":"2024-11-15T01:59:30.92333Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Kiểm tra các khóa trong history\nprint(history.history.keys())\n\n# Các metric bạn muốn vẽ (chỉ vẽ các accuracy mà không có \"val_\")\nmetrics_to_plot = [key for key in history.history.keys() if key.endswith('accuracy') and 'val_' not in key]\n\n# Tạo biểu đồ cho cả huấn luyện và validation\nfor metric in metrics_to_plot:\n    plt.figure(figsize=(8, 6))  # Tạo một biểu đồ mới cho mỗi metric\n    plt.plot(history.history[metric], label=f'Training {metric}', linestyle='-', marker='o')\n    \n    # Kiểm tra và vẽ độ chính xác của validation nếu có\n    val_metric = f'val_{metric}'\n    if val_metric in history.history:\n        plt.plot(history.history[val_metric], label=f'Validation {metric}', linestyle='--', marker='x')\n    \n    # Thêm số epoch vào trục x (tự động từ 0 đến n)\n    epochs = range(1, len(history.history[metric]) + 1)\n    \n    # Thêm ticks với khoảng cách 25\n    step_size = 25\n    epoch_ticks = [epoch for epoch in epochs if epoch % step_size == 0]\n    plt.xticks(epoch_ticks)\n    \n    # Cài đặt các thông số cho biểu đồ\n    plt.xlabel('Epochs', fontsize=12)\n    plt.ylabel('Accuracy', fontsize=12)\n    plt.title(f'{metric.capitalize()} over Epochs', fontsize=14)\n    plt.legend(loc='upper left')\n    plt.grid(True)\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T01:59:34.475128Z","iopub.execute_input":"2024-11-15T01:59:34.475756Z","iopub.status.idle":"2024-11-15T01:59:36.024465Z","shell.execute_reply.started":"2024-11-15T01:59:34.475723Z","shell.execute_reply":"2024-11-15T01:59:36.023527Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history.history.keys()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T01:59:39.700662Z","iopub.execute_input":"2024-11-15T01:59:39.701627Z","iopub.status.idle":"2024-11-15T01:59:39.707784Z","shell.execute_reply.started":"2024-11-15T01:59:39.701593Z","shell.execute_reply":"2024-11-15T01:59:39.706735Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"val_acc = ['val_bowel_bowel_binary_accuracy', 'val_extra_extra_binary_accuracy', 'val_liver_liver_cat_accuracy', 'val_kidney_kidney_cat_accuracy', 'val_spleen_spleen_cat_accuracy']","metadata":{"_kg_hide-input":true,"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T01:59:42.93466Z","iopub.execute_input":"2024-11-15T01:59:42.935049Z","iopub.status.idle":"2024-11-15T01:59:42.940046Z","shell.execute_reply.started":"2024-11-15T01:59:42.935019Z","shell.execute_reply":"2024-11-15T01:59:42.938951Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Lặp qua các key trong history để vẽ accuracy\nfor i in history.history.keys():\n    if i.endswith(\"_accuracy\") and not i == \"val_accuracy\":\n        plt.plot(history.history[i], label=i)\n\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Accuracy\")\nplt.legend(loc=(1.05, 0.0))\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T02:00:04.134627Z","iopub.execute_input":"2024-11-15T02:00:04.135712Z","iopub.status.idle":"2024-11-15T02:00:04.480878Z","shell.execute_reply.started":"2024-11-15T02:00:04.135647Z","shell.execute_reply":"2024-11-15T02:00:04.479988Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i in history.history.keys():\n    if i.endswith(\"_loss\") and not i ==\"val_loss\":\n        plt.plot(history.history[i], label=i)\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Loss\")\nplt.legend(loc=(1.05,0.0))\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T02:00:08.149426Z","iopub.execute_input":"2024-11-15T02:00:08.150046Z","iopub.status.idle":"2024-11-15T02:00:08.45391Z","shell.execute_reply.started":"2024-11-15T02:00:08.150012Z","shell.execute_reply":"2024-11-15T02:00:08.452938Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i in history.history.keys():\n    if i.endswith(\"accuracy\") and not i ==\"val_accuracy\":\n        plt.plot(history.history[i], label=i)\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Loss\")\nplt.legend(loc=(1.05,0.0))\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T02:00:11.917792Z","iopub.execute_input":"2024-11-15T02:00:11.918698Z","iopub.status.idle":"2024-11-15T02:00:12.298144Z","shell.execute_reply.started":"2024-11-15T02:00:11.918641Z","shell.execute_reply":"2024-11-15T02:00:12.297208Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Lặp qua các key trong history để vẽ accuracy (loại bỏ val_accuracy)\nfor i in history.history.keys():\n    if i.endswith(\"accuracy\") and \"val_\" not in i:\n        plt.plot(history.history[i], label=i)\n\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Accuracy\")\nplt.legend(loc=(1.05, 0.0))\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T02:00:16.248075Z","iopub.execute_input":"2024-11-15T02:00:16.248471Z","iopub.status.idle":"2024-11-15T02:00:16.57186Z","shell.execute_reply.started":"2024-11-15T02:00:16.248437Z","shell.execute_reply":"2024-11-15T02:00:16.570742Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Kiểm tra các khóa trong history\nprint(history.history.keys())\n\n# Các metric bạn muốn vẽ (chỉ vẽ các accuracy mà không có \"val_\")\nmetrics_to_plot = [key for key in history.history.keys() if key.endswith('accuracy') and 'val_' not in key]\n\n# Tạo biểu đồ cho cả huấn luyện và validation\nfor metric in metrics_to_plot:\n    plt.figure(figsize=(8, 6))  # Tạo một biểu đồ mới cho mỗi metric\n    plt.plot(history.history[metric], label=f'Training {metric}')\n    \n    # Kiểm tra và vẽ độ chính xác của validation nếu có\n    val_metric = f'val_{metric}'\n    if val_metric in history.history:\n        plt.plot(history.history[val_metric], label=f'Validation {metric}')\n    \n    # Thêm số epoch vào trục x (tự động từ 0 đến n)\n    epochs = range(1, len(history.history[metric]) + 1)\n    \n    # Thêm ticks với khoảng cách 25\n    step_size = 25\n    epoch_ticks = [epoch for epoch in epochs if epoch % step_size == 0]\n    plt.xticks(epoch_ticks)\n    \n    # Cài đặt các thông số cho biểu đồ\n    plt.xlabel('Epochs')\n    plt.ylabel('Accuracy')\n    plt.title(f'{metric.capitalize()} over Epochs')\n    plt.legend(loc='upper left')\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T02:00:19.9104Z","iopub.execute_input":"2024-11-15T02:00:19.911361Z","iopub.status.idle":"2024-11-15T02:00:21.431233Z","shell.execute_reply.started":"2024-11-15T02:00:19.91132Z","shell.execute_reply":"2024-11-15T02:00:21.430162Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\n\n# Giả sử bạn có các giá trị accuracy và các metrics Ep cho từng epoch hoặc mô hình\nepochs = np.arange(1, 11)  # Ví dụ: 10 epochs\naccuracy = np.random.rand(10)  # Accuracy giả định (tạo ngẫu nhiên từ 0 đến 1)\nprecision = np.random.rand(10)  # Precision giả định (tạo ngẫu nhiên)\nrecall = np.random.rand(10)  # Recall giả định\nf1_score = np.random.rand(10)  # F1-Score giả định\n\n# Vẽ đồ thị đường cho Accuracy và các metrics Ep\nplt.figure(figsize=(10, 6))\n\n# Accuracy\nplt.plot(epochs, accuracy, label='Accuracy', color='blue', marker='o', linestyle='-', linewidth=2)\n\n# Precision\nplt.plot(epochs, precision, label='Precision', color='green', marker='s', linestyle='--', linewidth=2)\n\n# Recall\nplt.plot(epochs, recall, label='Recall', color='red', marker='^', linestyle='-.', linewidth=2)\n\n# F1-Score\nplt.plot(epochs, f1_score, label='F1-Score', color='purple', marker='x', linestyle=':', linewidth=2)\n\n# Thiết lập nhãn và tiêu đề\nplt.xlabel('Epochs', fontsize=14)\nplt.ylabel('Scores', fontsize=14)\nplt.title('Accuracy and EP (Precision, Recall, F1-Score) vs Epochs', fontsize=16)\nplt.legend(loc='upper left')\n\n# Hiển thị đồ thị\nplt.grid(True)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T02:00:27.036295Z","iopub.execute_input":"2024-11-15T02:00:27.03716Z","iopub.status.idle":"2024-11-15T02:00:27.373766Z","shell.execute_reply.started":"2024-11-15T02:00:27.037126Z","shell.execute_reply":"2024-11-15T02:00:27.372782Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.plot(history.history[\"val_loss\"], label=\"val_loss\")\nplt.plot(history.history[\"loss\"], label=\"loss\")\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T02:00:32.777275Z","iopub.execute_input":"2024-11-15T02:00:32.778146Z","iopub.status.idle":"2024-11-15T02:00:33.041473Z","shell.execute_reply.started":"2024-11-15T02:00:32.778106Z","shell.execute_reply":"2024-11-15T02:00:33.04046Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Đọc dữ liệu từ file train.csv\ndf = pd.read_csv('/kaggle/input/rsna-atd-512x512-png-v2-dataset/train.csv')\n\n# Hiển thị các cột trong DataFrame\nprint(df.columns)\n\n# Giả sử nhãn thật là các cột từ \"bowel_healthy\" đến \"spleen_high\"\nbinary_columns = [\n    'bowel_healthy', 'bowel_injury', 'extravasation_healthy', 'extravasation_injury',\n    'kidney_healthy', 'kidney_low', 'kidney_high', 'liver_healthy', 'liver_low', 'liver_high',\n    'spleen_healthy', 'spleen_low', 'spleen_high'\n]\n\n# Lấy các nhãn thật từ DataFrame (các cột nhãn)\nall_targets_np = df[binary_columns].values\n\n# In ra 5 mẫu nhãn thật đầu tiên\nprint(all_targets_np[:5])  # In 5 mẫu đầu tiên\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T02:00:36.131251Z","iopub.execute_input":"2024-11-15T02:00:36.131598Z","iopub.status.idle":"2024-11-15T02:00:36.217573Z","shell.execute_reply.started":"2024-11-15T02:00:36.131571Z","shell.execute_reply":"2024-11-15T02:00:36.216658Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Đọc dữ liệu huấn luyện từ file CSV\ntrain_df = pd.read_csv('/kaggle/input/rsna-atd-512x512-png-v2-dataset/train.csv')\n\n# Kiểm tra cấu trúc dữ liệu\ntrain_df.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T02:01:46.454073Z","iopub.execute_input":"2024-11-15T02:01:46.454846Z","iopub.status.idle":"2024-11-15T02:01:46.515604Z","shell.execute_reply.started":"2024-11-15T02:01:46.454814Z","shell.execute_reply":"2024-11-15T02:01:46.514594Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Đọc dữ liệu từ file CSV\ntrain_df = pd.read_csv('/kaggle/input/rsna-atd-512x512-png-v2-dataset/train.csv')\n\n# Danh sách các cột nhị phân (binary_columns)\nbinary_columns = [\n    'bowel_healthy', 'bowel_injury', 'extravasation_healthy', 'extravasation_injury',\n    'kidney_healthy', 'kidney_low', 'kidney_high', 'liver_healthy', 'liver_low', 'liver_high',\n    'spleen_healthy', 'spleen_low', 'spleen_high'\n]\n\n# Giả sử bạn đã có nhãn dự đoán (y_pred) từ mô hình của bạn\n# Ví dụ: Nếu bạn có các dự đoán từ mô hình (thay 'y_pred' bằng kết quả dự đoán của bạn)\n# Ở đây, ta tạo ra các nhãn giả y_pred (thay thế bằng kết quả thực tế của bạn)\ny_true = train_df[binary_columns].values  # Nhãn thật\ny_pred = np.random.randint(0, 1, size=y_true.shape)  # Giả lập nhãn dự đoán (thay bằng giá trị thực tế từ mô hình)\n\n# Vẽ ma trận nhầm lẫn cho từng lớp\nfor idx, label in enumerate(binary_columns):\n    # Tạo ma trận nhầm lẫn cho từng lớp\n    cm = confusion_matrix(y_true[:, idx], y_pred[:, idx])\n    \n    # Vẽ ma trận nhầm lẫn\n    plt.figure(figsize=(6, 5))\n    sns.heatmap(cm, annot=True, fmt='d', cmap='Blues', xticklabels=['Pred Negative', 'Pred Positive'], yticklabels=['True Negative', 'True Positive'])\n    plt.title(f\"Confusion Matrix for {label}\")\n    plt.xlabel('Predicted')\n    plt.ylabel('True')\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T02:01:49.224897Z","iopub.execute_input":"2024-11-15T02:01:49.225633Z","iopub.status.idle":"2024-11-15T02:01:52.773644Z","shell.execute_reply.started":"2024-11-15T02:01:49.225602Z","shell.execute_reply":"2024-11-15T02:01:52.772756Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Đọc dữ liệu từ file CSV\ntrain_df = pd.read_csv('/kaggle/input/rsna-atd-512x512-png-v2-dataset/train.csv')\n\n# Danh sách các cột nhị phân (binary_columns)\nbinary_columns = [\n    'bowel_healthy', 'bowel_injury', 'extravasation_healthy', 'extravasation_injury',\n    'kidney_healthy', 'kidney_low', 'kidney_high', 'liver_healthy', 'liver_low', 'liver_high',\n    'spleen_healthy', 'spleen_low', 'spleen_high'\n]\n\n# Giả sử bạn đã có nhãn dự đoán (y_pred) từ mô hình của bạn\n# Ví dụ: Nếu bạn có các dự đoán từ mô hình (thay 'y_pred' bằng kết quả dự đoán của bạn)\n# Ở đây, ta tạo ra các nhãn giả y_pred (thay thế bằng giá trị thực tế của bạn)\ny_true = train_df[binary_columns].values  # Nhãn thật\ny_pred = np.random.randint(0, 10, size=y_true.shape)  # Giả lập nhãn dự đoán (thay bằng giá trị thực tế từ mô hình)\n\n# Tạo ma trận nhầm lẫn cho từng lớp\ncm_all = []\nfor idx, label in enumerate(binary_columns):\n    # Tính ma trận nhầm lẫn cho mỗi lớp\n    cm = confusion_matrix(y_true[:, idx], y_pred[:, idx])\n    cm_all.append(cm)\n\n# Vẽ ma trận nhầm lẫn cho tất cả các lớp\nfig, axes = plt.subplots(nrows=4, ncols=4, figsize=(15, 15))\naxes = axes.flatten()\n\nfor idx, cm in enumerate(cm_all):\n    ax = axes[idx]\n    sns.heatmap(cm, annot=True, fmt='d', cmap='Blues', xticklabels=['Predicted 0', 'Predicted 1'], \n                yticklabels=['True 0', 'True 1'], ax=ax)\n    ax.set_title(f\"Confusion Matrix for {binary_columns[idx]}\")\n    ax.set_xlabel('Predicted')\n    ax.set_ylabel('True')\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T02:03:27.748819Z","iopub.execute_input":"2024-11-15T02:03:27.74923Z","iopub.status.idle":"2024-11-15T02:03:43.612785Z","shell.execute_reply.started":"2024-11-15T02:03:27.7492Z","shell.execute_reply":"2024-11-15T02:03:43.611832Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Đọc dữ liệu từ file CSV\ntrain_df = pd.read_csv('/kaggle/input/rsna-atd-512x512-png-v2-dataset/train.csv')\n\n# Danh sách các cột nhị phân (binary_columns)\nbinary_columns = [\n    'bowel_healthy', 'bowel_injury', 'extravasation_healthy', 'extravasation_injury',\n    'kidney_healthy', 'kidney_low', 'kidney_high', 'liver_healthy', 'liver_low', 'liver_high',\n    'spleen_healthy', 'spleen_low', 'spleen_high'\n]\n\n# Giả sử bạn đã có nhãn dự đoán (y_pred) từ mô hình của bạn\n# Ví dụ: Nếu bạn có các dự đoán từ mô hình (thay 'y_pred' bằng kết quả dự đoán của bạn)\n# Ở đây, ta tạo ra các nhãn giả y_pred (thay thế bằng giá trị thực tế của bạn)\ny_true = train_df[binary_columns].values  # Nhãn thật\ny_pred = np.random.randint(0, 5, size=y_true.shape)  # Giả lập nhãn dự đoán (thay bằng giá trị thực tế từ mô hình)\n\n# Tạo ma trận nhầm lẫn cho từng lớp và vẽ\nfig, axes = plt.subplots(nrows=4, ncols=4, figsize=(15, 15))\naxes = axes.flatten()\n\n# Tạo ma trận nhầm lẫn cho từng lớp\nfor idx, label in enumerate(binary_columns):\n    cm = confusion_matrix(y_true[:, idx], y_pred[:, idx])\n    \n    # Vẽ ma trận nhầm lẫn cho từng lớp\n    ax = axes[idx]\n    sns.heatmap(cm, annot=True, fmt='d', cmap='Blues', \n                xticklabels=['Predicted 0', 'Predicted 1'], \n                yticklabels=['True 0', 'True 1'], ax=ax)\n    ax.set_title(f\"Confusion Matrix for {label}\")\n    ax.set_xlabel('Predicted')\n    ax.set_ylabel('True')\n\n# Điều chỉnh layout và hiển thị\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T02:04:07.781768Z","iopub.execute_input":"2024-11-15T02:04:07.782148Z","iopub.status.idle":"2024-11-15T02:04:17.776794Z","shell.execute_reply.started":"2024-11-15T02:04:07.782112Z","shell.execute_reply":"2024-11-15T02:04:17.775783Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Đảm bảo rằng tệp CSV hoặc nguồn dữ liệu của bạn đã được nạp vào DataFrame\ndf = pd.read_csv(\"/kaggle/input/rsna-2023-abdominal-trauma-detection/train_2024.csv\")  # Thay thế bằng đường dẫn đúng\n\n# Kiểm tra các giá trị thiếu trong DataFrame\nmissing_values = df.isnull().sum()\nprint(\"Missing Values:\")\nprint(missing_values)\n\n# Danh sách các cột nhị phân cần chuyển đổi sang kiểu boolean\nbinary_columns = [\n    'bowel_healthy', 'bowel_injury', 'extravasation_healthy', 'extravasation_injury',\n    'kidney_healthy', 'kidney_low', 'kidney_high', 'liver_healthy', 'liver_low', 'liver_high',\n    'spleen_healthy', 'spleen_low', 'spleen_high'\n]\n\n# Chuyển đổi các cột nhị phân thành kiểu boolean\ndf[binary_columns] = df[binary_columns].astype(bool)\n\n# Giải quyết các vấn đề về chất lượng dữ liệu (nếu có, bạn có thể thêm các bước xử lý dữ liệu ở đây)\n\n# Hiển thị DataFrame sau khi đã xử lý\nprint(\"\\nPreprocessed DataFrame:\")\nprint(df.head())  # In ra 5 dòng đầu tiên của DataFrame đã xử lý để kiểm tra\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T02:05:41.768543Z","iopub.execute_input":"2024-11-15T02:05:41.769183Z","iopub.status.idle":"2024-11-15T02:05:41.794888Z","shell.execute_reply.started":"2024-11-15T02:05:41.769151Z","shell.execute_reply":"2024-11-15T02:05:41.793911Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Đọc tệp CSV chứa nhãn\ndf = pd.read_csv('/kaggle/input/rsna-atd-512x512-png-v2-dataset/train.csv')\n\n# Loại bỏ khoảng trắng thừa trong tên cột (nếu có)\ndf.columns = df.columns.str.strip()\n\n# Tạo danh sách các cột nhãn cho các cơ quan bạn muốn hiển thị\norgan_columns = ['kidney', 'liver', 'spleen']  # Chỉ giữ lại các cơ quan bạn muốn hiển thị\n\n# Vẽ ma trận nhầm lẫn cho các cơ quan\nfor organ in organ_columns:\n    # Đảm bảo tên cột chính xác sau khi loại bỏ khoảng trắng\n    y_true = df[f'{organ}_healthy']  # Nhãn thực tế (ví dụ: kidney_healthy)\n    y_pred = df[f'{organ}_low']  # Hoặc chọn nhãn khác như kidney_low (tùy vào mục đích của bạn)\n\n    # Tạo ma trận nhầm lẫn với 5 lớp (0 đến 4)\n    cm = confusion_matrix(y_true, y_pred, labels=[False, True, 2, 3, 4,5,6,7,8,9])  # Chỉ định 5 lớp (tùy theo dữ liệu của bạn)\n\n    # Vẽ ma trận nhầm lẫn\n    plt.figure(figsize=(10, 8))  # Thay đổi kích thước đồ thị cho phù hợp với ma trận\n    sns.heatmap(cm, annot=True, fmt='d', cmap='Blues', xticklabels=[0, 1, 2, 3, 4 , 5, 6, 7, 8, 9, 10], yticklabels=[0, 1, 2, 3, 4 , 5, 6, 7, 8, 9, 10])\n\n    # Thiết lập tiêu đề và nhãn\n    plt.title(f'Confusion Matrix for {organ.capitalize()}')\n    plt.xlabel('Predicted')\n    plt.ylabel('Actual')\n\n    # Hiển thị ma trận nhầm lẫn\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T02:05:45.358464Z","iopub.execute_input":"2024-11-15T02:05:45.358833Z","iopub.status.idle":"2024-11-15T02:05:47.061665Z","shell.execute_reply.started":"2024-11-15T02:05:45.358803Z","shell.execute_reply":"2024-11-15T02:05:47.060696Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Kiểm tra tất cả các cột trong DataFrame\nprint(df.columns)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T02:05:50.997692Z","iopub.execute_input":"2024-11-15T02:05:50.998028Z","iopub.status.idle":"2024-11-15T02:05:51.003494Z","shell.execute_reply.started":"2024-11-15T02:05:50.998002Z","shell.execute_reply":"2024-11-15T02:05:51.002375Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Đọc tệp CSV chứa nhãn\ndf = pd.read_csv('/kaggle/input/rsna-atd-512x512-png-v2-dataset/train.csv')\n\n# Loại bỏ khoảng trắng thừa trong tên cột (nếu có)\ndf.columns = df.columns.str.strip()\n\n# Tạo danh sách các cơ quan bạn muốn hiển thị\norgan_columns = ['bowel', 'extravasation', 'kidney', 'liver', 'spleen']\n\n# Vẽ ma trận nhầm lẫn cho các cơ quan\nfor organ in organ_columns:\n    # Đảm bảo tên cột chính xác sau khi loại bỏ khoảng trắng\n    y_true = df[f'{organ}_healthy']  # Nhãn thực tế (ví dụ: bowel_healthy)\n    \n    # Chọn cột dự đoán như 'injury', 'low' hoặc 'high', tùy vào mục đích của bạn\n    y_pred = df[f'{organ}_injury']  # Hoặc thay bằng 'low' hoặc 'high' tùy vào nhu cầu\n\n    # Tạo ma trận nhầm lẫn với 2 lớp (0: Healthy, 1: Injury)\n    cm_binary = confusion_matrix(y_true, y_pred, labels=[False, True])  # Chỉ so sánh Healthy vs Injury\n\n    # Vẽ ma trận nhầm lẫn 2 lớp\n    plt.figure(figsize=(10, 8))  \n    sns.heatmap(cm_binary, annot=True, fmt='d', cmap='Blues', xticklabels=['Healthy', 'Injury'], yticklabels=['Healthy', 'Injury'])\n    plt.title(f'Confusion Matrix for {organ.capitalize()} (2-class: Healthy vs Injury)')\n    plt.xlabel('Predicted')\n    plt.ylabel('Actual')\n    plt.show()\n\n    # Tạo ma trận nhầm lẫn với 10 lớp (0 đến 9) nếu dữ liệu có thể hỗ trợ\n    cm_10class = confusion_matrix(y_true, y_pred, labels=range(10))  # Sử dụng 10 lớp (0 đến 9)\n    \n    # Vẽ ma trận nhầm lẫn 10 lớp\n    plt.figure(figsize=(10, 8))  \n    sns.heatmap(cm_10class, annot=True, fmt='d', cmap='Blues', xticklabels=range(10), yticklabels=range(10))\n    plt.title(f'Confusion Matrix for {organ.capitalize()} (10-class)')\n    plt.xlabel('Predicted')\n    plt.ylabel('Actual')\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T02:05:54.704549Z","iopub.execute_input":"2024-11-15T02:05:54.704926Z","iopub.status.idle":"2024-11-15T02:05:58.131019Z","shell.execute_reply.started":"2024-11-15T02:05:54.704894Z","shell.execute_reply":"2024-11-15T02:05:58.129594Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Đọc tệp CSV chứa nhãn\ndf = pd.read_csv('/kaggle/input/rsna-atd-512x512-png-v2-dataset/train.csv')\n\n# Loại bỏ khoảng trắng thừa trong tên cột (nếu có)\ndf.columns = df.columns.str.strip()\n\n# Tạo danh sách các cột nhãn cho các cơ quan bạn muốn hiển thị\norgan_columns = ['kidney', 'liver', 'spleen']  # Các cơ quan cần hiển thị\n\n# Vẽ ma trận nhầm lẫn cho các cơ quan\nfor organ in organ_columns:\n    # Đảm bảo tên cột chính xác sau khi loại bỏ khoảng trắng\n    y_true = df[f'{organ}_healthy']  # Nhãn thực tế (ví dụ: kidney_healthy)\n    y_pred = df[f'{organ}_low']  # Hoặc chọn nhãn khác như kidney_low (tùy vào mục đích của bạn)\n\n    # Tạo ma trận nhầm lẫn với 10 lớp (0 đến 9)\n    cm = confusion_matrix(y_true, y_pred, labels=[False, True, 2, 3, 4, 5, 6, 7, 8, 9])  # Giả sử dữ liệu có lớp 0-9\n\n    # Vẽ ma trận nhầm lẫn\n    plt.figure(figsize=(10, 8))  # Thay đổi kích thước đồ thị cho phù hợp với ma trận\n    sns.heatmap(cm, annot=True, fmt='d', cmap='Blues', xticklabels=[0, 1, 2, 3, 4 , 5, 6, 7, 8, 9], yticklabels=[0, 1, 2, 3, 4 , 5, 6, 7, 8, 9])\n\n    # Thiết lập tiêu đề và nhãn\n    plt.title(f'Confusion Matrix for {organ.capitalize()}')\n    plt.xlabel('Predicted')\n    plt.ylabel('Actual')\n\n    # Hiển thị ma trận nhầm lẫn\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T02:06:03.96062Z","iopub.execute_input":"2024-11-15T02:06:03.961519Z","iopub.status.idle":"2024-11-15T02:06:05.684711Z","shell.execute_reply.started":"2024-11-15T02:06:03.961489Z","shell.execute_reply":"2024-11-15T02:06:05.683848Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Đọc tệp CSV chứa nhãn\ndf = pd.read_csv('/kaggle/input/rsna-2023-abdominal-trauma-detection/train_2024.csv')\n\n# Trích xuất nhãn thật (y_true) và nhãn dự đoán (y_pred)\n# Giả sử rằng bạn có một mô hình dự đoán sẵn có hoặc đang huấn luyện mô hình để lấy y_pred\n\n# Dữ liệu về bowel (thay 'bowel' bằng các cơ quan khác như 'extravasation', 'kidney', v.v.)\ny_true_bowel = df['bowel_healthy']  # Hoặc nếu bạn muốn nhãn về injury thì dùng 'bowel_injury'\ny_pred_bowel = df['bowel_injury']  # Đây là nhãn dự đoán mà mô hình của bạn sẽ đưa ra (giả lập)\n\n# Dự đoán giả lập cho ví dụ\n# y_pred_bowel có thể là đầu ra của mô hình dự đoán (thay 'bowel_injury' bằng giá trị thực tế của mô hình của bạn)\n# Đoạn dưới đây là giả lập, bạn sẽ thay thế bằng mô hình thực tế của mình.\n\n# Xử lý các cơ quan khác (extravasation, kidney, liver, spleen)\ny_true_extra = df['extravasation_healthy']\ny_pred_extra= df['extravasation_injury']  # Tùy thuộc vào nhãn bạn muốn sử dụng\n\n# Tạo danh sách các cột nhãn cho các cơ quan khác\norgan_columns = ['bowel', 'extravasation', 'kidney', 'liver', 'spleen']\n\n# Vẽ ma trận nhầm lẫn cho các cơ quan\nfor organ in organ_columns:\n    y_true = df[f'{organ}_healthy']\n    y_pred = df[f'{organ}_injury']  # Hoặc thay đổi cột này tùy vào nhu cầu\n\n    # Tạo ma trận nhầm lẫn với 5 lớp (0 đến 4)\n    cm = confusion_matrix(y_true, y_pred, labels=range(10))  # Sử dụng 10 lớp (0 đến 9)\n\n    # Vẽ ma trận nhầm lẫn\n    plt.figure(figsize=(10, 8))  # Thay đổi kích thước đồ thị cho phù hợp với ma trận 11x11\n    sns.heatmap(cm, annot=True, fmt='d', cmap='Blues', xticklabels=range(11), yticklabels=range(11))\n\n    # Thiết lập tiêu đề và nhãn\n    plt.title(f'Confusion Matrix for {organ.capitalize()}')\n    plt.xlabel('Predicted')\n    plt.ylabel('Actual')\n\n    # Hiển thị ma trận nhầm lẫn\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T02:06:10.977346Z","iopub.execute_input":"2024-11-15T02:06:10.977746Z","iopub.status.idle":"2024-11-15T02:06:12.253886Z","shell.execute_reply.started":"2024-11-15T02:06:10.977714Z","shell.execute_reply":"2024-11-15T02:06:12.252551Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check for missing values\nmissing_values = df.isnull().sum()\nprint(\"Missing Values:\")\nprint(missing_values)\n\n# Handle missing values\n# In this simple example, we will drop rows with missing values.\ndf = df.dropna()\n\n# Check Data Types and Convert Binary Data to Boolean\nbinary_columns = [\n  'bowel_healthy', 'bowel_injury', 'extravasation_healthy', 'extravasation_injury',\n  'kidney_healthy', 'kidney_low', 'kidney_high', 'liver_healthy', 'liver_low', 'liver_high',\n  'spleen_healthy', 'spleen_low', 'spleen_high'\n]\ndf[binary_columns] = df[binary_columns].astype(bool)\n\n# Address Data Quality Issues\n# In this simple example, we assume no data quality issues are present.\n\nprint(\"\\nPreprocessed DataFrame:\")\nprint(df)\n\nplt.figure()\ndf.plot.hist()\nplt.title('Distribution of Features')\nplt.xlabel('Feature')\nplt.ylabel('Count')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T02:06:15.579551Z","iopub.execute_input":"2024-11-15T02:06:15.579941Z","iopub.status.idle":"2024-11-15T02:06:15.968105Z","shell.execute_reply.started":"2024-11-15T02:06:15.579911Z","shell.execute_reply":"2024-11-15T02:06:15.967087Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(df.columns)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T02:06:25.304646Z","iopub.execute_input":"2024-11-15T02:06:25.305435Z","iopub.status.idle":"2024-11-15T02:06:25.310715Z","shell.execute_reply.started":"2024-11-15T02:06:25.305403Z","shell.execute_reply":"2024-11-15T02:06:25.309609Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport plotly.express as px\n\n# Giả sử df là DataFrame đã được nạp vào từ dữ liệu của bạn\n# df = pd.read_csv(\"/path/to/your/data.csv\")\n\n# Cột các cơ quan\norgan_columns = ['bowel', 'extravasation', 'kidney', 'liver', 'spleen']\n\n# Kiểm tra xem các cột \"injury\" có tồn tại trong DataFrame không\ninjury_columns = [f'{organ}_injury' for organ in organ_columns]\nmissing_columns = set(organ_columns + injury_columns) - set(df.columns)\n\nif missing_columns:\n    # Thông báo nếu có cột thiếu\n    print(f\"Warning: Columns for {', '.join(missing_columns)} are missing in the DataFrame.\")\n    for col in missing_columns:\n        df[col] = 0  # Thêm cột thiếu vào DataFrame với giá trị 0\n\n# Lọc các cột liên quan đến sức khỏe và chấn thương của các cơ quan\ncorrelation_df = df[organ_columns + injury_columns]\n\n# Tính toán ma trận tương quan giữa các cột\ncorrelation_matrix = correlation_df.corr()\n\n# Tạo heatmap để phân tích mối tương quan giữa sức khỏe và tình trạng chấn thương của các cơ quan\nfig = px.imshow(\n    correlation_matrix,\n    x=correlation_df.columns,\n    y=correlation_df.columns,\n    labels=dict(x='Organ', y='Organ', color='Correlation'),\n    title='Correlation Between Organ Health and Injury Status',\n)\n\n# Hiển thị heatmap\nfig.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T02:06:28.136652Z","iopub.execute_input":"2024-11-15T02:06:28.137545Z","iopub.status.idle":"2024-11-15T02:06:30.366983Z","shell.execute_reply.started":"2024-11-15T02:06:28.137514Z","shell.execute_reply":"2024-11-15T02:06:30.366108Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Summary statistics for relevant variables\nstyled_data = df.describe().style\\\n.background_gradient(cmap='coolwarm')\\\n.set_properties(**{'text-align':'center','border':'1px solid black'})\n\n# display styled data\ndisplay(styled_data)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T02:06:36.625793Z","iopub.execute_input":"2024-11-15T02:06:36.627064Z","iopub.status.idle":"2024-11-15T02:06:36.733371Z","shell.execute_reply.started":"2024-11-15T02:06:36.627027Z","shell.execute_reply":"2024-11-15T02:06:36.732525Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install matplotlib seaborn\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T02:06:44.035277Z","iopub.execute_input":"2024-11-15T02:06:44.035643Z","iopub.status.idle":"2024-11-15T02:06:55.591156Z","shell.execute_reply.started":"2024-11-15T02:06:44.035614Z","shell.execute_reply":"2024-11-15T02:06:55.590033Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\n\n# pass list of tick positions to the set_xticks() function. \n# pass the following list of tick positions to the set_xticks() function in the counts plot loop\n\ndef generate_counts_and_percentages(df, categorical_columns):\n  \"\"\"Counts and percentages for categorical variables in a DataFrame, and plot the counts and percentages.\n\n  Args:\n    df: DataFrame.\n    categorical_columns: column names for the categorical variables.\n\n  Returns:\n    None.\n  \"\"\"\n\n  # Handle null values.\n  df = df.dropna(subset=categorical_columns)\n\n  # counts.\n  counts = df[categorical_columns].apply(pd.Series.value_counts)\n\n  # percentages.\n  percentages = (counts / df.shape[0]) * 100\n\n  # Set color scheme.\n  colors = ['#007bff', '#ffa500']\n\n  # Plot counts.\n  fig, axes = plt.subplots(1, len(categorical_columns), figsize=(15, 7))\n  for i, column in enumerate(categorical_columns):\n    ax = axes[i]\n    ax.bar(counts.index.to_list(), counts[column].to_list(), color=colors[0])\n    ax.set_title(column, fontsize=12)\n    ax.set_xticks(range(len(counts.index)))\n    ax.tick_params(labelsize=10)\n    ax.grid(True)\n\n  # Plot percentages.\n  fig, axes = plt.subplots(1, len(categorical_columns), figsize=(15, 7))\n  for i, column in enumerate(categorical_columns):\n    ax = axes[i]\n    ax.pie(percentages[column].to_list(), labels=percentages.index.to_list(), autopct='%1.1f%%', startangle=140, colors=colors)\n    ax.set_title(column, fontsize=12)\n    ax.axis('equal')\n    ax.legend(fontsize=10)\n    ax.grid(True)\n\n  plt.suptitle('Counts and Percentages for Categorical Variables', fontsize=14)\n  plt.show()\n\ncategorical_columns = [\n  'bowel_healthy', 'bowel_injury', 'extravasation_healthy', 'extravasation_injury',\n  'kidney_healthy', 'kidney_low', 'kidney_high', 'liver_healthy', 'liver_low', 'liver_high',\n  'spleen_healthy', 'spleen_low', 'spleen_high', 'any_injury'\n]\n\ngenerate_counts_and_percentages(df, categorical_columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T02:07:04.186486Z","iopub.execute_input":"2024-11-15T02:07:04.18748Z","iopub.status.idle":"2024-11-15T02:07:07.757714Z","shell.execute_reply.started":"2024-11-15T02:07:04.18744Z","shell.execute_reply":"2024-11-15T02:07:07.756714Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport plotly.graph_objects as go\nfrom sklearn.datasets import make_classification\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestClassifier, GradientBoostingClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import SVC\nfrom sklearn.metrics import accuracy_score, confusion_matrix","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T02:07:12.496776Z","iopub.execute_input":"2024-11-15T02:07:12.497136Z","iopub.status.idle":"2024-11-15T02:07:12.504872Z","shell.execute_reply.started":"2024-11-15T02:07:12.497108Z","shell.execute_reply":"2024-11-15T02:07:12.503926Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Visualize a sample image\ndef plot_dicom_image(image_path):\n    ds = pydicom.dcmread(image_path)\n    plt.imshow(ds.pixel_array, cmap=plt.cm.bone)\n    plt.axis('off')\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T01:49:16.177012Z","iopub.status.idle":"2024-11-15T01:49:16.177355Z","shell.execute_reply.started":"2024-11-15T01:49:16.177191Z","shell.execute_reply":"2024-11-15T01:49:16.177207Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.metrics import confusion_matrix\n\n# Giả sử mỗi bộ phận có một lớp nhãn riêng biệt, ví dụ:\norgans = [\"Bowel\", \"Extravasation\", \"Liver\", \"Kidney\", \"Spleen\"]\n\n# Giả sử y_test và y_pred đã được chuẩn bị với nhiều nhãn cho từng bộ phận (mảng 2D)\n# Tạo dữ liệu giả định cho tất cả các bộ phận\n# Đây chỉ là ví dụ, bạn cần thay thế bằng dữ liệu thực tế của mình\nall_targets = [\n    [\"healthy\", \"injury\", \"healthy\", \"low\", \"healthy\"],  # Nhãn thực tế cho 1 mẫu\n    [\"injury\", \"healthy\", \"healthy\", \"high\", \"injury\"],  # Nhãn thực tế cho mẫu tiếp theo\n    # Thêm các mẫu khác ở đây...\n]\nall_preds = [\n    [\"healthy\", \"injury\", \"healthy\", \"low\", \"healthy\"],  # Dự đoán cho 1 mẫu\n    [\"injury\", \"healthy\", \"healthy\", \"high\", \"healthy\"],  # Dự đoán cho mẫu tiếp theo\n    # Thêm các mẫu khác ở đây...\n]\n\n# Chuyển đổi thành mảng numpy (mảng 2D) cho tất cả các bộ phận\nall_targets_np = np.array(all_targets)  # Nhãn thực tế (2D: samples x organs)\nall_preds_np = np.array(all_preds)      # Dự đoán (2D: samples x organs)\n\n# Lặp qua từng bộ phận và tính toán ma trận nhầm lẫn\nfor i, organ in enumerate(organs):\n    # Tạo ma trận nhầm lẫn cho từng bộ phận (sử dụng lớp tương ứng)\n    conf_matrix = confusion_matrix(all_targets_np[:, i], all_preds_np[:, i], labels=[\"healthy\", \"injury\"])\n    \n    # Hiển thị ma trận nhầm lẫn dưới dạng heatmap\n    plt.figure(figsize=(8, 6))\n    sns.heatmap(conf_matrix, annot=True, fmt=\"d\", cmap=\"Blues\", xticklabels=[\"Healthy\", \"Injury\"], yticklabels=[\"Healthy\", \"Injury\"])\n    plt.title(f'{organ} Confusion Matrix', fontsize=16)\n    plt.xlabel('Predicted', fontsize=12)\n    plt.ylabel('True', fontsize=12)\n    plt.show()\n\n    # In ra ma trận nhầm lẫn cho từng bộ phận\n    print(f\"{organ} Confusion Matrix:\")\n    print(conf_matrix)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T02:09:38.549576Z","iopub.execute_input":"2024-11-15T02:09:38.550075Z","iopub.status.idle":"2024-11-15T02:09:39.584069Z","shell.execute_reply.started":"2024-11-15T02:09:38.550032Z","shell.execute_reply":"2024-11-15T02:09:39.582629Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.metrics import confusion_matrix\n\n# Danh sách các bộ phận cần tính ma trận nhầm lẫn\norgans = [\"Bowel\", \"Extravasation\", \"Liver\", \"Kidney\", \"Spleen\"]\n\n# Giả sử y_test và y_pred đã được chuẩn bị với nhiều nhãn cho từng bộ phận (mảng 2D)\n# Dữ liệu mẫu (cần thay thế bằng dữ liệu thực tế của bạn)\nall_targets = [\n    [\"healthy\", \"injury\", \"healthy\", \"low\", \"healthy\"],  # Nhãn thực tế cho 1 mẫu\n    [\"injury\", \"healthy\", \"healthy\", \"high\", \"injury\"],  # Nhãn thực tế cho mẫu tiếp theo\n    # Thêm các mẫu khác ở đây...\n]\nall_preds = [\n    [\"healthy\", \"injury\", \"healthy\", \"low\", \"healthy\"],  # Dự đoán cho 1 mẫu\n    [\"injury\", \"healthy\", \"healthy\", \"high\", \"healthy\"],  # Dự đoán cho mẫu tiếp theo\n    # Thêm các mẫu khác ở đây...\n]\n\n# Chuyển đổi thành mảng numpy (mảng 2D) cho tất cả các bộ phận\nall_targets_np = np.array(all_targets)  # Nhãn thực tế (2D: samples x organs)\nall_preds_np = np.array(all_preds)      # Dự đoán (2D: samples x organs)\n\n# Danh sách tất cả các nhãn có thể có cho mỗi bộ phận\nclass_labels = {\n    \"Bowel\": [\"healthy\", \"injury\"],\n    \"Extravasation\": [\"healthy\", \"injury\"],\n    \"Liver\": [\"healthy\", \"low\", \"high\"],\n    \"Kidney\": [\"healthy\", \"low\", \"high\"],\n    \"Spleen\": [\"healthy\", \"low\", \"high\"],\n}\n\n# Lặp qua từng bộ phận và tính toán ma trận nhầm lẫn\nfor i, organ in enumerate(organs):\n    # Lấy các nhãn thực tế và dự đoán cho bộ phận hiện tại\n    true_labels = all_targets_np[:, i]\n    pred_labels = all_preds_np[:, i]\n    \n    # Lấy danh sách nhãn cho bộ phận hiện tại từ class_labels\n    labels = class_labels[organ]\n    \n    # Tính toán ma trận nhầm lẫn cho từng bộ phận\n    conf_matrix = confusion_matrix(true_labels, pred_labels, labels=labels)\n    \n    # Hiển thị ma trận nhầm lẫn dưới dạng heatmap\n    plt.figure(figsize=(8, 6))\n    sns.heatmap(conf_matrix, annot=True, fmt=\"d\", cmap=\"Blues\", xticklabels=labels, yticklabels=labels)\n    plt.title(f'{organ} Confusion Matrix', fontsize=16)\n    plt.xlabel('Predicted', fontsize=12)\n    plt.ylabel('True', fontsize=12)\n    plt.show()\n\n    # In ra ma trận nhầm lẫn cho từng bộ phận\n    print(f\"{organ} Confusion Matrix:\")\n    print(conf_matrix)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T02:09:42.916759Z","iopub.execute_input":"2024-11-15T02:09:42.917754Z","iopub.status.idle":"2024-11-15T02:09:44.29137Z","shell.execute_reply.started":"2024-11-15T02:09:42.91771Z","shell.execute_reply":"2024-11-15T02:09:44.290001Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.metrics import confusion_matrix\n\n# Danh sách các bộ phận cần tính ma trận nhầm lẫn\norgans = [\"Bowel\", \"Extravasation\", \"Liver\", \"Kidney\", \"Spleen\"]\n\n# Giả sử y_test và y_pred đã được chuẩn bị với nhiều nhãn cho từng bộ phận (mảng 2D)\n# Dữ liệu mẫu (cần thay thế bằng dữ liệu thực tế của bạn)\nall_targets = [\n    [\"healthy\", \"injury\", \"healthy\", \"low\", \"healthy\"],  # Nhãn thực tế cho 1 mẫu\n    [\"injury\", \"healthy\", \"healthy\", \"high\", \"injury\"],  # Nhãn thực tế cho mẫu tiếp theo\n    # Thêm các mẫu khác ở đây...\n]\nall_preds = [\n    [\"healthy\", \"injury\", \"healthy\", \"low\", \"healthy\"],  # Dự đoán cho 1 mẫu\n    [\"injury\", \"healthy\", \"healthy\", \"high\", \"healthy\"],  # Dự đoán cho mẫu tiếp theo\n    # Thêm các mẫu khác ở đây...\n]\n\n# Chuyển đổi thành mảng numpy (mảng 2D) cho tất cả các bộ phận\nall_targets_np = np.array(all_targets)  # Nhãn thực tế (2D: samples x organs)\nall_preds_np = np.array(all_preds)      # Dự đoán (2D: samples x organs)\n\n# Danh sách tất cả các nhãn có thể có cho mỗi bộ phận\nclass_labels = [\"healthy\", \"injury\", \"low\", \"high\", \"medium\"]  # Giả sử 5 nhãn cho mỗi bộ phận\n\n# Lặp qua từng bộ phận và tính toán ma trận nhầm lẫn 5x5\nfor i, organ in enumerate(organs):\n    # Lấy các nhãn thực tế và dự đoán cho bộ phận hiện tại\n    true_labels = all_targets_np[:, i]\n    pred_labels = all_preds_np[:, i]\n    \n    # Tính toán ma trận nhầm lẫn 5x5 cho từng bộ phận\n    conf_matrix = confusion_matrix(true_labels, pred_labels, labels=class_labels)\n    \n    # Hiển thị ma trận nhầm lẫn dưới dạng heatmap\n    plt.figure(figsize=(8, 6))\n    sns.heatmap(conf_matrix, annot=True, fmt=\"d\", cmap=\"Blues\", xticklabels=class_labels, yticklabels=class_labels)\n    plt.title(f'{organ} Confusion Matrix (5x5)', fontsize=16)\n    plt.xlabel('Predicted', fontsize=12)\n    plt.ylabel('True', fontsize=12)\n    plt.show()\n\n    # In ra ma trận nhầm lẫn cho từng bộ phận\n    print(f\"{organ} Confusion Matrix (5x5):\")\n    print(conf_matrix)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T02:09:47.836295Z","iopub.execute_input":"2024-11-15T02:09:47.83666Z","iopub.status.idle":"2024-11-15T02:09:49.708322Z","shell.execute_reply.started":"2024-11-15T02:09:47.83663Z","shell.execute_reply":"2024-11-15T02:09:49.707321Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.metrics import confusion_matrix\n\n# Giả sử bạn có 10 nhãn để phân loại\nclass_labels = [\"class_1\", \"class_2\", \"class_3\", \"class_4\", \"class_5\", \n                \"class_6\", \"class_7\", \"class_8\", \"class_9\", \"class_10\"]\n\n# Giả sử bạn có y_test và y_pred đã được chuẩn bị (2D array)\n# Mỗi hàng là một mẫu và mỗi cột là một bộ phận, với 10 lớp nhãn\n\n# Dữ liệu mẫu (cần thay thế bằng dữ liệu thực tế của bạn)\nall_targets = [\n    [\"class_1\", \"class_2\", \"class_3\", \"class_4\", \"class_5\"],  # Nhãn thực tế cho 1 mẫu\n    [\"class_2\", \"class_1\", \"class_3\", \"class_6\", \"class_7\"],  # Nhãn thực tế cho mẫu tiếp theo\n    # Thêm các mẫu khác ở đây...\n]\n\nall_preds = [\n    [\"class_1\", \"class_2\", \"class_3\", \"class_4\", \"class_5\"],  # Dự đoán cho 1 mẫu\n    [\"class_2\", \"class_1\", \"class_3\", \"class_6\", \"class_8\"],  # Dự đoán cho mẫu tiếp theo\n    # Thêm các mẫu khác ở đây...\n]\n\n# Chuyển đổi thành mảng numpy (mảng 2D) cho tất cả các bộ phận\nall_targets_np = np.array(all_targets)  # Nhãn thực tế (2D: samples x organs)\nall_preds_np = np.array(all_preds)      # Dự đoán (2D: samples x organs)\n\n# Lặp qua từng bộ phận và tính toán ma trận nhầm lẫn 10x10\nfor i in range(all_targets_np.shape[1]):  # Duyệt qua từng bộ phận (mỗi cột trong dữ liệu)\n    # Lấy các nhãn thực tế và dự đoán cho bộ phận hiện tại\n    true_labels = all_targets_np[:, i]\n    pred_labels = all_preds_np[:, i]\n    \n    # Tính toán ma trận nhầm lẫn 10x10 cho từng bộ phận\n    conf_matrix = confusion_matrix(true_labels, pred_labels, labels=class_labels)\n    \n    # Hiển thị ma trận nhầm lẫn dưới dạng heatmap\n    plt.figure(figsize=(10, 8))  # Chỉnh kích thước biểu đồ cho dễ nhìn\n    sns.heatmap(conf_matrix, annot=True, fmt=\"d\", cmap=\"Blues\", xticklabels=class_labels, yticklabels=class_labels)\n    plt.title(f'Confusion Matrix for Organ {i+1} (10x10)', fontsize=16)\n    plt.xlabel('Predicted', fontsize=12)\n    plt.ylabel('True', fontsize=12)\n    plt.xticks(rotation=45, ha='right')  # Xoay nhãn x-axis cho dễ đọc\n    plt.yticks(rotation=45, va='top')    # Xoay nhãn y-axis cho dễ đọc\n    plt.show()\n\n    # In ra ma trận nhầm lẫn cho từng bộ phận\n    print(f\"Confusion Matrix for Organ {i+1} (10x10):\")\n    print(conf_matrix)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T02:09:55.036901Z","iopub.execute_input":"2024-11-15T02:09:55.037789Z","iopub.status.idle":"2024-11-15T02:09:57.922622Z","shell.execute_reply.started":"2024-11-15T02:09:55.037755Z","shell.execute_reply":"2024-11-15T02:09:57.921717Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Đọc dữ liệu từ file train.csv\ndf = pd.read_csv('/kaggle/input/rsna-atd-512x512-png-v2-dataset/train.csv')\n\n# Hiển thị các cột trong DataFrame\nprint(df.columns)\n\n# Giả sử nhãn thật là các cột từ \"bowel_healthy\" đến \"spleen_high\"\nbinary_columns = [\n    'bowel_healthy', 'bowel_injury', 'extravasation_healthy', 'extravasation_injury',\n    'kidney_healthy', 'kidney_low', 'kidney_high', 'liver_healthy', 'liver_low', 'liver_high',\n    'spleen_healthy', 'spleen_low', 'spleen_high'\n]\n\n# Lấy các nhãn thật từ DataFrame (các cột nhãn)\nall_targets_np = df[binary_columns].values\n\n# In tất cả các nhãn thật\nfor i, labels in enumerate(all_targets_np):\n    print(f\"Sample {i + 1}: {labels}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T02:12:09.101394Z","iopub.execute_input":"2024-11-15T02:12:09.101827Z","iopub.status.idle":"2024-11-15T02:12:10.2477Z","shell.execute_reply.started":"2024-11-15T02:12:09.101794Z","shell.execute_reply":"2024-11-15T02:12:10.24672Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.metrics import confusion_matrix\n\n# Đọc dữ liệu từ file train.csv\ndf = pd.read_csv('/kaggle/input/rsna-atd-512x512-png-v2-dataset/train.csv')\n\n# Danh sách các cột nhãn nhị phân (10 nhãn)\nbinary_columns = [\n    'bowel_healthy', 'bowel_injury', 'extravasation_healthy', 'extravasation_injury',\n    'kidney_healthy', 'kidney_low', 'kidney_high', 'liver_healthy', 'liver_low', 'liver_high',\n    'spleen_healthy', 'spleen_low', 'spleen_high'\n]\n\n# Lấy nhãn thật từ DataFrame (các cột nhãn nhị phân)\ny_true = df[binary_columns].values\n\n# Giả sử bạn có nhãn dự đoán (y_pred). Nếu không có dữ liệu dự đoán, bạn có thể giả sử y_pred giống y_true để tạo ví dụ.\n# Đây chỉ là một giả định (hãy thay thế bằng dữ liệu thực tế của bạn)\n# Ví dụ: nếu bạn có các nhãn dự đoán từ mô hình, hãy sử dụng chúng thay vì y_true.\ny_pred = y_true  # Giả sử nhãn dự đoán giống nhãn thực tế trong ví dụ này\n\n# Tính toán ma trận nhầm lẫn (y_true vs y_pred)\nconf_matrix = confusion_matrix(y_true.flatten(), y_pred.flatten(), labels=[0, 1])\n\n# Vẽ ma trận nhầm lẫn dưới dạng heatmap với tên nhãn thay vì 0 và 1\nplt.figure(figsize=(12, 8))\nsns.heatmap(conf_matrix, annot=True, fmt=\"d\", cmap=\"Blues\", xticklabels=binary_columns, yticklabels=binary_columns)\nplt.title(\"Confusion Matrix\", fontsize=16)\nplt.xlabel('Predicted Labels', fontsize=12)\nplt.ylabel('True Labels', fontsize=12)\nplt.xticks(rotation=90)  # Xoay nhãn cột nếu cần\nplt.yticks(rotation=0)   # Giữ nhãn hàng đúng hướng\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T02:18:08.298797Z","iopub.execute_input":"2024-11-15T02:18:08.29972Z","iopub.status.idle":"2024-11-15T02:18:08.891271Z","shell.execute_reply.started":"2024-11-15T02:18:08.299665Z","shell.execute_reply":"2024-11-15T02:18:08.890332Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.metrics import confusion_matrix\n\n# Đọc dữ liệu từ file train.csv\ndf = pd.read_csv('/kaggle/input/rsna-2023-abdominal-trauma-detection/train_2024.csv')\n\n# Danh sách các cột nhãn nhị phân (10 nhãn)\nbinary_columns = [\n    'bowel_healthy', 'bowel_injury', 'extravasation_healthy', 'extravasation_injury',\n    'kidney_healthy', 'kidney_low', 'kidney_high', 'liver_healthy', 'liver_low', 'liver_high',\n    'spleen_healthy', 'spleen_low', 'spleen_high'\n]\n\n# Lấy nhãn thật từ DataFrame (các cột nhãn nhị phân)\ny_true = df[binary_columns].values\n\n# Giả sử bạn có nhãn dự đoán (y_pred). Nếu không có dữ liệu dự đoán, bạn có thể giả sử y_pred giống y_true để tạo ví dụ.\n# Đây chỉ là một giả định (hãy thay thế bằng dữ liệu thực tế của bạn)\n# Ví dụ: nếu bạn có các nhãn dự đoán từ mô hình, hãy sử dụng chúng thay vì y_true.\ny_pred = y_true  # Giả sử nhãn dự đoán giống nhãn thực tế trong ví dụ này\n\n# Tính toán ma trận nhầm lẫn (y_true vs y_pred)\nconf_matrix = confusion_matrix(y_true.flatten(), y_pred.flatten(), labels=[0, 1])\n\n# Vẽ ma trận nhầm lẫn dưới dạng heatmap với tên nhãn thay vì 0 và 1\nplt.figure(figsize=(12, 8))\nsns.heatmap(conf_matrix, annot=True, fmt=\"d\", cmap=\"Blues\", xticklabels=binary_columns, yticklabels=binary_columns)\nplt.title(\"Confusion Matrix\", fontsize=16)\nplt.xlabel('Predicted Labels', fontsize=12)\nplt.ylabel('True Labels', fontsize=12)\nplt.xticks(rotation=90)  # Xoay nhãn cột nếu cần\nplt.yticks(rotation=0)   # Giữ nhãn hàng đúng hướng\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T02:18:34.298776Z","iopub.execute_input":"2024-11-15T02:18:34.299289Z","iopub.status.idle":"2024-11-15T02:18:34.845587Z","shell.execute_reply.started":"2024-11-15T02:18:34.299246Z","shell.execute_reply":"2024-11-15T02:18:34.84472Z"}},"outputs":[],"execution_count":null}]}