{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":52254,"databundleVersionId":9674523,"sourceType":"competition"},{"sourceId":6211844,"sourceType":"datasetVersion","datasetId":3567114}],"dockerImageVersionId":30554,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# <center>Multi-label classification of abdominal trauma from CT images</center>\n\n","metadata":{"papermill":{"duration":0.025483,"end_time":"2022-02-01T10:21:43.84374","exception":false,"start_time":"2022-02-01T10:21:43.818257","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"## To commence this project, the neccesary libraries need to be installed.\n## We would be using the **TensorFlow** framework and the pretrained **EfficientNetB1**\n","metadata":{}},{"cell_type":"code","source":"# Imporitng libraries\nimport pandas as pd\nimport os\nimport random\nimport numpy as np\nimport tensorflow as tf\nimport cv2\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom IPython.display import clear_output\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.model_selection import StratifiedKFold\n\n\nrandom.seed(42)","metadata":{"id":"fUPKjZaaJHkM","papermill":{"duration":5.021373,"end_time":"2022-02-01T10:21:48.88737","exception":false,"start_time":"2022-02-01T10:21:43.865997","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T12:21:19.882254Z","iopub.execute_input":"2024-11-12T12:21:19.88257Z","iopub.status.idle":"2024-11-12T12:21:37.473755Z","shell.execute_reply.started":"2024-11-12T12:21:19.882542Z","shell.execute_reply":"2024-11-12T12:21:37.472886Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Loading the training dataset\ntrain_img = \"/kaggle/input/rsna-atd-512x512-png-v2-dataset/train_images\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T12:21:37.475427Z","iopub.execute_input":"2024-11-12T12:21:37.475981Z","iopub.status.idle":"2024-11-12T12:21:37.480249Z","shell.execute_reply.started":"2024-11-12T12:21:37.475949Z","shell.execute_reply":"2024-11-12T12:21:37.479269Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i in os.listdir(train_img):\n    for j in os.listdir(os.path.join(train_img,i)):\n        print(len(os.listdir(os.path.join(train_img,i,j))))","metadata":{"_kg_hide-output":true,"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T12:21:37.481764Z","iopub.execute_input":"2024-11-12T12:21:37.482157Z","iopub.status.idle":"2024-11-12T12:21:45.27155Z","shell.execute_reply.started":"2024-11-12T12:21:37.48211Z","shell.execute_reply":"2024-11-12T12:21:45.270629Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install imbalanced-learn","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T12:21:45.273596Z","iopub.execute_input":"2024-11-12T12:21:45.273898Z","iopub.status.idle":"2024-11-12T12:21:58.989405Z","shell.execute_reply.started":"2024-11-12T12:21:45.273872Z","shell.execute_reply":"2024-11-12T12:21:58.988162Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Part A\n## Import and explore the data","metadata":{"papermill":{"duration":0.02193,"end_time":"2022-02-01T10:21:48.93356","exception":false,"start_time":"2022-02-01T10:21:48.91163","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Making a list containing all unique classes in the training set\n\n# Reading the training labels\ntraining_labels = pd.read_csv(\"/kaggle/input/rsna-2023-abdominal-trauma-detection/train_2024.csv\")","metadata":{"id":"PBlH8ae3LbG0","papermill":{"duration":0.069238,"end_time":"2022-02-01T10:21:49.024476","exception":false,"start_time":"2022-02-01T10:21:48.955238","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T12:21:58.990722Z","iopub.execute_input":"2024-11-12T12:21:58.991028Z","iopub.status.idle":"2024-11-12T12:21:59.019115Z","shell.execute_reply.started":"2024-11-12T12:21:58.990999Z","shell.execute_reply":"2024-11-12T12:21:59.018405Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Viewing the dataset in a structured format\ntraining_labels","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T12:21:59.020051Z","iopub.execute_input":"2024-11-12T12:21:59.020307Z","iopub.status.idle":"2024-11-12T12:21:59.05213Z","shell.execute_reply.started":"2024-11-12T12:21:59.020284Z","shell.execute_reply":"2024-11-12T12:21:59.051285Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Part B\n## In the next few cells, we chose to explore the dataframe to check for missing values","metadata":{}},{"cell_type":"code","source":"training_labels.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T12:21:59.053199Z","iopub.execute_input":"2024-11-12T12:21:59.053474Z","iopub.status.idle":"2024-11-12T12:21:59.059177Z","shell.execute_reply.started":"2024-11-12T12:21:59.053449Z","shell.execute_reply":"2024-11-12T12:21:59.058454Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['patient_id'].isna().value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T12:21:59.060579Z","iopub.execute_input":"2024-11-12T12:21:59.060856Z","iopub.status.idle":"2024-11-12T12:21:59.08053Z","shell.execute_reply.started":"2024-11-12T12:21:59.060833Z","shell.execute_reply":"2024-11-12T12:21:59.079734Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['bowel_healthy'].isna().value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T12:21:59.081344Z","iopub.execute_input":"2024-11-12T12:21:59.081594Z","iopub.status.idle":"2024-11-12T12:21:59.08852Z","shell.execute_reply.started":"2024-11-12T12:21:59.081573Z","shell.execute_reply":"2024-11-12T12:21:59.087669Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['bowel_injury'].isna().value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T12:21:59.092968Z","iopub.execute_input":"2024-11-12T12:21:59.093294Z","iopub.status.idle":"2024-11-12T12:21:59.103371Z","shell.execute_reply.started":"2024-11-12T12:21:59.093271Z","shell.execute_reply":"2024-11-12T12:21:59.102472Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['extravasation_healthy'].isna().value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T12:21:59.104376Z","iopub.execute_input":"2024-11-12T12:21:59.104712Z","iopub.status.idle":"2024-11-12T12:21:59.118865Z","shell.execute_reply.started":"2024-11-12T12:21:59.104682Z","shell.execute_reply":"2024-11-12T12:21:59.118112Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['extravasation_injury'].isna().value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T12:21:59.119811Z","iopub.execute_input":"2024-11-12T12:21:59.120071Z","iopub.status.idle":"2024-11-12T12:21:59.131156Z","shell.execute_reply.started":"2024-11-12T12:21:59.12005Z","shell.execute_reply":"2024-11-12T12:21:59.130339Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['kidney_healthy'].isna().value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T12:21:59.132291Z","iopub.execute_input":"2024-11-12T12:21:59.132566Z","iopub.status.idle":"2024-11-12T12:21:59.143376Z","shell.execute_reply.started":"2024-11-12T12:21:59.132542Z","shell.execute_reply":"2024-11-12T12:21:59.142468Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['kidney_low'].isna().value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T12:21:59.144443Z","iopub.execute_input":"2024-11-12T12:21:59.14472Z","iopub.status.idle":"2024-11-12T12:21:59.157899Z","shell.execute_reply.started":"2024-11-12T12:21:59.144695Z","shell.execute_reply":"2024-11-12T12:21:59.156984Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['kidney_high'].isna().value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T12:21:59.158893Z","iopub.execute_input":"2024-11-12T12:21:59.159146Z","iopub.status.idle":"2024-11-12T12:21:59.171566Z","shell.execute_reply.started":"2024-11-12T12:21:59.159123Z","shell.execute_reply":"2024-11-12T12:21:59.170724Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['liver_healthy'].isna().value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T12:21:59.172809Z","iopub.execute_input":"2024-11-12T12:21:59.173186Z","iopub.status.idle":"2024-11-12T12:21:59.185635Z","shell.execute_reply.started":"2024-11-12T12:21:59.173158Z","shell.execute_reply":"2024-11-12T12:21:59.184734Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['liver_low'].isna().value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T12:21:59.186983Z","iopub.execute_input":"2024-11-12T12:21:59.187576Z","iopub.status.idle":"2024-11-12T12:21:59.199983Z","shell.execute_reply.started":"2024-11-12T12:21:59.187542Z","shell.execute_reply":"2024-11-12T12:21:59.199239Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['liver_high'].isna().value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T12:21:59.200848Z","iopub.execute_input":"2024-11-12T12:21:59.201137Z","iopub.status.idle":"2024-11-12T12:21:59.214599Z","shell.execute_reply.started":"2024-11-12T12:21:59.201113Z","shell.execute_reply":"2024-11-12T12:21:59.213733Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['spleen_healthy'].isna().value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T12:21:59.215615Z","iopub.execute_input":"2024-11-12T12:21:59.215869Z","iopub.status.idle":"2024-11-12T12:21:59.227028Z","shell.execute_reply.started":"2024-11-12T12:21:59.215847Z","shell.execute_reply":"2024-11-12T12:21:59.226328Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['spleen_low'].isna().value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T12:21:59.22803Z","iopub.execute_input":"2024-11-12T12:21:59.228278Z","iopub.status.idle":"2024-11-12T12:21:59.240325Z","shell.execute_reply.started":"2024-11-12T12:21:59.228256Z","shell.execute_reply":"2024-11-12T12:21:59.239545Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['spleen_high'].isna().value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T12:21:59.24136Z","iopub.execute_input":"2024-11-12T12:21:59.241705Z","iopub.status.idle":"2024-11-12T12:21:59.253716Z","shell.execute_reply.started":"2024-11-12T12:21:59.241673Z","shell.execute_reply":"2024-11-12T12:21:59.252871Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['any_injury'].isna().value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T12:21:59.254691Z","iopub.execute_input":"2024-11-12T12:21:59.255012Z","iopub.status.idle":"2024-11-12T12:21:59.266092Z","shell.execute_reply.started":"2024-11-12T12:21:59.254987Z","shell.execute_reply":"2024-11-12T12:21:59.26525Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Using the EfficientNetB1 model and tweaking the last two layers to suit our work\n#def create_model(decay_steps=10,warmup_steps=10):\n#    base_model = tf.keras.applications.EfficientNetB1(\n#    weights= \"imagenet\", include_top=False, input_shape= (512,512,3)\n#    )\n#    num_classes=61\n\n#    x = base_model.output\n#    x = tf.keras.layers.GlobalAveragePooling2D()(x)\n#    x = tf.keras.layers.Dropout(0.2)(x)\n#    x_bowel = tf.keras.layers.Dense(32, activation='silu')(x)\n#    x_extra = tf.keras.layers.Dense(32, activation='silu')(x)\n#    x_liver = tf.keras.layers.Dense(32, activation='silu')(x)\n #   x_kidney = tf.keras.layers.Dense(32, activation='silu')(x)\n#    x_spleen = tf.keras.layers.Dense(32, activation='silu')(x)\n\n    # Define heads\n#    out_bowel = tf.keras.layers.Dense(1, name='bowel', activation='sigmoid')(x_bowel) # use sigmoid to convert predictions to [0-1]\n#    out_extra = tf.keras.layers.Dense(1, name='extra', activation='sigmoid')(x_extra) # use sigmoid to convert predictions to [0-1]\n#    out_liver = tf.keras.layers.Dense(3, name='liver', activation='softmax')(x_liver) # use softmax for the liver head\n#    out_kidney = tf.keras.layers.Dense(3, name='kidney', activation='softmax')(x_kidney) # use softmax for the kidney head\n#    out_spleen = tf.keras.layers.Dense(3, name='spleen', activation='softmax')(x_spleen) # use softmax for the spleen head\n\n#    model = tf.keras.Model(inputs = base_model.input, outputs = [out_bowel,out_extra,out_liver,out_kidney,out_spleen])\n        # Cosine Decay\n#    cosine_decay = tf.keras.optimizers.schedules.CosineDecay(\n#        initial_learning_rate=1e-4,\n#        decay_steps=decay_steps,\n#        alpha=0.0,\n        #warmup_target=1e-3,\n        #warmup_steps=warmup_steps,\n #   )\n\n    # Compile the model\n #   optimizer = tf.keras.optimizers.Adam(learning_rate=cosine_decay)\n #   loss = [\n #       tf.keras.losses.BinaryCrossentropy(),\n #       tf.keras.losses.BinaryCrossentropy(),\n #       tf.keras.losses.CategoricalCrossentropy(),\n #       tf.keras.losses.CategoricalCrossentropy(),\n #       tf.keras.losses.CategoricalCrossentropy()]\n    \n #   metrics = [\n #       [tf.keras.metrics.BinaryAccuracy(name=\"bowel_binary_accuracy\")],\n #       [tf.keras.metrics.BinaryAccuracy(name=\"extra_binary_accuracy\")],\n  #      [tf.keras.metrics.CategoricalAccuracy(name=\"liver_cat_accuracy\")],\n #       [tf.keras.metrics.CategoricalAccuracy(name=\"kidney_cat_accuracy\")],\n #       [tf.keras.metrics.CategoricalAccuracy(name=\"spleen_cat_accuracy\")]]\n    #    \"bowel\":[\"accuracy\"],\n    #    \"extra\":[\"accuracy\"],\n    #    \"liver\":[\"accuracy\"],\n    #    \"kidney\":[\"accuracy\"],\n    #    \"spleen\":[\"accuracy\"],\n    #}\n  #  print(\"[INFO] Compiling the model...\")\n #   model.compile(\n #       optimizer=optimizer,\n #     loss=loss,\n #     metrics=metrics\n #   )\n #   return model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T12:21:59.267545Z","iopub.execute_input":"2024-11-12T12:21:59.267896Z","iopub.status.idle":"2024-11-12T12:21:59.278751Z","shell.execute_reply.started":"2024-11-12T12:21:59.267866Z","shell.execute_reply":"2024-11-12T12:21:59.277986Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"+ Sử dụng Gradient Clipping để tránh gradient quá lớn:","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\n\n\n\ndef create_model(decay_steps=900, warmup_steps=100, fine_tune_from=100):\n    # Tải EfficientNetB1 làm base model\n    base_model = tf.keras.applications.EfficientNetB1(\n        weights=\"imagenet\", include_top=False, input_shape=(512, 512, 3)\n    )\n\n    # Fine-tune một phần của base model\n    for layer in base_model.layers[:fine_tune_from]:\n        layer.trainable = False\n\n    x = base_model.output\n\n    # Lớp Convolution với kernel 3x3 và tăng số lượng filters\n    x = tf.keras.layers.Conv2D(128, (5, 5), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.MaxPooling2D((2, 2), padding='same')(x)\n    \n    x = tf.keras.layers.Conv2D(256, (5, 5), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.MaxPooling2D((2, 2), padding='same')(x)\n\n    x = tf.keras.layers.Conv2D(512, (5, 5), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.MaxPooling2D((2, 2), padding='same')(x)\n\n    x = tf.keras.layers.Conv2D(1024, (5, 5), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.MaxPooling2D((2, 2), padding='same')(x)\n\n    # Residual Block\n    residual = x\n    x = tf.keras.layers.Conv2D(1024, (2, 2), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.Add()([x, residual])  # Skip connection\n\n    # Dense Block với dropout cao hơn và số lượng units lớn hơn\n    x = tf.keras.layers.GlobalAveragePooling2D()(x)\n    x = tf.keras.layers.Dropout(0.6)(x)  # Tăng dropout\n\n    # Dense layers cho các đầu ra\n    x_bowel = tf.keras.layers.Dense(512, activation='relu')(x)\n    x_extra = tf.keras.layers.Dense(512, activation='relu')(x)\n    x_liver = tf.keras.layers.Dense(512, activation='relu')(x)\n    x_kidney = tf.keras.layers.Dense(512, activation='relu')(x)\n    x_spleen = tf.keras.layers.Dense(512, activation='relu')(x)\n\n    # Output Layers\n    out_bowel = tf.keras.layers.Dense(1, name='bowel', activation='sigmoid')(x_bowel)  # Bowel (binary)\n    out_extra = tf.keras.layers.Dense(1, name='extra', activation='sigmoid')(x_extra)  # Extra (binary)\n    out_liver = tf.keras.layers.Dense(3, name='liver', activation='softmax')(x_liver)  # Liver (multiclass)\n    out_kidney = tf.keras.layers.Dense(3, name='kidney', activation='softmax')(x_kidney)  # Kidney (multiclass)\n    out_spleen = tf.keras.layers.Dense(3, name='spleen', activation='softmax')(x_spleen)  # Spleen (multiclass)\n\n    # Create model\n    model = tf.keras.Model(inputs=base_model.input, outputs=[out_bowel, out_extra, out_liver, out_kidney, out_spleen])\n    \n    # Cosine Decay with a warmup phase\n    cosine_decay = tf.keras.optimizers.schedules.CosineDecayRestarts(\n        initial_learning_rate=3e-4,  # Lower initial learning rate\n        first_decay_steps=warmup_steps,\n        t_mul=1.8,\n        m_mul=0.8,\n        alpha=0.2\n    )\n\n    # Compile the model\n    optimizer = tf.keras.optimizers.Adam(learning_rate=cosine_decay)\n    loss = [\n        tf.keras.losses.BinaryCrossentropy(),\n        tf.keras.losses.BinaryCrossentropy(),\n        tf.keras.losses.CategoricalCrossentropy(),\n        tf.keras.losses.CategoricalCrossentropy(),\n        tf.keras.losses.CategoricalCrossentropy()\n    ]\n    \n    metrics = [\n        [tf.keras.metrics.BinaryAccuracy(name=\"bowel_binary_accuracy\")],\n        [tf.keras.metrics.BinaryAccuracy(name=\"extra_binary_accuracy\")],\n        [tf.keras.metrics.CategoricalAccuracy(name=\"liver_cat_accuracy\")],\n        [tf.keras.metrics.CategoricalAccuracy(name=\"kidney_cat_accuracy\")],\n        [tf.keras.metrics.CategoricalAccuracy(name=\"spleen_cat_accuracy\")]\n    ]\n    \n    print(\"[INFO] Compiling the model...\")\n    model.compile(\n        optimizer=optimizer,\n        loss=loss,\n        metrics=metrics\n    )\n    \n    return model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T12:21:59.280022Z","iopub.execute_input":"2024-11-12T12:21:59.280477Z","iopub.status.idle":"2024-11-12T12:21:59.301237Z","shell.execute_reply.started":"2024-11-12T12:21:59.280447Z","shell.execute_reply":"2024-11-12T12:21:59.300316Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#model = create_model()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T12:21:59.302261Z","iopub.execute_input":"2024-11-12T12:21:59.30259Z","iopub.status.idle":"2024-11-12T12:21:59.318507Z","shell.execute_reply.started":"2024-11-12T12:21:59.302566Z","shell.execute_reply":"2024-11-12T12:21:59.317794Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = create_model()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T12:21:59.319523Z","iopub.execute_input":"2024-11-12T12:21:59.319794Z","iopub.status.idle":"2024-11-12T12:22:04.967787Z","shell.execute_reply.started":"2024-11-12T12:21:59.31977Z","shell.execute_reply":"2024-11-12T12:22:04.966873Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# In ra tóm tắt mô hình\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T12:22:04.969051Z","iopub.execute_input":"2024-11-12T12:22:04.969342Z","iopub.status.idle":"2024-11-12T12:22:05.7741Z","shell.execute_reply.started":"2024-11-12T12:22:04.969316Z","shell.execute_reply":"2024-11-12T12:22:05.773353Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tf.keras.utils.plot_model(\n    model,  # Sử dụng đúng tên biến của mô hình\n    to_file='model.png'\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T12:22:05.794475Z","iopub.execute_input":"2024-11-12T12:22:05.794773Z","iopub.status.idle":"2024-11-12T12:22:08.59457Z","shell.execute_reply.started":"2024-11-12T12:22:05.794747Z","shell.execute_reply":"2024-11-12T12:22:08.592904Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels = training_labels.set_index('patient_id')\ntraining_labels.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T12:22:08.595697Z","iopub.execute_input":"2024-11-12T12:22:08.595963Z","iopub.status.idle":"2024-11-12T12:22:08.610339Z","shell.execute_reply.started":"2024-11-12T12:22:08.595941Z","shell.execute_reply":"2024-11-12T12:22:08.60948Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Part C\n## Balancing the imbalanced training dataset ","metadata":{}},{"cell_type":"code","source":"#Using histogram to get the distribution of the labels and to check if there are outliers and wrong labels\ntraining_labels.hist(figsize=(20,12),bins=2)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T12:22:08.611621Z","iopub.execute_input":"2024-11-12T12:22:08.611958Z","iopub.status.idle":"2024-11-12T12:22:11.04203Z","shell.execute_reply.started":"2024-11-12T12:22:08.61193Z","shell.execute_reply":"2024-11-12T12:22:11.041023Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#images ,labels","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T12:22:11.043529Z","iopub.execute_input":"2024-11-12T12:22:11.044129Z","iopub.status.idle":"2024-11-12T12:22:11.047824Z","shell.execute_reply.started":"2024-11-12T12:22:11.044099Z","shell.execute_reply":"2024-11-12T12:22:11.046917Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"images = []\nlabels = []\nfor i in sorted(os.listdir(train_img)):\n    folder = os.listdir(os.path.join(train_img,i))[0]\n    file = os.listdir(os.path.join(train_img,i,folder))[0]\n    images.append(cv2.imread(os.path.join(train_img,i,folder,file), cv2.IMREAD_COLOR ))\n    labels.append(np.asarray(training_labels.loc[int(i)]))\nimages = np.asarray(images)\nlabels = np.asarray(labels)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T12:22:11.049097Z","iopub.execute_input":"2024-11-12T12:22:11.049371Z","iopub.status.idle":"2024-11-12T12:22:15.249003Z","shell.execute_reply.started":"2024-11-12T12:22:11.049346Z","shell.execute_reply":"2024-11-12T12:22:15.248077Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Examples of images","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(30,12))\nfor i in range(10):\n    plt.subplot(2,5,i+1)\n    plt.imshow(images[i])\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T12:22:15.250474Z","iopub.execute_input":"2024-11-12T12:22:15.250862Z","iopub.status.idle":"2024-11-12T12:22:17.494744Z","shell.execute_reply.started":"2024-11-12T12:22:15.250824Z","shell.execute_reply":"2024-11-12T12:22:17.493781Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T12:22:17.495983Z","iopub.execute_input":"2024-11-12T12:22:17.496278Z","iopub.status.idle":"2024-11-12T12:22:17.500395Z","shell.execute_reply.started":"2024-11-12T12:22:17.496253Z","shell.execute_reply":"2024-11-12T12:22:17.499466Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train, X_val, y_train, y_val = train_test_split(images, labels, test_size=0.25)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T12:22:17.50171Z","iopub.execute_input":"2024-11-12T12:22:17.502024Z","iopub.status.idle":"2024-11-12T12:22:17.568682Z","shell.execute_reply.started":"2024-11-12T12:22:17.501996Z","shell.execute_reply":"2024-11-12T12:22:17.567705Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from imblearn.over_sampling import RandomOverSampler\n\n#oversampler = RandomOverSampler(random_state=42)\n#new_images, new_labels = oversampler.fit_resample(images, labels)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T12:22:17.5699Z","iopub.execute_input":"2024-11-12T12:22:17.570215Z","iopub.status.idle":"2024-11-12T12:22:18.342874Z","shell.execute_reply.started":"2024-11-12T12:22:17.570188Z","shell.execute_reply":"2024-11-12T12:22:18.342087Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Part D\n## Training the model with the training dataset","metadata":{}},{"cell_type":"code","source":"bowel_labels = labels[:,2]\nextravasation_labels = labels[:,4]\nkidney_labels = labels[:,4:7]\nliver_labels = labels[:,7:10]\nspleen_labels = labels[:,10:13]\nany_labels = labels[:,-1]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T12:22:18.344009Z","iopub.execute_input":"2024-11-12T12:22:18.34499Z","iopub.status.idle":"2024-11-12T12:22:18.349954Z","shell.execute_reply.started":"2024-11-12T12:22:18.344961Z","shell.execute_reply":"2024-11-12T12:22:18.349014Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"bowel_val = y_val[:,2]\nextravasation_val = y_val[:,4]\nkidney_val = y_val[:,4:7]\nliver_val = y_val[:,7:10]\nspleen_val = y_val[:,10:13]\nany_val = y_val[:,-1]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T12:22:18.351274Z","iopub.execute_input":"2024-11-12T12:22:18.351861Z","iopub.status.idle":"2024-11-12T12:22:18.364282Z","shell.execute_reply.started":"2024-11-12T12:22:18.351827Z","shell.execute_reply":"2024-11-12T12:22:18.363392Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"bowel_train = y_train[:,2]\nextravasation_train = y_train[:,4]\nkidney_train = y_train[:,4:7]\nliver_train = y_train[:,7:10]\nspleen_train = y_train[:,10:13]\nany_train = y_train[:,-1]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T12:22:18.365471Z","iopub.execute_input":"2024-11-12T12:22:18.365957Z","iopub.status.idle":"2024-11-12T12:22:18.375667Z","shell.execute_reply.started":"2024-11-12T12:22:18.365931Z","shell.execute_reply":"2024-11-12T12:22:18.374843Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#batch_size = 8\n#num_epoch = 2\n#history = model.fit(x=X_train,y=[bowel_train,extravasation_train,kidney_train,liver_train,spleen_train],batch_size=batch_size, epochs=num_epoch, verbose=1, validation_data=(X_val,[bowel_val,extravasation_val,kidney_val,liver_val,spleen_val]))\n\n\n#validation_data=(X_val,[bowel_val,extravasation_val,kidney_val,liver_val,spleen_val])","metadata":{"papermill":{"duration":1176.379334,"end_time":"2022-02-01T14:07:30.902597","exception":false,"start_time":"2022-02-01T13:47:54.523263","status":"completed"},"tags":[],"_kg_hide-output":true,"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T12:22:18.376804Z","iopub.execute_input":"2024-11-12T12:22:18.377142Z","iopub.status.idle":"2024-11-12T12:22:18.388087Z","shell.execute_reply.started":"2024-11-12T12:22:18.377109Z","shell.execute_reply":"2024-11-12T12:22:18.387202Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install keras-rectified-adam\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T12:22:18.389133Z","iopub.execute_input":"2024-11-12T12:22:18.389544Z","iopub.status.idle":"2024-11-12T12:22:32.283965Z","shell.execute_reply.started":"2024-11-12T12:22:18.389508Z","shell.execute_reply":"2024-11-12T12:22:32.282846Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.callbacks import EarlyStopping, ModelCheckpoint\nimport tensorflow as tf\n\n\n\ndef create_model(decay_steps=900, warmup_steps=60, fine_tune_from=100):\n    # Tải EfficientNetB1 làm base model\n    base_model = tf.keras.applications.EfficientNetB1(\n        weights=\"imagenet\", include_top=False, input_shape=(512, 512, 3)\n    )\n\n    # Fine-tune một phần của base model\n    for layer in base_model.layers[:fine_tune_from]:\n        layer.trainable = False\n\n    x = base_model.output\n\n    # Lớp Convolution với kernel 3x3 và tăng số lượng filters\n    x = tf.keras.layers.Conv2D(128, (5, 5), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.MaxPooling2D((2, 2), padding='same')(x)\n    \n    x = tf.keras.layers.Conv2D(256, (5, 5), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.MaxPooling2D((2, 2), padding='same')(x)\n\n    x = tf.keras.layers.Conv2D(512, (5, 5), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.MaxPooling2D((2, 2), padding='same')(x)\n\n    x = tf.keras.layers.Conv2D(1024, (5, 5), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.MaxPooling2D((2, 2), padding='same')(x)\n\n    # Residual Block\n    residual = x\n    x = tf.keras.layers.Conv2D(1024, (2, 2), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.Add()([x, residual])  # Skip connection\n\n    # Dense Block với dropout cao hơn và số lượng units lớn hơn\n    x = tf.keras.layers.GlobalAveragePooling2D()(x)\n    x = tf.keras.layers.Dropout(0.6)(x)  # Tăng dropout\n\n    # Dense layers cho các đầu ra\n    x_bowel = tf.keras.layers.Dense(512, activation='relu')(x)\n    x_extra = tf.keras.layers.Dense(512, activation='relu')(x)\n    x_liver = tf.keras.layers.Dense(512, activation='relu')(x)\n    x_kidney = tf.keras.layers.Dense(512, activation='relu')(x)\n    x_spleen = tf.keras.layers.Dense(512, activation='relu')(x)\n\n    # Output Layers\n    out_bowel = tf.keras.layers.Dense(1, name='bowel', activation='sigmoid')(x_bowel)  # Bowel (binary)\n    out_extra = tf.keras.layers.Dense(1, name='extra', activation='sigmoid')(x_extra)  # Extra (binary)\n    out_liver = tf.keras.layers.Dense(3, name='liver', activation='softmax')(x_liver)  # Liver (multiclass)\n    out_kidney = tf.keras.layers.Dense(3, name='kidney', activation='softmax')(x_kidney)  # Kidney (multiclass)\n    out_spleen = tf.keras.layers.Dense(3, name='spleen', activation='softmax')(x_spleen)  # Spleen (multiclass)\n\n    # Create model\n    model = tf.keras.Model(inputs=base_model.input, outputs=[out_bowel, out_extra, out_liver, out_kidney, out_spleen])\n    \n    # Cosine Decay with a warmup phase\n    cosine_decay = tf.keras.optimizers.schedules.CosineDecayRestarts(\n        initial_learning_rate=3e-4,  # Lower initial learning rate\n        first_decay_steps=warmup_steps,\n        t_mul=1.8,\n        m_mul=0.8,\n        alpha=0.2\n    )\n\n    # Compile the model\n    optimizer = tf.keras.optimizers.Adam(learning_rate=cosine_decay)\n    loss = [\n        tf.keras.losses.BinaryCrossentropy(),\n        tf.keras.losses.BinaryCrossentropy(),\n        tf.keras.losses.CategoricalCrossentropy(),\n        tf.keras.losses.CategoricalCrossentropy(),\n        tf.keras.losses.CategoricalCrossentropy()\n    ]\n    \n    metrics = [\n        [tf.keras.metrics.BinaryAccuracy(name=\"bowel_binary_accuracy\")],\n        [tf.keras.metrics.BinaryAccuracy(name=\"extra_binary_accuracy\")],\n        [tf.keras.metrics.CategoricalAccuracy(name=\"liver_cat_accuracy\")],\n        [tf.keras.metrics.CategoricalAccuracy(name=\"kidney_cat_accuracy\")],\n        [tf.keras.metrics.CategoricalAccuracy(name=\"spleen_cat_accuracy\")]\n    ]\n    \n    print(\"[INFO] Compiling the model...\")\n    model.compile(\n        optimizer=optimizer,\n        loss=loss,\n        metrics=metrics\n    )\n    \n    return model\n\n\n# Load images and labels\nimages = []\nlabels = []\nfor i in sorted(os.listdir(train_img)):\n    folder = os.listdir(os.path.join(train_img, i))[0]\n    file = os.listdir(os.path.join(train_img, i, folder))[0]\n    images.append(cv2.imread(os.path.join(train_img, i, folder, file), cv2.IMREAD_COLOR))\n    labels.append(np.asarray(training_labels.loc[int(i)]))\nimages = np.asarray(images)\nlabels = np.asarray(labels)\n\n# Split data\nX_train, X_val, y_train, y_val = train_test_split(images, labels, test_size=0.25)\n\n# Prepare labels for each class\nbowel_labels = labels[:, 2]\nextravasation_labels = labels[:, 4]\nkidney_labels = labels[:, 4:7]\nliver_labels = labels[:, 7:10]\nspleen_labels = labels[:, 10:13]\nany_labels = labels[:, -1]\n\nbowel_val = y_val[:, 2]\nextravasation_val = y_val[:, 4]\nkidney_val = y_val[:, 4:7]\nliver_val = y_val[:, 7:10]\nspleen_val = y_val[:, 10:13]\nany_val = y_val[:, -1]\n\n# KFold Cross Validation\nkf = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n\n# Store results for each fold\nfold_results = []\n\n# Data Augmentation\ntrain_datagen = ImageDataGenerator(\n    rotation_range=40,\n    width_shift_range=0.2,\n    height_shift_range=0.2,\n    shear_range=0.2,\n    zoom_range=0.2,\n    horizontal_flip=True,\n    fill_mode='nearest'\n)\n\n# Validation data generator (no augmentation)\nval_datagen = ImageDataGenerator()\n\nfor fold, (train_index, val_index) in enumerate(kf.split(images, any_labels)):\n    print(f\"Fold {fold}\")  # Changed to print fold number from 0 to 4\n\n    # Split data into train and validation according to fold\n    X_train_fold, X_val_fold = images[train_index], images[val_index]\n    y_train_fold, y_val_fold = labels[train_index], labels[val_index]\n\n    # Prepare labels for each class for the fold\n    bowel_train = y_train_fold[:, 2]\n    extravasation_train = y_train_fold[:, 4]\n    kidney_train = y_train_fold[:, 4:7]\n    liver_train = y_train_fold[:, 7:10]\n    spleen_train = y_train_fold[:, 10:13]\n\n    bowel_val = y_val_fold[:, 2]\n    extravasation_val = y_val_fold[:, 4]\n    kidney_val = y_val_fold[:, 4:7]\n    liver_val = y_val_fold[:, 7:10]\n    spleen_val = y_val_fold[:, 10:13]\n\n    # Create a new model for each fold\n    model = create_model()\n\n    # EarlyStopping and ModelCheckpoint\n    early_stopping = EarlyStopping(monitor='val_loss', patience=10, restore_best_weights=True)\n    checkpoint = ModelCheckpoint(f'best_model_fold_{fold}.h5', save_best_only=True, monitor='val_loss')\n\n    # Train the model\n    batch_size = 16\n    num_epoch = 200  # You can modify this value based on the training time and model convergence\n    \n    history = model.fit(\n        x=X_train_fold,\n        y=[bowel_train, extravasation_train, kidney_train, liver_train, spleen_train],\n        batch_size=batch_size,\n        epochs=num_epoch,\n        verbose=1,\n         validation_data=(X_val_fold, [bowel_val, extravasation_val, kidney_val, liver_val, spleen_val])\n    )\n\n    # Store results for this fold\n    fold_results.append(history.history)\n\n# Check results\nfor fold, result in enumerate(fold_results):\n    print(f\"Fold {fold} Results:\")  # Changed to print fold number from 0 to 4\n    for key in result.keys():\n        print(f\"{key}: {result[key][-1]}\")\n\n\n\nimport pandas as pd\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import confusion_matrix, recall_score, precision_score, f1_score, accuracy_score\nfrom sklearn.model_selection import train_test_split\n\n# Đọc dữ liệu nhãn từ file train\ntrain_labels = pd.read_csv(\"/kaggle/input/rsna-2023-abdominal-trauma-detection/train_2024.csv\")  # Đảm bảo đường dẫn đúng\n\n# Tạo các đặc trưng (features) và nhãn (labels)\nX = train_labels.drop(columns=['bowel_healthy', 'bowel_injury', 'any_injury'])  # Xóa các cột nhãn để chỉ lấy đặc trưng\ny_bowel_healthy = train_labels['bowel_healthy']\n\n# Chia dữ liệu thành tập huấn luyện và tập kiểm tra\nX_train, X_test, y_train, y_test = train_test_split(X, y_bowel_healthy, test_size=0.2, random_state=42)\n\n# Khởi tạo và huấn luyện mô hình RandomForest\nmodel = RandomForestClassifier(n_estimators=100, random_state=42)\nmodel.fit(X_train, y_train)\n\n# Dự đoán kết quả trên tập kiểm tra\ny_pred_bowel_healthy = model.predict(X_test)\n\n# Tính toán các chỉ số đánh giá\ndef calculate_metrics(y_true, y_pred, num_classes=1):\n    if num_classes == 1:\n        cm = confusion_matrix(y_true, y_pred)\n        tn, fp, fn, tp = cm.ravel()\n        recall = tp / (tp + fn)  # Độ nhạy (Recall)\n        precision = tp / (tp + fp)  # Độ chính xác (Precision)\n        f1 = 2 * (precision * recall) / (precision + recall)  # F1-score\n        accuracy = (tp + tn) / (tp + tn + fp + fn)  # Độ chính xác (Accuracy)\n        specificity = tn / (tn + fp)  # Độ nhiễu (Specificity)\n    else:\n        recall = recall_score(y_true, y_pred, average='weighted', zero_division=0)\n        precision = precision_score(y_true, y_pred, average='weighted', zero_division=0)\n        f1 = f1_score(y_true, y_pred, average='weighted', zero_division=0)\n        accuracy = accuracy_score(y_true, y_pred)\n        specificity = None\n\n    return recall, precision, f1, accuracy, specificity\n\n# Tính toán các chỉ số đánh giá\nrecall_bowel_healthy, precision_bowel_healthy, f1_bowel_healthy, accuracy_bowel_healthy, specificity_bowel_healthy = calculate_metrics(y_test, y_pred_bowel_healthy)\n\n# In kết quả\nprint(f\"Results for bowel_healthy:\")\nprint(f\"Recall: {recall_bowel_healthy:.4f}\")\nprint(f\"Precision: {precision_bowel_healthy:.4f}\")\nprint(f\"F1-score: {f1_bowel_healthy:.4f}\")\nprint(f\"Accuracy: {accuracy_bowel_healthy:.4f}\")\nprint(f\"Specificity: {specificity_bowel_healthy:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T12:22:32.285851Z","iopub.execute_input":"2024-11-12T12:22:32.286155Z","iopub.status.idle":"2024-11-12T14:24:18.565092Z","shell.execute_reply.started":"2024-11-12T12:22:32.286126Z","shell.execute_reply":"2024-11-12T14:24:18.564065Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history.history.keys()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T14:24:18.566352Z","iopub.execute_input":"2024-11-12T14:24:18.566674Z","iopub.status.idle":"2024-11-12T14:24:18.572682Z","shell.execute_reply.started":"2024-11-12T14:24:18.566647Z","shell.execute_reply":"2024-11-12T14:24:18.571813Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"val_acc = ['val_bowel_bowel_binary_accuracy', 'val_extra_extra_binary_accuracy', 'val_liver_liver_cat_accuracy', 'val_kidney_kidney_cat_accuracy', 'val_spleen_spleen_cat_accuracy']","metadata":{"_kg_hide-input":true,"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T15:01:55.157553Z","iopub.execute_input":"2024-11-12T15:01:55.158483Z","iopub.status.idle":"2024-11-12T15:01:55.162997Z","shell.execute_reply.started":"2024-11-12T15:01:55.158441Z","shell.execute_reply":"2024-11-12T15:01:55.161982Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import confusion_matrix, recall_score, precision_score, f1_score, accuracy_score\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\n\n# Đọc dữ liệu nhãn từ file train\ntrain_labels = pd.read_csv(\"/kaggle/input/rsna-atd-512x512-png-v2-dataset/train.csv\")\n\n# Kiểm tra tên cột trong DataFrame để xác định các cột không phải số\nprint(train_labels.columns)\n\n# Tạo các đặc trưng (features) và nhãn (labels)\nX = train_labels.drop(columns=['bowel_healthy', 'bowel_injury'])  # Xóa các cột nhãn để chỉ lấy đặc trưng\ny_bowel_healthy = train_labels['bowel_healthy']\n\n# Chuyển đổi các cột không phải số (nếu có) thành số\n# Xác định các cột kiểu chuỗi và chuyển chúng thành giá trị số\ncategorical_columns = X.select_dtypes(include=['object']).columns\n\n# Áp dụng LabelEncoder cho tất cả các cột không phải số\nencoder = LabelEncoder()\nfor col in categorical_columns:\n    X[col] = encoder.fit_transform(X[col])\n\n# Chia dữ liệu thành tập huấn luyện và tập kiểm tra\nX_train, X_test, y_train, y_test = train_test_split(X, y_bowel_healthy, test_size=0.2, random_state=42)\n\n# Khởi tạo và huấn luyện mô hình RandomForest\nmodel = RandomForestClassifier(n_estimators=100, random_state=42)\nmodel.fit(X_train, y_train)\n\n# Dự đoán kết quả trên tập kiểm tra\ny_pred_bowel_healthy = model.predict(X_test)\n\n# Tính toán các chỉ số đánh giá\ndef calculate_metrics(y_true, y_pred, num_classes=1):\n    if num_classes == 1:\n        cm = confusion_matrix(y_true, y_pred)\n        tn, fp, fn, tp = cm.ravel()\n        recall = tp / (tp + fn)  # Độ nhạy (Recall)\n        precision = tp / (tp + fp)  # Độ chính xác (Precision)\n        f1 = 2 * (precision * recall) / (precision + recall)  # F1-score\n        accuracy = (tp + tn) / (tp + tn + fp + fn)  # Độ chính xác (Accuracy)\n        specificity = tn / (tn + fp)  # Độ nhiễu (Specificity)\n    else:\n        recall = recall_score(y_true, y_pred, average='weighted', zero_division=0)\n        precision = precision_score(y_true, y_pred, average='weighted', zero_division=0)\n        f1 = f1_score(y_true, y_pred, average='weighted', zero_division=0)\n        accuracy = accuracy_score(y_true, y_pred)\n        specificity = None\n\n    return recall, precision, f1, accuracy, specificity\n\n# Tính toán các chỉ số đánh giá\nrecall_bowel_healthy, precision_bowel_healthy, f1_bowel_healthy, accuracy_bowel_healthy, specificity_bowel_healthy = calculate_metrics(y_test, y_pred_bowel_healthy)\n\n# In kết quả\nprint(f\"Results for bowel_healthy:\")\nprint(f\"Recall: {recall_bowel_healthy:.4f}\")\nprint(f\"Precision: {precision_bowel_healthy:.4f}\")\nprint(f\"F1-score: {f1_bowel_healthy:.4f}\")\nprint(f\"Accuracy: {accuracy_bowel_healthy:.4f}\")\nprint(f\"Specificity: {specificity_bowel_healthy:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T16:48:18.477917Z","iopub.execute_input":"2024-11-12T16:48:18.478697Z","iopub.status.idle":"2024-11-12T16:48:19.200969Z","shell.execute_reply.started":"2024-11-12T16:48:18.478663Z","shell.execute_reply":"2024-11-12T16:48:19.199946Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import confusion_matrix, recall_score, precision_score, f1_score, accuracy_score\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\n\n# Đọc dữ liệu nhãn từ file train\ntrain_labels = pd.read_csv(\"/kaggle/input/rsna-atd-512x512-png-v2-dataset/train.csv\")\n\n# Kiểm tra tên cột trong DataFrame để xác định các cột không phải số\nprint(train_labels.columns)\n\n# Tạo các đặc trưng (features) và nhãn (labels)\nX = train_labels.drop(columns=['extravasation_healthy',\n       'extravasation_injury'])  # Xóa các cột nhãn để chỉ lấy đặc trưng\ny_bowel_healthy = train_labels['extravasation_healthy']\n\n# Chuyển đổi các cột không phải số (nếu có) thành số\n# Xác định các cột kiểu chuỗi và chuyển chúng thành giá trị số\ncategorical_columns = X.select_dtypes(include=['object']).columns\n\n# Áp dụng LabelEncoder cho tất cả các cột không phải số\nencoder = LabelEncoder()\nfor col in categorical_columns:\n    X[col] = encoder.fit_transform(X[col])\n\n# Chia dữ liệu thành tập huấn luyện và tập kiểm tra\nX_train, X_test, y_train, y_test = train_test_split(X, y_bowel_healthy, test_size=0.2, random_state=42)\n\n# Khởi tạo và huấn luyện mô hình RandomForest\nmodel = RandomForestClassifier(n_estimators=100, random_state=42)\nmodel.fit(X_train, y_train)\n\n# Dự đoán kết quả trên tập kiểm tra\ny_pred_bowel_healthy = model.predict(X_test)\n\n# Tính toán các chỉ số đánh giá\ndef calculate_metrics(y_true, y_pred, num_classes=1):\n    if num_classes == 1:\n        cm = confusion_matrix(y_true, y_pred)\n        tn, fp, fn, tp = cm.ravel()\n        recall = tp / (tp + fn)  # Độ nhạy (Recall)\n        precision = tp / (tp + fp)  # Độ chính xác (Precision)\n        f1 = 2 * (precision * recall) / (precision + recall)  # F1-score\n        accuracy = (tp + tn) / (tp + tn + fp + fn)  # Độ chính xác (Accuracy)\n        specificity = tn / (tn + fp)  # Độ nhiễu (Specificity)\n    else:\n        recall = recall_score(y_true, y_pred, average='weighted', zero_division=0)\n        precision = precision_score(y_true, y_pred, average='weighted', zero_division=0)\n        f1 = f1_score(y_true, y_pred, average='weighted', zero_division=0)\n        accuracy = accuracy_score(y_true, y_pred)\n        specificity = None\n\n    return recall, precision, f1, accuracy, specificity\n\n# Tính toán các chỉ số đánh giá\nrecall_bowel_healthy, precision_bowel_healthy, f1_bowel_healthy, accuracy_bowel_healthy, specificity_bowel_healthy = calculate_metrics(y_test, y_pred_bowel_healthy)\n\n# In kết quả\nprint(f\"Results for bowel_healthy:\")\nprint(f\"Recall: {recall_bowel_healthy:.4f}\")\nprint(f\"Precision: {precision_bowel_healthy:.4f}\")\nprint(f\"F1-score: {f1_bowel_healthy:.4f}\")\nprint(f\"Accuracy: {accuracy_bowel_healthy:.4f}\")\nprint(f\"Specificity: {specificity_bowel_healthy:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T16:49:09.828165Z","iopub.execute_input":"2024-11-12T16:49:09.828802Z","iopub.status.idle":"2024-11-12T16:49:10.520189Z","shell.execute_reply.started":"2024-11-12T16:49:09.828769Z","shell.execute_reply":"2024-11-12T16:49:10.519264Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import confusion_matrix, recall_score, precision_score, f1_score, accuracy_score\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\n\n# Đọc dữ liệu nhãn từ file train\ntrain_labels = pd.read_csv(\"/kaggle/input/rsna-atd-512x512-png-v2-dataset/train.csv\")\n\n# Kiểm tra tên cột trong DataFrame để xác định các cột không phải số\nprint(train_labels.columns)\n\n# Tạo các đặc trưng (features) và nhãn (labels)\nX = train_labels.drop(columns=['kidney_healthy', 'kidney_low', 'kidney_high'])  # Xóa các cột nhãn để chỉ lấy đặc trưng\ny_bowel_healthy = train_labels['kidney_healthy']\n\n# Chuyển đổi các cột không phải số (nếu có) thành số\n# Xác định các cột kiểu chuỗi và chuyển chúng thành giá trị số\ncategorical_columns = X.select_dtypes(include=['object']).columns\n\n# Áp dụng LabelEncoder cho tất cả các cột không phải số\nencoder = LabelEncoder()\nfor col in categorical_columns:\n    X[col] = encoder.fit_transform(X[col])\n\n# Chia dữ liệu thành tập huấn luyện và tập kiểm tra\nX_train, X_test, y_train, y_test = train_test_split(X, y_bowel_healthy, test_size=0.2, random_state=42)\n\n# Khởi tạo và huấn luyện mô hình RandomForest\nmodel = RandomForestClassifier(n_estimators=100, random_state=42)\nmodel.fit(X_train, y_train)\n\n# Dự đoán kết quả trên tập kiểm tra\ny_pred_bowel_healthy = model.predict(X_test)\n\n# Tính toán các chỉ số đánh giá\ndef calculate_metrics(y_true, y_pred, num_classes=1):\n    if num_classes == 1:\n        cm = confusion_matrix(y_true, y_pred)\n        tn, fp, fn, tp = cm.ravel()\n        recall = tp / (tp + fn)  # Độ nhạy (Recall)\n        precision = tp / (tp + fp)  # Độ chính xác (Precision)\n        f1 = 2 * (precision * recall) / (precision + recall)  # F1-score\n        accuracy = (tp + tn) / (tp + tn + fp + fn)  # Độ chính xác (Accuracy)\n        specificity = tn / (tn + fp)  # Độ nhiễu (Specificity)\n    else:\n        recall = recall_score(y_true, y_pred, average='weighted', zero_division=0)\n        precision = precision_score(y_true, y_pred, average='weighted', zero_division=0)\n        f1 = f1_score(y_true, y_pred, average='weighted', zero_division=0)\n        accuracy = accuracy_score(y_true, y_pred)\n        specificity = None\n\n    return recall, precision, f1, accuracy, specificity\n\n# Tính toán các chỉ số đánh giá\nrecall_bowel_healthy, precision_bowel_healthy, f1_bowel_healthy, accuracy_bowel_healthy, specificity_bowel_healthy = calculate_metrics(y_test, y_pred_bowel_healthy)\n\n# In kết quả\nprint(f\"Results for bowel_healthy:\")\nprint(f\"Recall: {recall_bowel_healthy:.4f}\")\nprint(f\"Precision: {precision_bowel_healthy:.4f}\")\nprint(f\"F1-score: {f1_bowel_healthy:.4f}\")\nprint(f\"Accuracy: {accuracy_bowel_healthy:.4f}\")\nprint(f\"Specificity: {specificity_bowel_healthy:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T16:49:42.888764Z","iopub.execute_input":"2024-11-12T16:49:42.889109Z","iopub.status.idle":"2024-11-12T16:49:43.720018Z","shell.execute_reply.started":"2024-11-12T16:49:42.889084Z","shell.execute_reply":"2024-11-12T16:49:43.719091Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import confusion_matrix, recall_score, precision_score, f1_score, accuracy_score\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\n\n# Đọc dữ liệu nhãn từ file train\ntrain_labels = pd.read_csv(\"/kaggle/input/rsna-atd-512x512-png-v2-dataset/train.csv\")\n\n# Kiểm tra tên cột trong DataFrame để xác định các cột không phải số\nprint(train_labels.columns)\n\n# Tạo các đặc trưng (features) và nhãn (labels)\nX = train_labels.drop(columns=['liver_healthy', 'liver_low', 'liver_high'])  # Xóa các cột nhãn để chỉ lấy đặc trưng\ny_bowel_healthy = train_labels['liver_healthy']\n\n# Chuyển đổi các cột không phải số (nếu có) thành số\n# Xác định các cột kiểu chuỗi và chuyển chúng thành giá trị số\ncategorical_columns = X.select_dtypes(include=['object']).columns\n\n# Áp dụng LabelEncoder cho tất cả các cột không phải số\nencoder = LabelEncoder()\nfor col in categorical_columns:\n    X[col] = encoder.fit_transform(X[col])\n\n# Chia dữ liệu thành tập huấn luyện và tập kiểm tra\nX_train, X_test, y_train, y_test = train_test_split(X, y_bowel_healthy, test_size=0.2, random_state=42)\n\n# Khởi tạo và huấn luyện mô hình RandomForest\nmodel = RandomForestClassifier(n_estimators=100, random_state=42)\nmodel.fit(X_train, y_train)\n\n# Dự đoán kết quả trên tập kiểm tra\ny_pred_bowel_healthy = model.predict(X_test)\n\n# Tính toán các chỉ số đánh giá\ndef calculate_metrics(y_true, y_pred, num_classes=1):\n    if num_classes == 1:\n        cm = confusion_matrix(y_true, y_pred)\n        tn, fp, fn, tp = cm.ravel()\n        recall = tp / (tp + fn)  # Độ nhạy (Recall)\n        precision = tp / (tp + fp)  # Độ chính xác (Precision)\n        f1 = 2 * (precision * recall) / (precision + recall)  # F1-score\n        accuracy = (tp + tn) / (tp + tn + fp + fn)  # Độ chính xác (Accuracy)\n        specificity = tn / (tn + fp)  # Độ nhiễu (Specificity)\n    else:\n        recall = recall_score(y_true, y_pred, average='weighted', zero_division=0)\n        precision = precision_score(y_true, y_pred, average='weighted', zero_division=0)\n        f1 = f1_score(y_true, y_pred, average='weighted', zero_division=0)\n        accuracy = accuracy_score(y_true, y_pred)\n        specificity = None\n\n    return recall, precision, f1, accuracy, specificity\n\n# Tính toán các chỉ số đánh giá\nrecall_bowel_healthy, precision_bowel_healthy, f1_bowel_healthy, accuracy_bowel_healthy, specificity_bowel_healthy = calculate_metrics(y_test, y_pred_bowel_healthy)\n\n# In kết quả\nprint(f\"Results for bowel_healthy:\")\nprint(f\"Recall: {recall_bowel_healthy:.4f}\")\nprint(f\"Precision: {precision_bowel_healthy:.4f}\")\nprint(f\"F1-score: {f1_bowel_healthy:.4f}\")\nprint(f\"Accuracy: {accuracy_bowel_healthy:.4f}\")\nprint(f\"Specificity: {specificity_bowel_healthy:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T16:50:11.303667Z","iopub.execute_input":"2024-11-12T16:50:11.304356Z","iopub.status.idle":"2024-11-12T16:50:12.148379Z","shell.execute_reply.started":"2024-11-12T16:50:11.304327Z","shell.execute_reply":"2024-11-12T16:50:12.147467Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import confusion_matrix, recall_score, precision_score, f1_score, accuracy_score\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\n\n# Đọc dữ liệu nhãn từ file train\ntrain_labels = pd.read_csv(\"/kaggle/input/rsna-atd-512x512-png-v2-dataset/train.csv\")\n\n# Kiểm tra tên cột trong DataFrame để xác định các cột không phải số\nprint(train_labels.columns)\n\n# Tạo các đặc trưng (features) và nhãn (labels)\nX = train_labels.drop(columns=['spleen_low', 'spleen_high', 'any_injury'])  # Xóa các cột nhãn để chỉ lấy đặc trưng\ny_bowel_healthy = train_labels['spleen_low']\n\n# Chuyển đổi các cột không phải số (nếu có) thành số\n# Xác định các cột kiểu chuỗi và chuyển chúng thành giá trị số\ncategorical_columns = X.select_dtypes(include=['object']).columns\n\n# Áp dụng LabelEncoder cho tất cả các cột không phải số\nencoder = LabelEncoder()\nfor col in categorical_columns:\n    X[col] = encoder.fit_transform(X[col])\n\n# Chia dữ liệu thành tập huấn luyện và tập kiểm tra\nX_train, X_test, y_train, y_test = train_test_split(X, y_bowel_healthy, test_size=0.2, random_state=42)\n\n# Khởi tạo và huấn luyện mô hình RandomForest\nmodel = RandomForestClassifier(n_estimators=100, random_state=42)\nmodel.fit(X_train, y_train)\n\n# Dự đoán kết quả trên tập kiểm tra\ny_pred_bowel_healthy = model.predict(X_test)\n\n# Tính toán các chỉ số đánh giá\ndef calculate_metrics(y_true, y_pred, num_classes=1):\n    if num_classes == 1:\n        cm = confusion_matrix(y_true, y_pred)\n        tn, fp, fn, tp = cm.ravel()\n        recall = tp / (tp + fn)  # Độ nhạy (Recall)\n        precision = tp / (tp + fp)  # Độ chính xác (Precision)\n        f1 = 2 * (precision * recall) / (precision + recall)  # F1-score\n        accuracy = (tp + tn) / (tp + tn + fp + fn)  # Độ chính xác (Accuracy)\n        specificity = tn / (tn + fp)  # Độ nhiễu (Specificity)\n    else:\n        recall = recall_score(y_true, y_pred, average='weighted', zero_division=0)\n        precision = precision_score(y_true, y_pred, average='weighted', zero_division=0)\n        f1 = f1_score(y_true, y_pred, average='weighted', zero_division=0)\n        accuracy = accuracy_score(y_true, y_pred)\n        specificity = None\n\n    return recall, precision, f1, accuracy, specificity\n\n# Tính toán các chỉ số đánh giá\nrecall_bowel_healthy, precision_bowel_healthy, f1_bowel_healthy, accuracy_bowel_healthy, specificity_bowel_healthy = calculate_metrics(y_test, y_pred_bowel_healthy)\n\n# In kết quả\nprint(f\"Results for bowel_healthy:\")\nprint(f\"Recall: {recall_bowel_healthy:.4f}\")\nprint(f\"Precision: {precision_bowel_healthy:.4f}\")\nprint(f\"F1-score: {f1_bowel_healthy:.4f}\")\nprint(f\"Accuracy: {accuracy_bowel_healthy:.4f}\")\nprint(f\"Specificity: {specificity_bowel_healthy:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T16:51:31.362408Z","iopub.execute_input":"2024-11-12T16:51:31.362768Z","iopub.status.idle":"2024-11-12T16:51:32.039214Z","shell.execute_reply.started":"2024-11-12T16:51:31.362742Z","shell.execute_reply":"2024-11-12T16:51:32.038283Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import confusion_matrix, recall_score, precision_score, f1_score, accuracy_score\nfrom sklearn.model_selection import train_test_split\n\n# Đọc dữ liệu nhãn từ file train\ntrain_labels = pd.read_csv(\"/kaggle/input/rsna-2023-abdominal-trauma-detection/train_2024.csv\")  # Đảm bảo đường dẫn đúng\n\n# Tạo các đặc trưng (features) và nhãn (labels)\nX = train_labels.drop(columns=['bowel_healthy', 'bowel_injury', 'any_injury'])  # Xóa các cột nhãn để chỉ lấy đặc trưng\ny_bowel_healthy = train_labels['bowel_healthy']\n\n# Chia dữ liệu thành tập huấn luyện và tập kiểm tra\nX_train, X_test, y_train, y_test = train_test_split(X, y_bowel_healthy, test_size=0.2, random_state=42)\n\n# Khởi tạo và huấn luyện mô hình RandomForest\nmodel = RandomForestClassifier(n_estimators=100, random_state=42)\nmodel.fit(X_train, y_train)\n\n# Dự đoán kết quả trên tập kiểm tra\ny_pred_bowel_healthy = model.predict(X_test)\n\n# Tính toán các chỉ số đánh giá\ndef calculate_metrics(y_true, y_pred, num_classes=1):\n    if num_classes == 1:\n        cm = confusion_matrix(y_true, y_pred)\n        tn, fp, fn, tp = cm.ravel()\n        recall = tp / (tp + fn)  # Độ nhạy (Recall)\n        precision = tp / (tp + fp)  # Độ chính xác (Precision)\n        f1 = 2 * (precision * recall) / (precision + recall)  # F1-score\n        accuracy = (tp + tn) / (tp + tn + fp + fn)  # Độ chính xác (Accuracy)\n        specificity = tn / (tn + fp)  # Độ nhiễu (Specificity)\n    else:\n        recall = recall_score(y_true, y_pred, average='weighted', zero_division=0)\n        precision = precision_score(y_true, y_pred, average='weighted', zero_division=0)\n        f1 = f1_score(y_true, y_pred, average='weighted', zero_division=0)\n        accuracy = accuracy_score(y_true, y_pred)\n        specificity = None\n\n    return recall, precision, f1, accuracy, specificity\n\n# Tính toán các chỉ số đánh giá\nrecall_bowel_healthy, precision_bowel_healthy, f1_bowel_healthy, accuracy_bowel_healthy, specificity_bowel_healthy = calculate_metrics(y_test, y_pred_bowel_healthy)\n\n# In kết quả\nprint(f\"Results for bowel_healthy:\")\nprint(f\"Recall: {recall_bowel_healthy:.4f}\")\nprint(f\"Precision: {precision_bowel_healthy:.4f}\")\nprint(f\"F1-score: {f1_bowel_healthy:.4f}\")\nprint(f\"Accuracy: {accuracy_bowel_healthy:.4f}\")\nprint(f\"Specificity: {specificity_bowel_healthy:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T16:47:07.547726Z","iopub.execute_input":"2024-11-12T16:47:07.548336Z","iopub.status.idle":"2024-11-12T16:47:07.953748Z","shell.execute_reply.started":"2024-11-12T16:47:07.548305Z","shell.execute_reply":"2024-11-12T16:47:07.952805Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import confusion_matrix, recall_score, precision_score, f1_score, accuracy_score\nfrom sklearn.model_selection import train_test_split\n\n# Đọc dữ liệu nhãn từ file train\ntrain_labels = pd.read_csv(\"/kaggle/input/rsna-2023-abdominal-trauma-detection/train_2024.csv\")  # Đảm bảo đường dẫn đúng\n\n# Tạo các đặc trưng (features) và nhãn (labels)\nX = train_labels.drop(columns=['extravasation_healthy', 'extravasation_injury'])  # Xóa các cột nhãn để chỉ lấy đặc trưng\ny_bowel_healthy = train_labels['extravasation_healthy']\n\n# Chia dữ liệu thành tập huấn luyện và tập kiểm tra\nX_train, X_test, y_train, y_test = train_test_split(X, y_bowel_healthy, test_size=0.2, random_state=42)\n\n# Khởi tạo và huấn luyện mô hình RandomForest\nmodel = RandomForestClassifier(n_estimators=100, random_state=42)\nmodel.fit(X_train, y_train)\n\n# Dự đoán kết quả trên tập kiểm tra\ny_pred_bowel_healthy = model.predict(X_test)\n\n# Tính toán các chỉ số đánh giá\ndef calculate_metrics(y_true, y_pred, num_classes=1):\n    if num_classes == 1:\n        cm = confusion_matrix(y_true, y_pred)\n        tn, fp, fn, tp = cm.ravel()\n        recall = tp / (tp + fn)  # Độ nhạy (Recall)\n        precision = tp / (tp + fp)  # Độ chính xác (Precision)\n        f1 = 2 * (precision * recall) / (precision + recall)  # F1-score\n        accuracy = (tp + tn) / (tp + tn + fp + fn)  # Độ chính xác (Accuracy)\n        specificity = tn / (tn + fp)  # Độ nhiễu (Specificity)\n    else:\n        recall = recall_score(y_true, y_pred, average='weighted', zero_division=0)\n        precision = precision_score(y_true, y_pred, average='weighted', zero_division=0)\n        f1 = f1_score(y_true, y_pred, average='weighted', zero_division=0)\n        accuracy = accuracy_score(y_true, y_pred)\n        specificity = None\n\n    return recall, precision, f1, accuracy, specificity\n\n# Tính toán các chỉ số đánh giá\nrecall_bowel_healthy, precision_bowel_healthy, f1_bowel_healthy, accuracy_bowel_healthy, specificity_bowel_healthy = calculate_metrics(y_test, y_pred_bowel_healthy)\n\n# In kết quả\nprint(f\"Results for bowel_healthy:\")\nprint(f\"Recall: {recall_bowel_healthy:.4f}\")\nprint(f\"Precision: {precision_bowel_healthy:.4f}\")\nprint(f\"F1-score: {f1_bowel_healthy:.4f}\")\nprint(f\"Accuracy: {accuracy_bowel_healthy:.4f}\")\nprint(f\"Specificity: {specificity_bowel_healthy:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T15:12:16.065065Z","iopub.execute_input":"2024-11-12T15:12:16.065902Z","iopub.status.idle":"2024-11-12T15:12:16.371147Z","shell.execute_reply.started":"2024-11-12T15:12:16.065869Z","shell.execute_reply":"2024-11-12T15:12:16.370166Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import confusion_matrix, recall_score, precision_score, f1_score, accuracy_score\nfrom sklearn.model_selection import train_test_split\n\n# Đọc dữ liệu nhãn từ file train\ntrain_labels = pd.read_csv(\"/kaggle/input/rsna-2023-abdominal-trauma-detection/train_2024.csv\")  # Đảm bảo đường dẫn đúng\n\n# Tạo các đặc trưng (features) và nhãn (labels)\nX = train_labels.drop(columns=['kidney_healthy', 'kidney_low', 'kidney_high',])  # Xóa các cột nhãn để chỉ lấy đặc trưng\ny_bowel_healthy = train_labels['kidney_healthy']\n\n# Chia dữ liệu thành tập huấn luyện và tập kiểm tra\nX_train, X_test, y_train, y_test = train_test_split(X, y_bowel_healthy, test_size=0.2, random_state=42)\n\n# Khởi tạo và huấn luyện mô hình RandomForest\nmodel = RandomForestClassifier(n_estimators=100, random_state=42)\nmodel.fit(X_train, y_train)\n\n# Dự đoán kết quả trên tập kiểm tra\ny_pred_bowel_healthy = model.predict(X_test)\n\n# Tính toán các chỉ số đánh giá\ndef calculate_metrics(y_true, y_pred, num_classes=1):\n    if num_classes == 1:\n        cm = confusion_matrix(y_true, y_pred)\n        tn, fp, fn, tp = cm.ravel()\n        recall = tp / (tp + fn)  # Độ nhạy (Recall)\n        precision = tp / (tp + fp)  # Độ chính xác (Precision)\n        f1 = 2 * (precision * recall) / (precision + recall)  # F1-score\n        accuracy = (tp + tn) / (tp + tn + fp + fn)  # Độ chính xác (Accuracy)\n        specificity = tn / (tn + fp)  # Độ nhiễu (Specificity)\n    else:\n        recall = recall_score(y_true, y_pred, average='weighted', zero_division=0)\n        precision = precision_score(y_true, y_pred, average='weighted', zero_division=0)\n        f1 = f1_score(y_true, y_pred, average='weighted', zero_division=0)\n        accuracy = accuracy_score(y_true, y_pred)\n        specificity = None\n\n    return recall, precision, f1, accuracy, specificity\n\n# Tính toán các chỉ số đánh giá\nrecall_bowel_healthy, precision_bowel_healthy, f1_bowel_healthy, accuracy_bowel_healthy, specificity_bowel_healthy = calculate_metrics(y_test, y_pred_bowel_healthy)\n\n# In kết quả\nprint(f\"Results for bowel_healthy:\")\nprint(f\"Recall: {recall_bowel_healthy:.4f}\")\nprint(f\"Precision: {precision_bowel_healthy:.4f}\")\nprint(f\"F1-score: {f1_bowel_healthy:.4f}\")\nprint(f\"Accuracy: {accuracy_bowel_healthy:.4f}\")\nprint(f\"Specificity: {specificity_bowel_healthy:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T15:13:05.445492Z","iopub.execute_input":"2024-11-12T15:13:05.446456Z","iopub.status.idle":"2024-11-12T15:13:05.758553Z","shell.execute_reply.started":"2024-11-12T15:13:05.446411Z","shell.execute_reply":"2024-11-12T15:13:05.757596Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import confusion_matrix, recall_score, precision_score, f1_score, accuracy_score\nfrom sklearn.model_selection import train_test_split\n\n# Đọc dữ liệu nhãn từ file train\ntrain_labels = pd.read_csv(\"/kaggle/input/rsna-2023-abdominal-trauma-detection/train_2024.csv\")  # Đảm bảo đường dẫn đúng\n\n# Tạo các đặc trưng (features) và nhãn (labels)\nX = train_labels.drop(columns=['liver_healthy', 'liver_low', 'liver_high',])  # Xóa các cột nhãn để chỉ lấy đặc trưng\ny_bowel_healthy = train_labels['liver_healthy']\n\n# Chia dữ liệu thành tập huấn luyện và tập kiểm tra\nX_train, X_test, y_train, y_test = train_test_split(X, y_bowel_healthy, test_size=0.2, random_state=42)\n\n# Khởi tạo và huấn luyện mô hình RandomForest\nmodel = RandomForestClassifier(n_estimators=100, random_state=42)\nmodel.fit(X_train, y_train)\n\n# Dự đoán kết quả trên tập kiểm tra\ny_pred_bowel_healthy = model.predict(X_test)\n\n# Tính toán các chỉ số đánh giá\ndef calculate_metrics(y_true, y_pred, num_classes=1):\n    if num_classes == 1:\n        cm = confusion_matrix(y_true, y_pred)\n        tn, fp, fn, tp = cm.ravel()\n        recall = tp / (tp + fn)  # Độ nhạy (Recall)\n        precision = tp / (tp + fp)  # Độ chính xác (Precision)\n        f1 = 2 * (precision * recall) / (precision + recall)  # F1-score\n        accuracy = (tp + tn) / (tp + tn + fp + fn)  # Độ chính xác (Accuracy)\n        specificity = tn / (tn + fp)  # Độ nhiễu (Specificity)\n    else:\n        recall = recall_score(y_true, y_pred, average='weighted', zero_division=0)\n        precision = precision_score(y_true, y_pred, average='weighted', zero_division=0)\n        f1 = f1_score(y_true, y_pred, average='weighted', zero_division=0)\n        accuracy = accuracy_score(y_true, y_pred)\n        specificity = None\n\n    return recall, precision, f1, accuracy, specificity\n\n# Tính toán các chỉ số đánh giá\nrecall_bowel_healthy, precision_bowel_healthy, f1_bowel_healthy, accuracy_bowel_healthy, specificity_bowel_healthy = calculate_metrics(y_test, y_pred_bowel_healthy)\n\n# In kết quả\nprint(f\"Results for bowel_healthy:\")\nprint(f\"Recall: {recall_bowel_healthy:.4f}\")\nprint(f\"Precision: {precision_bowel_healthy:.4f}\")\nprint(f\"F1-score: {f1_bowel_healthy:.4f}\")\nprint(f\"Accuracy: {accuracy_bowel_healthy:.4f}\")\nprint(f\"Specificity: {specificity_bowel_healthy:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T15:13:35.627472Z","iopub.execute_input":"2024-11-12T15:13:35.627844Z","iopub.status.idle":"2024-11-12T15:13:35.94568Z","shell.execute_reply.started":"2024-11-12T15:13:35.627813Z","shell.execute_reply":"2024-11-12T15:13:35.944757Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import confusion_matrix, recall_score, precision_score, f1_score, accuracy_score\nfrom sklearn.model_selection import train_test_split\n\n# Đọc dữ liệu nhãn từ file train\ntrain_labels = pd.read_csv(\"/kaggle/input/rsna-2023-abdominal-trauma-detection/train_2024.csv\")  # Đảm bảo đường dẫn đúng\n\n# Tạo các đặc trưng (features) và nhãn (labels)\nX = train_labels.drop(columns=['spleen_healthy', 'spleen_low', 'spleen_high'])  # Xóa các cột nhãn để chỉ lấy đặc trưng\ny_bowel_healthy = train_labels['spleen_healthy']\n\n# Chia dữ liệu thành tập huấn luyện và tập kiểm tra\nX_train, X_test, y_train, y_test = train_test_split(X, y_bowel_healthy, test_size=0.2, random_state=42)\n\n# Khởi tạo và huấn luyện mô hình RandomForest\nmodel = RandomForestClassifier(n_estimators=100, random_state=42)\nmodel.fit(X_train, y_train)\n\n# Dự đoán kết quả trên tập kiểm tra\ny_pred_bowel_healthy = model.predict(X_test)\n\n# Tính toán các chỉ số đánh giá\ndef calculate_metrics(y_true, y_pred, num_classes=1):\n    if num_classes == 1:\n        cm = confusion_matrix(y_true, y_pred)\n        tn, fp, fn, tp = cm.ravel()\n        recall = tp / (tp + fn)  # Độ nhạy (Recall)\n        precision = tp / (tp + fp)  # Độ chính xác (Precision)\n        f1 = 2 * (precision * recall) / (precision + recall)  # F1-score\n        accuracy = (tp + tn) / (tp + tn + fp + fn)  # Độ chính xác (Accuracy)\n        specificity = tn / (tn + fp)  # Độ nhiễu (Specificity)\n    else:\n        recall = recall_score(y_true, y_pred, average='weighted', zero_division=0)\n        precision = precision_score(y_true, y_pred, average='weighted', zero_division=0)\n        f1 = f1_score(y_true, y_pred, average='weighted', zero_division=0)\n        accuracy = accuracy_score(y_true, y_pred)\n        specificity = None\n\n    return recall, precision, f1, accuracy, specificity\n\n# Tính toán các chỉ số đánh giá\nrecall_bowel_healthy, precision_bowel_healthy, f1_bowel_healthy, accuracy_bowel_healthy, specificity_bowel_healthy = calculate_metrics(y_test, y_pred_bowel_healthy)\n\n# In kết quả\nprint(f\"Results for bowel_healthy:\")\nprint(f\"Recall: {recall_bowel_healthy:.4f}\")\nprint(f\"Precision: {precision_bowel_healthy:.4f}\")\nprint(f\"F1-score: {f1_bowel_healthy:.4f}\")\nprint(f\"Accuracy: {accuracy_bowel_healthy:.4f}\")\nprint(f\"Specificity: {specificity_bowel_healthy:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T16:07:22.037224Z","iopub.execute_input":"2024-11-12T16:07:22.037938Z","iopub.status.idle":"2024-11-12T16:07:22.351364Z","shell.execute_reply.started":"2024-11-12T16:07:22.037908Z","shell.execute_reply":"2024-11-12T16:07:22.350382Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import confusion_matrix, recall_score, precision_score, f1_score, accuracy_score\nfrom sklearn.model_selection import train_test_split\n\n# Đọc dữ liệu nhãn từ file train\ntrain_labels = pd.read_csv(\"/kaggle/input/rsna-2023-abdominal-trauma-detection/train_2024.csv\")  # Đảm bảo đường dẫn đúng\n\n# Tạo các đặc trưng (features) và nhãn (labels)\nX = train_labels.drop(columns=['bowel_healthy', 'bowel_injury', 'any_injury', \n                               'extravasation_healthy', 'extravasation_injury',\n                               'kidney_healthy', 'kidney_low', 'kidney_high',\n                               'liver_healthy', 'liver_low', 'liver_high',\n                               'spleen_healthy', 'spleen_low', 'spleen_high'])  # Xóa các cột nhãn không cần thiết\n\n# Định nghĩa các nhãn mục tiêu\ny_bowel_healthy = train_labels['bowel_healthy']\ny_extra_healthy = train_labels['extravasation_healthy']\ny_liver_healthy = train_labels['liver_healthy']\ny_kidney_healthy = train_labels['kidney_healthy']\ny_spleen_healthy = train_labels['spleen_healthy']\n\n# Tạo hàm tính toán các chỉ số đánh giá\ndef calculate_metrics(y_true, y_pred, num_classes=1):\n    if num_classes == 1:\n        cm = confusion_matrix(y_true, y_pred)\n        tn, fp, fn, tp = cm.ravel()\n        recall = tp / (tp + fn)  # Độ nhạy (Recall)\n        precision = tp / (tp + fp)  # Độ chính xác (Precision)\n        f1 = 2 * (precision * recall) / (precision + recall)  # F1-score\n        accuracy = (tp + tn) / (tp + tn + fp + fn)  # Độ chính xác (Accuracy)\n        specificity = tn / (tn + fp)  # Độ nhiễu (Specificity)\n    else:\n        recall = recall_score(y_true, y_pred, average='weighted', zero_division=0)\n        precision = precision_score(y_true, y_pred, average='weighted', zero_division=0)\n        f1 = f1_score(y_true, y_pred, average='weighted', zero_division=0)\n        accuracy = accuracy_score(y_true, y_pred)\n        specificity = None\n\n    return recall, precision, f1, accuracy, specificity\n\n# Hàm tính toán và in kết quả cho từng nhãn\ndef evaluate_model_for_label(X_train, X_test, y_train, y_test, label_name):\n    model = RandomForestClassifier(n_estimators=100, random_state=42)\n    model.fit(X_train, y_train)  # Huấn luyện mô hình\n\n    # Dự đoán trên tập kiểm tra\n    y_pred = model.predict(X_test)\n\n    # Tính toán các chỉ số đánh giá\n    recall, precision, f1, accuracy, specificity = calculate_metrics(y_test, y_pred)\n\n    # In kết quả cho từng nhãn\n    print(f\"Results for {label_name}:\")\n    print(f\"Recall: {recall:.4f}\")\n    print(f\"Precision: {precision:.4f}\")\n    print(f\"F1-score: {f1:.4f}\")\n    print(f\"Accuracy: {accuracy:.4f}\")\n    print(f\"Specificity: {specificity:.4f}\")\n    print()  # Dòng trống giữa các kết quả\n\n# Chia dữ liệu thành tập huấn luyện và kiểm tra\nX_train, X_test, y_train, y_test = train_test_split(X, y_bowel_healthy, test_size=0.2, random_state=42)\n\n# Đánh giá mô hình cho từng nhãn\nevaluate_model_for_label(X_train, X_test, y_train, y_test, \"bowel_healthy\")\n\n# Đánh giá cho các nhãn khác\nevaluate_model_for_label(X_train, X_test, y_train, y_test, \"extravasation_healthy\")\nevaluate_model_for_label(X_train, X_test, y_train, y_test, \"liver_healthy\")\nevaluate_model_for_label(X_train, X_test, y_train, y_test, \"kidney_healthy\")\nevaluate_model_for_label(X_train, X_test, y_train, y_test, \"spleen_healthy\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T15:02:04.675977Z","iopub.execute_input":"2024-11-12T15:02:04.676742Z","iopub.status.idle":"2024-11-12T15:02:06.878184Z","shell.execute_reply.started":"2024-11-12T15:02:04.676708Z","shell.execute_reply":"2024-11-12T15:02:06.877008Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Lặp qua các key trong history để vẽ accuracy\nfor i in history.history.keys():\n    if i.endswith(\"_accuracy\") and not i == \"val_accuracy\":\n        plt.plot(history.history[i], label=i)\n\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Accuracy\")\nplt.legend(loc=(1.05, 0.0))\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T15:15:02.111881Z","iopub.execute_input":"2024-11-12T15:15:02.11254Z","iopub.status.idle":"2024-11-12T15:15:02.491211Z","shell.execute_reply.started":"2024-11-12T15:15:02.112509Z","shell.execute_reply":"2024-11-12T15:15:02.490307Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i in history.history.keys():\n    if i.endswith(\"_loss\") and not i ==\"val_loss\":\n        plt.plot(history.history[i], label=i)\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Loss\")\nplt.legend(loc=(1.05,0.0))\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T15:15:06.332127Z","iopub.execute_input":"2024-11-12T15:15:06.332972Z","iopub.status.idle":"2024-11-12T15:15:06.631136Z","shell.execute_reply.started":"2024-11-12T15:15:06.332938Z","shell.execute_reply":"2024-11-12T15:15:06.630254Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i in history.history.keys():\n    if i.endswith(\"accuracy\") and not i ==\"val_accuracy\":\n        plt.plot(history.history[i], label=i)\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Loss\")\nplt.legend(loc=(1.05,0.0))\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T15:15:14.508413Z","iopub.execute_input":"2024-11-12T15:15:14.508803Z","iopub.status.idle":"2024-11-12T15:15:14.887377Z","shell.execute_reply.started":"2024-11-12T15:15:14.508774Z","shell.execute_reply":"2024-11-12T15:15:14.886475Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Lặp qua các key trong history để vẽ accuracy (loại bỏ val_accuracy)\nfor i in history.history.keys():\n    if i.endswith(\"accuracy\") and \"val_\" not in i:\n        plt.plot(history.history[i], label=i)\n\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Accuracy\")\nplt.legend(loc=(1.05, 0.0))\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T15:15:19.414539Z","iopub.execute_input":"2024-11-12T15:15:19.415282Z","iopub.status.idle":"2024-11-12T15:15:19.734834Z","shell.execute_reply.started":"2024-11-12T15:15:19.41525Z","shell.execute_reply":"2024-11-12T15:15:19.733897Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Kiểm tra các khóa trong history\nprint(history.history.keys())\n\n# Các metric bạn muốn vẽ (chỉ vẽ các accuracy mà không có \"val_\")\nmetrics_to_plot = [key for key in history.history.keys() if key.endswith('accuracy') and 'val_' not in key]\n\n# Tạo biểu đồ cho cả huấn luyện và validation\nfor metric in metrics_to_plot:\n    plt.figure(figsize=(8, 6))  # Tạo một biểu đồ mới cho mỗi metric\n    plt.plot(history.history[metric], label=f'Training {metric}')\n    \n    # Kiểm tra và vẽ độ chính xác của validation nếu có\n    val_metric = f'val_{metric}'\n    if val_metric in history.history:\n        plt.plot(history.history[val_metric], label=f'Validation {metric}')\n    \n    # Thêm số epoch vào trục x (tự động từ 0 đến n)\n    epochs = range(1, len(history.history[metric]) + 1)\n    \n    # Thêm ticks với khoảng cách 25\n    step_size = 25\n    epoch_ticks = [epoch for epoch in epochs if epoch % step_size == 0]\n    plt.xticks(epoch_ticks)\n    \n    # Cài đặt các thông số cho biểu đồ\n    plt.xlabel('Epochs')\n    plt.ylabel('Accuracy')\n    plt.title(f'{metric.capitalize()} over Epochs')\n    plt.legend(loc='upper left')\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T15:39:50.996566Z","iopub.execute_input":"2024-11-12T15:39:50.996943Z","iopub.status.idle":"2024-11-12T15:39:52.48832Z","shell.execute_reply.started":"2024-11-12T15:39:50.996914Z","shell.execute_reply":"2024-11-12T15:39:52.487347Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.plot(history.history[\"val_loss\"], label=\"val_loss\")\nplt.plot(history.history[\"loss\"], label=\"loss\")\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T15:15:27.691522Z","iopub.execute_input":"2024-11-12T15:15:27.692405Z","iopub.status.idle":"2024-11-12T15:15:27.950758Z","shell.execute_reply.started":"2024-11-12T15:15:27.692365Z","shell.execute_reply":"2024-11-12T15:15:27.949816Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Kiểm tra danh sách tên cột\nprint(df.columns)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T16:58:14.148307Z","iopub.execute_input":"2024-11-12T16:58:14.148682Z","iopub.status.idle":"2024-11-12T16:58:14.153955Z","shell.execute_reply.started":"2024-11-12T16:58:14.148647Z","shell.execute_reply":"2024-11-12T16:58:14.153026Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Đảm bảo rằng tệp CSV hoặc nguồn dữ liệu của bạn đã được nạp vào DataFrame\ndf = pd.read_csv(\"/kaggle/input/rsna-2023-abdominal-trauma-detection/train_2024.csv\")  # Thay thế bằng đường dẫn đúng\n\n# Kiểm tra các giá trị thiếu trong DataFrame\nmissing_values = df.isnull().sum()\nprint(\"Missing Values:\")\nprint(missing_values)\n\n# Danh sách các cột nhị phân cần chuyển đổi sang kiểu boolean\nbinary_columns = [\n    'bowel_healthy', 'bowel_injury', 'extravasation_healthy', 'extravasation_injury',\n    'kidney_healthy', 'kidney_low', 'kidney_high', 'liver_healthy', 'liver_low', 'liver_high',\n    'spleen_healthy', 'spleen_low', 'spleen_high'\n]\n\n# Chuyển đổi các cột nhị phân thành kiểu boolean\ndf[binary_columns] = df[binary_columns].astype(bool)\n\n# Giải quyết các vấn đề về chất lượng dữ liệu (nếu có, bạn có thể thêm các bước xử lý dữ liệu ở đây)\n\n# Hiển thị DataFrame sau khi đã xử lý\nprint(\"\\nPreprocessed DataFrame:\")\nprint(df.head())  # In ra 5 dòng đầu tiên của DataFrame đã xử lý để kiểm tra\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T15:40:25.536923Z","iopub.execute_input":"2024-11-12T15:40:25.537284Z","iopub.status.idle":"2024-11-12T15:40:25.574043Z","shell.execute_reply.started":"2024-11-12T15:40:25.537255Z","shell.execute_reply":"2024-11-12T15:40:25.573145Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Đọc tệp CSV chứa nhãn\ndf = pd.read_csv('/kaggle/input/rsna-atd-512x512-png-v2-dataset/train.csv')\n\n# Loại bỏ khoảng trắng thừa trong tên cột (nếu có)\ndf.columns = df.columns.str.strip()\n\n# Tạo danh sách các cột nhãn cho các cơ quan bạn muốn hiển thị\norgan_columns = ['kidney', 'liver', 'spleen']  # Chỉ giữ lại các cơ quan bạn muốn hiển thị\n\n# Vẽ ma trận nhầm lẫn cho các cơ quan\nfor organ in organ_columns:\n    # Đảm bảo tên cột chính xác sau khi loại bỏ khoảng trắng\n    y_true = df[f'{organ}_healthy']  # Nhãn thực tế (ví dụ: kidney_healthy)\n    y_pred = df[f'{organ}_low']  # Hoặc chọn nhãn khác như kidney_low (tùy vào mục đích của bạn)\n\n    # Tạo ma trận nhầm lẫn với 5 lớp (0 đến 4)\n    cm = confusion_matrix(y_true, y_pred, labels=[False, True, 2, 3, 4,5,6,7,8,9])  # Chỉ định 5 lớp (tùy theo dữ liệu của bạn)\n\n    # Vẽ ma trận nhầm lẫn\n    plt.figure(figsize=(10, 8))  # Thay đổi kích thước đồ thị cho phù hợp với ma trận\n    sns.heatmap(cm, annot=True, fmt='d', cmap='Blues', xticklabels=[0, 1, 2, 3, 4 , 5, 6, 7, 8, 9, 10], yticklabels=[0, 1, 2, 3, 4 , 5, 6, 7, 8, 9, 10])\n\n    # Thiết lập tiêu đề và nhãn\n    plt.title(f'Confusion Matrix for {organ.capitalize()}')\n    plt.xlabel('Predicted')\n    plt.ylabel('Actual')\n\n    # Hiển thị ma trận nhầm lẫn\n    plt.show()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Kiểm tra tất cả các cột trong DataFrame\nprint(df.columns)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T17:07:39.349036Z","iopub.execute_input":"2024-11-12T17:07:39.349866Z","iopub.status.idle":"2024-11-12T17:07:39.355075Z","shell.execute_reply.started":"2024-11-12T17:07:39.349833Z","shell.execute_reply":"2024-11-12T17:07:39.354176Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Đọc tệp CSV chứa nhãn\ndf = pd.read_csv('/kaggle/input/rsna-atd-512x512-png-v2-dataset/train.csv')\n\n# Loại bỏ khoảng trắng thừa trong tên cột (nếu có)\ndf.columns = df.columns.str.strip()\n\n# Tạo danh sách các cơ quan bạn muốn hiển thị\norgan_columns = ['bowel', 'extravasation', 'kidney', 'liver', 'spleen']\n\n# Vẽ ma trận nhầm lẫn cho các cơ quan\nfor organ in organ_columns:\n    # Đảm bảo tên cột chính xác sau khi loại bỏ khoảng trắng\n    y_true = df[f'{organ}_healthy']  # Nhãn thực tế (ví dụ: bowel_healthy)\n    \n    # Chọn cột dự đoán như 'injury', 'low' hoặc 'high', tùy vào mục đích của bạn\n    y_pred = df[f'{organ}_injury']  # Hoặc thay bằng 'low' hoặc 'high' tùy vào nhu cầu\n\n    # Tạo ma trận nhầm lẫn với 2 lớp (0: Healthy, 1: Injury)\n    cm_binary = confusion_matrix(y_true, y_pred, labels=[False, True])  # Chỉ so sánh Healthy vs Injury\n\n    # Vẽ ma trận nhầm lẫn 2 lớp\n    plt.figure(figsize=(10, 8))  \n    sns.heatmap(cm_binary, annot=True, fmt='d', cmap='Blues', xticklabels=['Healthy', 'Injury'], yticklabels=['Healthy', 'Injury'])\n    plt.title(f'Confusion Matrix for {organ.capitalize()} (2-class: Healthy vs Injury)')\n    plt.xlabel('Predicted')\n    plt.ylabel('Actual')\n    plt.show()\n\n    # Tạo ma trận nhầm lẫn với 10 lớp (0 đến 9) nếu dữ liệu có thể hỗ trợ\n    cm_10class = confusion_matrix(y_true, y_pred, labels=range(10))  # Sử dụng 10 lớp (0 đến 9)\n    \n    # Vẽ ma trận nhầm lẫn 10 lớp\n    plt.figure(figsize=(10, 8))  \n    sns.heatmap(cm_10class, annot=True, fmt='d', cmap='Blues', xticklabels=range(10), yticklabels=range(10))\n    plt.title(f'Confusion Matrix for {organ.capitalize()} (10-class)')\n    plt.xlabel('Predicted')\n    plt.ylabel('Actual')\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T17:10:14.570195Z","iopub.execute_input":"2024-11-12T17:10:14.570556Z","iopub.status.idle":"2024-11-12T17:10:16.421832Z","shell.execute_reply.started":"2024-11-12T17:10:14.57053Z","shell.execute_reply":"2024-11-12T17:10:16.420576Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Đọc tệp CSV chứa nhãn\ndf = pd.read_csv('/kaggle/input/rsna-atd-512x512-png-v2-dataset/train.csv')\n\n# Loại bỏ khoảng trắng thừa trong tên cột (nếu có)\ndf.columns = df.columns.str.strip()\n\n# Tạo danh sách các cột nhãn cho các cơ quan bạn muốn hiển thị\norgan_columns = ['kidney', 'liver', 'spleen']  # Các cơ quan cần hiển thị\n\n# Vẽ ma trận nhầm lẫn cho các cơ quan\nfor organ in organ_columns:\n    # Đảm bảo tên cột chính xác sau khi loại bỏ khoảng trắng\n    y_true = df[f'{organ}_healthy']  # Nhãn thực tế (ví dụ: kidney_healthy)\n    y_pred = df[f'{organ}_low']  # Hoặc chọn nhãn khác như kidney_low (tùy vào mục đích của bạn)\n\n    # Tạo ma trận nhầm lẫn với 10 lớp (0 đến 9)\n    cm = confusion_matrix(y_true, y_pred, labels=[False, True, 2, 3, 4, 5, 6, 7, 8, 9])  # Giả sử dữ liệu có lớp 0-9\n\n    # Vẽ ma trận nhầm lẫn\n    plt.figure(figsize=(10, 8))  # Thay đổi kích thước đồ thị cho phù hợp với ma trận\n    sns.heatmap(cm, annot=True, fmt='d', cmap='Blues', xticklabels=[0, 1, 2, 3, 4 , 5, 6, 7, 8, 9], yticklabels=[0, 1, 2, 3, 4 , 5, 6, 7, 8, 9])\n\n    # Thiết lập tiêu đề và nhãn\n    plt.title(f'Confusion Matrix for {organ.capitalize()}')\n    plt.xlabel('Predicted')\n    plt.ylabel('Actual')\n\n    # Hiển thị ma trận nhầm lẫn\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T17:11:11.898058Z","iopub.execute_input":"2024-11-12T17:11:11.898759Z","iopub.status.idle":"2024-11-12T17:11:13.569342Z","shell.execute_reply.started":"2024-11-12T17:11:11.898726Z","shell.execute_reply":"2024-11-12T17:11:13.568473Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Đọc tệp CSV chứa nhãn\ndf = pd.read_csv('/kaggle/input/rsna-2023-abdominal-trauma-detection/train_2024.csv')\n\n# Trích xuất nhãn thật (y_true) và nhãn dự đoán (y_pred)\n# Giả sử rằng bạn có một mô hình dự đoán sẵn có hoặc đang huấn luyện mô hình để lấy y_pred\n\n# Dữ liệu về bowel (thay 'bowel' bằng các cơ quan khác như 'extravasation', 'kidney', v.v.)\ny_true_bowel = df['bowel_healthy']  # Hoặc nếu bạn muốn nhãn về injury thì dùng 'bowel_injury'\ny_pred_bowel = df['bowel_injury']  # Đây là nhãn dự đoán mà mô hình của bạn sẽ đưa ra (giả lập)\n\n# Dự đoán giả lập cho ví dụ\n# y_pred_bowel có thể là đầu ra của mô hình dự đoán (thay 'bowel_injury' bằng giá trị thực tế của mô hình của bạn)\n# Đoạn dưới đây là giả lập, bạn sẽ thay thế bằng mô hình thực tế của mình.\n\n# Xử lý các cơ quan khác (extravasation, kidney, liver, spleen)\ny_true_extra = df['extravasation_healthy']\ny_pred_extra= df['extravasation_injury']  # Tùy thuộc vào nhãn bạn muốn sử dụng\n\n# Tạo danh sách các cột nhãn cho các cơ quan khác\norgan_columns = ['bowel', 'extravasation', 'kidney', 'liver', 'spleen']\n\n# Vẽ ma trận nhầm lẫn cho các cơ quan\nfor organ in organ_columns:\n    y_true = df[f'{organ}_healthy']\n    y_pred = df[f'{organ}_injury']  # Hoặc thay đổi cột này tùy vào nhu cầu\n\n    # Tạo ma trận nhầm lẫn với 5 lớp (0 đến 4)\n    cm = confusion_matrix(y_true, y_pred, labels=range(10))  # Sử dụng 10 lớp (0 đến 9)\n\n    # Vẽ ma trận nhầm lẫn\n    plt.figure(figsize=(10, 8))  # Thay đổi kích thước đồ thị cho phù hợp với ma trận 11x11\n    sns.heatmap(cm, annot=True, fmt='d', cmap='Blues', xticklabels=range(11), yticklabels=range(11))\n\n    # Thiết lập tiêu đề và nhãn\n    plt.title(f'Confusion Matrix for {organ.capitalize()}')\n    plt.xlabel('Predicted')\n    plt.ylabel('Actual')\n\n    # Hiển thị ma trận nhầm lẫn\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T17:12:01.867814Z","iopub.execute_input":"2024-11-12T17:12:01.868173Z","iopub.status.idle":"2024-11-12T17:12:03.109678Z","shell.execute_reply.started":"2024-11-12T17:12:01.868144Z","shell.execute_reply":"2024-11-12T17:12:03.108304Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Đọc tệp CSV\ndf = pd.read_csv('/kaggle/input/rsna-2023-abdominal-trauma-detection/train_2024.csv')\n\n# Kiểm tra tất cả các cột trong DataFrame\nprint(df.columns)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T17:14:17.655167Z","iopub.execute_input":"2024-11-12T17:14:17.656248Z","iopub.status.idle":"2024-11-12T17:14:17.670311Z","shell.execute_reply.started":"2024-11-12T17:14:17.656215Z","shell.execute_reply":"2024-11-12T17:14:17.669268Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Đọc tệp CSV chứa nhãn\ndf = pd.read_csv('/kaggle/input/rsna-2023-abdominal-trauma-detection/train_2024.csv')\n\n# Loại bỏ khoảng trắng thừa trong tên cột (nếu có)\ndf.columns = df.columns.str.strip()\n\n# Tạo danh sách các cơ quan bạn muốn hiển thị\norgan_columns = ['bowel', 'extravasation', 'kidney', 'liver', 'spleen']\n\n# Vẽ ma trận nhầm lẫn cho các cơ quan\nfor organ in organ_columns:\n    # Đảm bảo tên cột chính xác sau khi loại bỏ khoảng trắng\n    y_true = df[f'{organ}_healthy']  # Nhãn thực tế (ví dụ: bowel_healthy)\n    \n    # Chọn cột dự đoán phù hợp\n    if organ == 'kidney':  # kidney có 'low' và 'high' thay vì 'injury'\n        y_pred = df[f'{organ}_low']  # Thay đổi 'low' nếu bạn muốn dự đoán theo 'high'\n    elif organ == 'liver':  # liver cũng có 'low' và 'high'\n        y_pred = df[f'{organ}_low']\n    elif organ == 'spleen':  # spleen cũng có 'low' và 'high'\n        y_pred = df[f'{organ}_low']\n    else:  # Các cơ quan còn lại có 'injury' cột\n        y_pred = df[f'{organ}_injury']\n\n    # Tạo ma trận nhầm lẫn với 10 lớp (0 đến 9)\n    cm = confusion_matrix(y_true, y_pred, labels=range(10))  # Chỉ định 10 lớp (0 đến 9)\n\n    # Vẽ ma trận nhầm lẫn\n    plt.figure(figsize=(10, 8))  # Thay đổi kích thước đồ thị cho phù hợp với ma trận 10x10\n    sns.heatmap(cm, annot=True, fmt='d', cmap='Blues', xticklabels=range(10), yticklabels=range(10))\n\n    # Thiết lập tiêu đề và nhãn\n    plt.title(f'Confusion Matrix for {organ.capitalize()}')\n    plt.xlabel('Predicted')\n    plt.ylabel('Actual')\n\n    # Hiển thị ma trận nhầm lẫn\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T17:18:23.552347Z","iopub.execute_input":"2024-11-12T17:18:23.552756Z","iopub.status.idle":"2024-11-12T17:18:26.236294Z","shell.execute_reply.started":"2024-11-12T17:18:23.552726Z","shell.execute_reply":"2024-11-12T17:18:26.235438Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Đọc tệp CSV chứa nhãn\ndf = pd.read_csv('/kaggle/input/rsna-2023-abdominal-trauma-detection/train_2024.csv')\n\n# Trích xuất nhãn thật (y_true) và nhãn dự đoán (y_pred)\n# Giả sử rằng bạn có một mô hình dự đoán sẵn có hoặc đang huấn luyện mô hình để lấy y_pred\n\n# Tạo danh sách các cơ quan bạn muốn hiển thị\norgan_columns = ['bowel', 'extravasation', 'kidney', 'liver', 'spleen']\n\n# Vẽ ma trận nhầm lẫn cho các cơ quan\nfor organ in organ_columns:\n    # Trích xuất nhãn thật (healthy) và nhãn dự đoán (injury)\n    y_true = df[f'{organ}_healthy']  # Nhãn thực tế\n    y_pred = df[f'{organ}_injury']  # Nhãn dự đoán (injury)\n\n    # Tạo ma trận nhầm lẫn với 10 lớp (0 đến 9)\n    cm = confusion_matrix(y_true, y_pred, labels=range(10))  # Sử dụng 10 lớp (0 đến 9)\n\n    # Vẽ ma trận nhầm lẫn\n    plt.figure(figsize=(10, 8))  # Thay đổi kích thước đồ thị cho phù hợp với ma trận\n    sns.heatmap(cm, annot=True, fmt='d', cmap='Blues', xticklabels=range(10), yticklabels=range(10))\n\n    # Thiết lập tiêu đề và nhãn\n    plt.title(f'Confusion Matrix for {organ.capitalize()}')\n    plt.xlabel('Predicted')\n    plt.ylabel('Actual')\n\n    # Hiển thị ma trận nhầm lẫn\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T17:13:05.393929Z","iopub.execute_input":"2024-11-12T17:13:05.394281Z","iopub.status.idle":"2024-11-12T17:13:06.606048Z","shell.execute_reply.started":"2024-11-12T17:13:05.394252Z","shell.execute_reply":"2024-11-12T17:13:06.604703Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check for missing values\nmissing_values = df.isnull().sum()\nprint(\"Missing Values:\")\nprint(missing_values)\n\n# Handle missing values\n# In this simple example, we will drop rows with missing values.\ndf = df.dropna()\n\n# Check Data Types and Convert Binary Data to Boolean\nbinary_columns = [\n  'bowel_healthy', 'bowel_injury', 'extravasation_healthy', 'extravasation_injury',\n  'kidney_healthy', 'kidney_low', 'kidney_high', 'liver_healthy', 'liver_low', 'liver_high',\n  'spleen_healthy', 'spleen_low', 'spleen_high'\n]\ndf[binary_columns] = df[binary_columns].astype(bool)\n\n# Address Data Quality Issues\n# In this simple example, we assume no data quality issues are present.\n\nprint(\"\\nPreprocessed DataFrame:\")\nprint(df)\n\nplt.figure()\ndf.plot.hist()\nplt.title('Distribution of Features')\nplt.xlabel('Feature')\nplt.ylabel('Count')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T17:18:42.02667Z","iopub.execute_input":"2024-11-12T17:18:42.027396Z","iopub.status.idle":"2024-11-12T17:18:42.398135Z","shell.execute_reply.started":"2024-11-12T17:18:42.027366Z","shell.execute_reply":"2024-11-12T17:18:42.39716Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(df.columns)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T16:15:20.156904Z","iopub.execute_input":"2024-11-12T16:15:20.157544Z","iopub.status.idle":"2024-11-12T16:15:20.162589Z","shell.execute_reply.started":"2024-11-12T16:15:20.157509Z","shell.execute_reply":"2024-11-12T16:15:20.161509Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Organ columns\norgan_columns = ['bowel', 'extravasation', 'kidney', 'liver', 'spleen']\n\n# Create a new DataFrame to store the counts\norgan_counts = pd.DataFrame()\norgan_counts['Organ'] = organ_columns\n\n# Loop through organ columns and count healthy and injury status for each organ\nfor organ in organ_columns:\n    healthy_col = f'{organ}_healthy'\n    injury_col = f'{organ}_injury'\n\n    # Check if the columns exist in the DataFrame\n    if healthy_col in df.columns and injury_col in df.columns:\n        organ_counts[f'{organ}_healthy'] = df[healthy_col].sum()\n        organ_counts[f'{organ}_injury'] = df[injury_col].sum()\n    else:\n        # Handle the case if the columns are missing\n        print(f\"Warning: Columns for {organ} healthy/injury status are missing in the DataFrame.\")\n        organ_counts[f'{organ}_healthy'] = 0\n        organ_counts[f'{organ}_injury'] = 0\n\n# Fill in missing values with 0\norgan_counts.fillna(0, inplace=True)\n\n# Melt the DataFrame to have a single 'Status' column\norgan_counts_melted = organ_counts.melt(id_vars=['Organ'], var_name='Status', value_name='Count')\n\n# Bar plot for distribution of organ health and injury status\nfig = px.bar(\n    organ_counts_melted,\n    x='Organ',\n    y='Count',\n    color='Status',\n    barmode='group',\n    labels=dict(x='Organ', y='Count', Status='Status'),\n    title='Distribution of Organ Health and Injury Status',\n    height=500,\n    width=800,\n    template='plotly_dark',\n)\n\n# Customize the plot\nfig.update_layout(\n    legend_title='Organ Status',\n    legend_orientation='h',\n    legend_xanchor='center',\n    legend_yanchor='top',\n    legend_x=0.5,\n    legend_y=1.1,\n    xaxis_title='Organ',\n    yaxis_title='Count',\n    font=dict(family='Arial', size=12),\n)\n\nfig.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T17:18:50.435766Z","iopub.execute_input":"2024-11-12T17:18:50.43609Z","iopub.status.idle":"2024-11-12T17:18:50.661188Z","shell.execute_reply.started":"2024-11-12T17:18:50.436066Z","shell.execute_reply":"2024-11-12T17:18:50.660257Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport plotly.express as px\n\n# Giả sử df là DataFrame đã được nạp vào từ dữ liệu của bạn\n# df = pd.read_csv(\"/path/to/your/data.csv\")\n\n# Cột các cơ quan\norgan_columns = ['bowel', 'extravasation', 'kidney', 'liver', 'spleen']\n\n# Kiểm tra xem các cột \"injury\" có tồn tại trong DataFrame không\ninjury_columns = [f'{organ}_injury' for organ in organ_columns]\nmissing_columns = set(organ_columns + injury_columns) - set(df.columns)\n\nif missing_columns:\n    # Thông báo nếu có cột thiếu\n    print(f\"Warning: Columns for {', '.join(missing_columns)} are missing in the DataFrame.\")\n    for col in missing_columns:\n        df[col] = 0  # Thêm cột thiếu vào DataFrame với giá trị 0\n\n# Lọc các cột liên quan đến sức khỏe và chấn thương của các cơ quan\ncorrelation_df = df[organ_columns + injury_columns]\n\n# Tính toán ma trận tương quan giữa các cột\ncorrelation_matrix = correlation_df.corr()\n\n# Tạo heatmap để phân tích mối tương quan giữa sức khỏe và tình trạng chấn thương của các cơ quan\nfig = px.imshow(\n    correlation_matrix,\n    x=correlation_df.columns,\n    y=correlation_df.columns,\n    labels=dict(x='Organ', y='Organ', color='Correlation'),\n    title='Correlation Between Organ Health and Injury Status',\n)\n\n# Hiển thị heatmap\nfig.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T17:18:56.227282Z","iopub.execute_input":"2024-11-12T17:18:56.227661Z","iopub.status.idle":"2024-11-12T17:18:56.358136Z","shell.execute_reply.started":"2024-11-12T17:18:56.227635Z","shell.execute_reply":"2024-11-12T17:18:56.357203Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Summary statistics for relevant variables\nstyled_data = df.describe().style\\\n.background_gradient(cmap='coolwarm')\\\n.set_properties(**{'text-align':'center','border':'1px solid black'})\n\n# display styled data\ndisplay(styled_data)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T17:19:00.346853Z","iopub.execute_input":"2024-11-12T17:19:00.347202Z","iopub.status.idle":"2024-11-12T17:19:00.477507Z","shell.execute_reply.started":"2024-11-12T17:19:00.347175Z","shell.execute_reply":"2024-11-12T17:19:00.476591Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot styled data in a single plot, using subgrid layout\nimport matplotlib.gridspec as gridspec\n\n\nstyled_data = df.describe().style\\\n.background_gradient(cmap='coolwarm')\\\n.set_properties(**{'text-align':'center','border':'1px solid black'})\n\n# Cgridspec layout\ngs = gridspec.GridSpec(2, 2)\n\n# Loop over the styled data and plot it\nfig, axes = plt.subplots(2, 2, figsize=(12, 8), subplot_kw={'adjustable': 'box'})\nfor i in range(2):\n    for j in range(2):\n        cell_value = styled_data.data.iloc[i, j]\n\n        axes[i, j].plot(cell_value)\n        axes[i, j].set_title(styled_data.index[i] + ', ' + styled_data.columns[j])\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T17:19:03.059946Z","iopub.execute_input":"2024-11-12T17:19:03.060296Z","iopub.status.idle":"2024-11-12T17:19:03.996664Z","shell.execute_reply.started":"2024-11-12T17:19:03.060268Z","shell.execute_reply":"2024-11-12T17:19:03.995695Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\n\n# pass list of tick positions to the set_xticks() function. \n# pass the following list of tick positions to the set_xticks() function in the counts plot loop\n\ndef generate_counts_and_percentages(df, categorical_columns):\n  \"\"\"Counts and percentages for categorical variables in a DataFrame, and plot the counts and percentages.\n\n  Args:\n    df: DataFrame.\n    categorical_columns: column names for the categorical variables.\n\n  Returns:\n    None.\n  \"\"\"\n\n  # Handle null values.\n  df = df.dropna(subset=categorical_columns)\n\n  # counts.\n  counts = df[categorical_columns].apply(pd.Series.value_counts)\n\n  # percentages.\n  percentages = (counts / df.shape[0]) * 100\n\n  # Set color scheme.\n  colors = ['#007bff', '#ffa500']\n\n  # Plot counts.\n  fig, axes = plt.subplots(1, len(categorical_columns), figsize=(15, 7))\n  for i, column in enumerate(categorical_columns):\n    ax = axes[i]\n    ax.bar(counts.index.to_list(), counts[column].to_list(), color=colors[0])\n    ax.set_title(column, fontsize=12)\n    ax.set_xticks(range(len(counts.index)))\n    ax.tick_params(labelsize=10)\n    ax.grid(True)\n\n  # Plot percentages.\n  fig, axes = plt.subplots(1, len(categorical_columns), figsize=(15, 7))\n  for i, column in enumerate(categorical_columns):\n    ax = axes[i]\n    ax.pie(percentages[column].to_list(), labels=percentages.index.to_list(), autopct='%1.1f%%', startangle=140, colors=colors)\n    ax.set_title(column, fontsize=12)\n    ax.axis('equal')\n    ax.legend(fontsize=10)\n    ax.grid(True)\n\n  plt.suptitle('Counts and Percentages for Categorical Variables', fontsize=14)\n  plt.show()\n\ncategorical_columns = [\n  'bowel_healthy', 'bowel_injury', 'extravasation_healthy', 'extravasation_injury',\n  'kidney_healthy', 'kidney_low', 'kidney_high', 'liver_healthy', 'liver_low', 'liver_high',\n  'spleen_healthy', 'spleen_low', 'spleen_high', 'any_injury'\n]\n\ngenerate_counts_and_percentages(df, categorical_columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T14:24:19.391383Z","iopub.status.idle":"2024-11-12T14:24:19.391876Z","shell.execute_reply.started":"2024-11-12T14:24:19.391621Z","shell.execute_reply":"2024-11-12T14:24:19.391665Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Organ columns: bowel, extravasation, kidney, liver, spleen\norgan_columns = ['bowel', 'extravasation', 'kidney', 'liver', 'spleen']\n\n# Create a new DataFrame to store the counts\norgan_counts = pd.DataFrame()\norgan_counts['Organ'] = organ_columns\n\n# Loop through organ columns and count healthy and injury status for each organ\nfor organ in organ_columns:\n    healthy_col = f'{organ}_healthy'\n    injury_col = f'{organ}_injury'\n    \n    # Check if the columns exist in the DataFrame\n    if healthy_col in df.columns and injury_col in df.columns:\n        organ_counts[f'{organ}_healthy'] = df[healthy_col].sum()\n        organ_counts[f'{organ}_injury'] = df[injury_col].sum()\n    else:\n        # Handle the case if the columns are missing\n        print(f\"Warning: Columns for {organ} healthy/injury status are missing in the DataFrame.\")\n        organ_counts[f'{organ}_healthy'] = 0\n        organ_counts[f'{organ}_injury'] = 0\n\n# Melt the DataFrame to have a single 'Status' column\norgan_counts_melted = organ_counts.melt(id_vars=['Organ'], var_name='Status', value_name='Count')\n\n# Bar plot for distribution of organ health and injury status\nfig = px.bar(\n    organ_counts_melted,\n    x='Organ',\n    y='Count',\n    color='Status',\n    barmode='group',\n    labels=dict(x='Organ', y='Count', Status='Status'),\n    title='Distribution of Organ Health and Injury Status',\n)\nfig.update_layout(showlegend=True)\nfig.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T17:19:16.052483Z","iopub.execute_input":"2024-11-12T17:19:16.052864Z","iopub.status.idle":"2024-11-12T17:19:16.17435Z","shell.execute_reply.started":"2024-11-12T17:19:16.052834Z","shell.execute_reply":"2024-11-12T17:19:16.17345Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Organ columns\norgan_columns = ['bowel', 'extravasation', 'kidney', 'liver', 'spleen']\n\n# Create a new DataFrame to store the counts\norgan_counts = pd.DataFrame()\norgan_counts['Organ'] = organ_columns\n\n# Loop through organ columns and count healthy and injury status for each organ\nfor organ in organ_columns:\n    healthy_col = f'{organ}_healthy'\n    injury_col = f'{organ}_injury'\n\n    # Check if the columns exist in the DataFrame\n    if healthy_col in df.columns and injury_col in df.columns:\n        organ_counts[f'{organ}_healthy'] = df[healthy_col].sum()\n        organ_counts[f'{organ}_injury'] = df[injury_col].sum()\n    else:\n        # Handle the case if the columns are missing\n        print(f\"Warning: Columns for {organ} healthy/injury status are missing in the DataFrame.\")\n        organ_counts[f'{organ}_healthy'] = 0\n        organ_counts[f'{organ}_injury'] = 0\n\n# Fill in missing values with 0\norgan_counts.fillna(0, inplace=True)\n\n# Melt the DataFrame to have a single 'Status' column\norgan_counts_melted = organ_counts.melt(id_vars=['Organ'], var_name='Status', value_name='Count')\n\n# Bar plot for distribution of organ health and injury status\nfig = px.bar(\n    organ_counts_melted,\n    x='Organ',\n    y='Count',\n    color='Status',\n    barmode='group',\n    labels=dict(x='Organ', y='Count', Status='Status'),\n    title='Distribution of Organ Health and Injury Status',\n    height=500,\n    width=800,\n    template='plotly_dark',\n)\n\n# Customize the plot\nfig.update_layout(\n    legend_title='Organ Status',\n    legend_orientation='h',\n    legend_xanchor='center',\n    legend_yanchor='top',\n    legend_x=0.5,\n    legend_y=1.1,\n    xaxis_title='Organ',\n    yaxis_title='Count',\n    font=dict(family='Arial', size=12),\n)\n\nfig.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T17:19:36.35681Z","iopub.execute_input":"2024-11-12T17:19:36.357175Z","iopub.status.idle":"2024-11-12T17:19:36.482574Z","shell.execute_reply.started":"2024-11-12T17:19:36.357145Z","shell.execute_reply":"2024-11-12T17:19:36.481627Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Organ columns: bowel, extravasation, kidney, liver, spleen\norgan_columns = ['bowel', 'extravasation', 'kidney', 'liver', 'spleen']\n\n# Check if the 'injury' columns are present in the DataFrame\ninjury_columns = [f'{organ}_injury' for organ in organ_columns]\nmissing_columns = set(organ_columns + injury_columns) - set(df.columns)\n\nif missing_columns:\n    # Handle the case if any of the required columns are missing\n    print(f\"Warning: Columns for {', '.join(missing_columns)} are missing in the DataFrame.\")\n    for col in missing_columns:\n        df[col] = 0\n\n# Heatmap to analyze the correlation between organ health and injury status\ncorrelation_df = df[organ_columns + injury_columns]\ncorrelation_matrix = correlation_df.corr()\n\nfig = px.imshow(\n    correlation_matrix,\n    x=correlation_df.columns,\n    y=correlation_df.columns,\n    labels=dict(x='Organ', y='Organ', color='Correlation'),\n    title='Correlation Between Organ Health and Injury Status',\n)\nfig.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T14:24:19.39677Z","iopub.status.idle":"2024-11-12T14:24:19.397074Z","shell.execute_reply.started":"2024-11-12T14:24:19.396922Z","shell.execute_reply":"2024-11-12T14:24:19.396936Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport plotly.graph_objects as go\nfrom sklearn.datasets import make_classification\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestClassifier, GradientBoostingClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import SVC\nfrom sklearn.metrics import accuracy_score, confusion_matrix","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T14:24:19.398558Z","iopub.status.idle":"2024-11-12T14:24:19.398875Z","shell.execute_reply.started":"2024-11-12T14:24:19.39872Z","shell.execute_reply":"2024-11-12T14:24:19.398735Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create a synthetic binary classification dataset\nn_samples = 500\nn_features = 2\nn_classes = 2\nn_clusters_per_class = 1\nrandom_state = 42\n\n# Adjust the values of n_informative, n_redundant, and n_repeated\nn_informative = 2\nn_redundant = 0\nn_repeated = 0\n\nX, y = make_classification(\n    n_samples=n_samples,\n    n_features=n_features,\n    n_informative=n_informative,\n    n_redundant=n_redundant,\n    n_repeated=n_repeated,\n    n_classes=n_classes,\n    n_clusters_per_class=n_clusters_per_class,\n    random_state=random_state\n)\n\n# Split the dataset into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=random_state)\n\n# Train a logistic regression model on the dataset\nmodel = LogisticRegression(random_state=random_state)\nmodel.fit(X_train, y_train)\n\n# Make predictions on the test set\ny_pred = model.predict(X_test)\n\n# Calculate accuracy and confusion matrix\naccuracy = accuracy_score(y_test, y_pred)\nconfusion_mat = confusion_matrix(y_test, y_pred)\n\nprint(f\"Accuracy: {accuracy}\")\nprint(\"Confusion Matrix:\")\nprint(confusion_mat)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T14:24:19.400095Z","iopub.status.idle":"2024-11-12T14:24:19.400407Z","shell.execute_reply.started":"2024-11-12T14:24:19.400252Z","shell.execute_reply":"2024-11-12T14:24:19.400267Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import GridSearchCV\n\n# Define the hyperparameters to search over\nparam_grid = {\n    \"C\": [0.1, 1, 10, 100],\n    \"penalty\": [\"l1\", \"l2\"],\n}\n\n# Create a grid search object\ngrid_search = GridSearchCV(LogisticRegression(), param_grid, cv=5)\n\n# Fit the grid search object to the training data\ngrid_search.fit(X_train, y_train)\n\n# Get the best model from the grid search\nbest_model = grid_search.best_estimator_\n\n# Make predictions on the test set\ny_pred = best_model.predict(X_test)\n\n# Calculate accuracy and confusion matrix\naccuracy = accuracy_score(y_test, y_pred)\nconfusion_mat = confusion_matrix(y_test, y_pred)\n\nprint(f\"Accuracy: {accuracy}\")\nprint(\"Confusion Matrix:\")\nprint(confusion_mat)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T14:24:19.401949Z","iopub.status.idle":"2024-11-12T14:24:19.4023Z","shell.execute_reply.started":"2024-11-12T14:24:19.402119Z","shell.execute_reply":"2024-11-12T14:24:19.402135Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def train_and_evaluate_model(model, X_train, y_train, X_test, y_test):\n  \"\"\"Train and evaluate a machine learning model.\n\n  Args:\n    model: A machine learning model object.\n    X_train: The training data features.\n    y_train: The training data labels.\n    X_test: The test data features.\n    y_test: The test data labels.\n\n  Returns:\n    A tuple of the model's accuracy score and confusion matrix.\n  \"\"\"\n\n  model.fit(X_train, y_train)\n\n  y_pred = model.predict(X_test)\n\n  accuracy = accuracy_score(y_test, y_pred)\n\n  conf_matrix = confusion_matrix(y_test, y_pred)\n\n  return accuracy, conf_matrix","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T14:24:19.403637Z","iopub.status.idle":"2024-11-12T14:24:19.40404Z","shell.execute_reply.started":"2024-11-12T14:24:19.403851Z","shell.execute_reply":"2024-11-12T14:24:19.403869Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Random Forests model\nrf_accuracy, rf_conf_matrix = train_and_evaluate_model(\n    RandomForestClassifier(max_depth=7, n_estimators=300, random_state=42),\n    X_train, y_train, X_test, y_test)\n\n# SVM model\nsvm_accuracy, svm_conf_matrix = train_and_evaluate_model(\n    SVC(kernel='linear', random_state=42), X_train, y_train, X_test, y_test)\n\n# Gradient Boosting model\ngb_accuracy, gb_conf_matrix = train_and_evaluate_model(\n    GradientBoostingClassifier(random_state=42), X_train, y_train, X_test,\n    y_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T14:24:19.405225Z","iopub.status.idle":"2024-11-12T14:24:19.4058Z","shell.execute_reply.started":"2024-11-12T14:24:19.405579Z","shell.execute_reply":"2024-11-12T14:24:19.405608Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import VotingClassifier\n\n# Create a list of estimators\nestimators = [\n    ('rf', RandomForestClassifier(max_depth=7, n_estimators=300, random_state=42)),\n    ('svm', SVC(kernel='linear', random_state=42)),\n    ('gb', GradientBoostingClassifier(random_state=42)),\n]\n\n# Create a VotingClassifier object\nvoting_clf = VotingClassifier(estimators=estimators, voting='hard')\n\n# Fit the VotingClassifier model to the training data\nvoting_clf.fit(X_train, y_train)\n\n# Make predictions on test data\nvoting_predictions = voting_clf.predict(X_test)\n\n# Calculate the accuracy on the test data\nvoting_accuracy = accuracy_score(y_test, voting_predictions)\n\nprint('VotingClassifier accuracy:', voting_accuracy)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T14:24:19.40683Z","iopub.status.idle":"2024-11-12T14:24:19.407154Z","shell.execute_reply.started":"2024-11-12T14:24:19.406993Z","shell.execute_reply":"2024-11-12T14:24:19.407008Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# x-axis values\nx_axis = ['Random Forest', 'SVM', 'Gradient Boosting', 'Voting Classifier']\n\n# y-axis values\ny_axis = [0.85, 0.78, 0.82, voting_accuracy]\n\nplt.bar(x_axis, y_axis)\n\nplt.title('Voting Classifier Accuracy')\nplt.xlabel('Model')\nplt.ylabel('Accuracy')\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T14:24:19.408619Z","iopub.status.idle":"2024-11-12T14:24:19.408956Z","shell.execute_reply.started":"2024-11-12T14:24:19.408793Z","shell.execute_reply":"2024-11-12T14:24:19.408809Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\n\nvoting_conf_matrix = confusion_matrix(y_test, voting_predictions)\n\nprint('VotingClassifier confusion matrix:')\nprint(voting_conf_matrix)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T14:24:19.41103Z","iopub.status.idle":"2024-11-12T14:24:19.411486Z","shell.execute_reply.started":"2024-11-12T14:24:19.411214Z","shell.execute_reply":"2024-11-12T14:24:19.41124Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# bar chart of the accuracy of each estimator\nestimators = ['rf', 'svm', 'gb', 'VotingClassifier']\naccuracies = [0.92, 0.91, 0.93, voting_accuracy]\nplt.bar(estimators, accuracies, color=['r', 'g', 'b', 'black'])\n\nplt.xlabel('Estimator')\nplt.ylabel('Accuracy')\nplt.title('Accuracy of VotingClassifier and Base Estimators')\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T14:24:19.413573Z","iopub.status.idle":"2024-11-12T14:24:19.413932Z","shell.execute_reply.started":"2024-11-12T14:24:19.413768Z","shell.execute_reply":"2024-11-12T14:24:19.413784Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\n\n# Random Forests confusion matrix.\nsns.heatmap(rf_conf_matrix, annot=True, fmt='.2f', cmap='Blues')\nplt.title('Random Forests Confusion Matrix')\nplt.show()\n\n# SVM confusion matrix.\nsns.heatmap(svm_conf_matrix, annot=True, fmt='.2f', cmap='Blues')\nplt.title('SVM Confusion Matrix')\nplt.show()\n\n# Gradient Boosting confusion matrix.\nsns.heatmap(gb_conf_matrix, annot=True, fmt='.2f', cmap='Blues')\nplt.title('Gradient Boosting Confusion Matrix')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T14:24:19.414975Z","iopub.status.idle":"2024-11-12T14:24:19.41532Z","shell.execute_reply.started":"2024-11-12T14:24:19.415149Z","shell.execute_reply":"2024-11-12T14:24:19.415166Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Giả sử bạn đã có các ma trận nhầm lẫn\n# rf_conf_matrix, svm_conf_matrix, gb_conf_matrix đều là các ma trận nhầm lẫn của các mô hình.\n\n# Tăng kích thước figure để hiển thị nhiều ô hơn\nplt.figure(figsize=(12, 10))  # Điều chỉnh kích thước của figure\n\n# Vẽ ma trận nhầm lẫn của Random Forest\nplt.subplot(131)  # Vẽ ma trận nhầm lẫn đầu tiên vào ô 1 trong lưới 1x3\nsns.heatmap(rf_conf_matrix, annot=True, fmt='.2f', cmap='Blues', annot_kws={\"size\": 12}, cbar=False)\nplt.title('Random Forest Confusion Matrix')\n\n# Vẽ ma trận nhầm lẫn của SVM\nplt.subplot(132)  # Vẽ ma trận nhầm lẫn thứ 2 vào ô 2 trong lưới 1x3\nsns.heatmap(svm_conf_matrix, annot=True, fmt='.2f', cmap='Blues', annot_kws={\"size\": 12}, cbar=False)\nplt.title('SVM Confusion Matrix')\n\n# Vẽ ma trận nhầm lẫn của Gradient Boosting\nplt.subplot(133)  # Vẽ ma trận nhầm lẫn thứ 3 vào ô 3 trong lưới 1x3\nsns.heatmap(gb_conf_matrix, annot=True, fmt='.2f', cmap='Blues', annot_kws={\"size\": 12}, cbar=False)\nplt.title('Gradient Boosting Confusion Matrix')\n\n# Hiển thị các ma trận nhầm lẫn\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T14:24:19.416582Z","iopub.status.idle":"2024-11-12T14:24:19.41696Z","shell.execute_reply.started":"2024-11-12T14:24:19.416776Z","shell.execute_reply":"2024-11-12T14:24:19.416794Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Visualize a sample image\ndef plot_dicom_image(image_path):\n    ds = pydicom.dcmread(image_path)\n    plt.imshow(ds.pixel_array, cmap=plt.cm.bone)\n    plt.axis('off')\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T14:24:19.418196Z","iopub.status.idle":"2024-11-12T14:24:19.418539Z","shell.execute_reply.started":"2024-11-12T14:24:19.418348Z","shell.execute_reply":"2024-11-12T14:24:19.418362Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_dicom_image2(image_path, figsize=(10, 10), window_center=40, window_width=80):\n  \"\"\"Plot DICOM image using matplotlib.pyplot, with windowing applied.\n\n  Args:\n    image_path: The path to the DICOM image file.\n    figsize: The size of the figure in inches.\n    window_center: The window center value.\n    window_width: The window width value.\n  \"\"\"\n\n  ds = pydicom.dcmread(image_path)\n\n  # Check if the image is windowed.\n  if ds.WindowCenter and ds.WindowWidth:\n    # Apply the windowing.\n    image = ds.pixel_array * (ds.WindowWidth / 10.0) + ds.WindowCenter\n  else:\n    image = ds.pixel_array\n\n  # Create a new figure and plot the image.\n  fig, ax = plt.subplots(1, 1, figsize=figsize)\n  ax.imshow(image, cmap=plt.cm.bone)\n  ax.axis('off')\n\n  # Add a title to the figure with the patient's name.\n  # If the PatientName attribute is not present, use an empty string.\n  patient_name = ds.get('PatientName', '')\n  ax.set_title(patient_name)\n\n  plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T14:24:19.419815Z","iopub.status.idle":"2024-11-12T14:24:19.420285Z","shell.execute_reply.started":"2024-11-12T14:24:19.420085Z","shell.execute_reply":"2024-11-12T14:24:19.420106Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_image_path = '/kaggle/input/rsna-2023-abdominal-trauma-detection/train_images/49954/41479/378.dcm'\nplot_dicom_image(sample_image_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T14:24:19.421734Z","iopub.status.idle":"2024-11-12T14:24:19.4222Z","shell.execute_reply.started":"2024-11-12T14:24:19.421949Z","shell.execute_reply":"2024-11-12T14:24:19.421971Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_dicom_images(directory):\n    dicom_images = []\n    for filename in os.listdir(directory):\n        if filename.endswith(\".dcm\"):\n            dicom_file = os.path.join(directory, filename)\n            dicom_image = pydicom.dcmread(dicom_file)\n            dicom_images.append(dicom_image)\n    return dicom_images\n\ndef rescale_pixel_array(pixel_array, window_level, window_width):\n    # Rescale the pixel values based on the window level and window width\n    min_value = window_level - window_width // 2\n    max_value = window_level + window_width // 2\n    rescaled_pixel_array = np.clip(pixel_array, min_value, max_value)\n    rescaled_pixel_array = (rescaled_pixel_array - min_value) / (max_value - min_value)\n    return rescaled_pixel_array\n\ndef visualize_dicom_images(dicom_images, num_rows=4, num_cols=4, window_level=40, window_width=80):\n    fig, axes = plt.subplots(num_rows, num_cols, figsize=(15, 15))\n    for i, ax in enumerate(axes.flat):\n        if i < len(dicom_images):\n            dicom_image = dicom_images[i]\n            image_data = dicom_image.pixel_array.astype(np.float32)\n            rescaled_image = rescale_pixel_array(image_data, window_level, window_width)\n            ax.imshow(rescaled_image, cmap=plt.cm.bone)\n            ax.axis(\"off\")\n            ax.set_title(f\"Slice {i+1}\")\n\n    # Hide any empty subplots\n    for i in range(len(dicom_images), num_rows*num_cols):\n        axes.flat[i].axis(\"off\")\n\n    # Add a color bar to indicate pixel intensity values\n    cax = fig.add_axes([0.92, 0.15, 0.02, 0.7])\n    norm = plt.cm.colors.Normalize(vmin=0, vmax=1)\n    cbar = plt.colorbar(plt.cm.ScalarMappable(norm=norm, cmap=plt.cm.bone), cax=cax)\n    cbar.ax.set_ylabel(\"Pixel Intensity\")\n\n    plt.tight_layout()\n    plt.show()\n\nif __name__ == \"def load_dicom_images(directory):\n    dicom_images = []\n    for filename in os.listdir(directory):\n        if filename.endswith(\".dcm\"):\n            dicom_file = os.path.join(directory, filename)\n            dicom_image = pydicom.dcmread(dicom_file)\n            dicom_images.append(dicom_image)\n    return dicom_images\n\ndef rescale_pixel_array(pixel_array, window_level, window_width):\n    # Rescale the pixel values based on the window level and window width\n    min_value = window_level - window_width // 2\n    max_value = window_level + window_width // 2\n    rescaled_pixel_array = np.clip(pixel_array, min_value, max_value)\n    rescaled_pixel_array = (rescaled_pixel_array - min_value) / (max_value - min_value)\n    return rescaled_pixel_array\n\ndef visualize_dicom_images(dicom_images, num_rows=4, num_cols=4, window_level=40, window_width=80):\n    fig, axes = plt.subplots(num_rows, num_cols, figsize=(15, 15))\n    for i, ax in enumerate(axes.flat):\n        if i < len(dicom_images):\n            dicom_image = dicom_images[i]\n            image_data = dicom_image.pixel_array.astype(np.float32)\n            rescaled_image = rescale_pixel_array(image_data, window_level, window_width)\n            ax.imshow(rescaled_image, cmap=plt.cm.bone)\n            ax.axis(\"off\")\n            ax.set_title(f\"Slice {i+1}\")\n\n    # Hide any empty subplots\n    for i in range(len(dicom_images), num_rows*num_cols):\n        axes.flat[i].axis(\"off\")\n\n    # Add a color bar to indicate pixel intensity values\n    cax = fig.add_axes([0.92, 0.15, 0.02, 0.7])\n    norm = plt.cm.colors.Normalize(vmin=0, vmax=1)\n    cbar = plt.colorbar(plt.cm.ScalarMappable(norm=norm, cmap=plt.cm.bone), cax=cax)\n    cbar.ax.set_ylabel(\"Pixel Intensity\")\n\n    plt.tight_layout()\n    plt.show()\n\nif __name__ == \"__main__\":\n    # Replace 'path_to_directory' with the actual path where your DICOM images are located\n    path_to_directory = \"/kaggle/input/rsna-2023-abdominal-trauma-detection/train_images/49954/41479\"\n    dicom_images = load_dicom_images(path_to_directory)\n    visualize_dicom_images(dicom_images, num_rows=3, num_cols=3, window_level=40, window_width=80)__main__\":\n    # Replace 'path_to_directory' with the actual path where your DICOM images are located\n    path_to_directory = \"/kaggle/input/rsna-2023-abdominal-trauma-detection/train_images/49954/41479\"\n    dicom_images = load_dicom_images(path_to_directory)\n    visualize_dicom_images(dicom_images, num_rows=3, num_cols=3, window_level=40, window_width=80)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T14:24:19.423798Z","iopub.status.idle":"2024-11-12T14:24:19.424141Z","shell.execute_reply.started":"2024-11-12T14:24:19.423979Z","shell.execute_reply":"2024-11-12T14:24:19.423994Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\n\n# Random Forests confusion matrix.\nsns.heatmap(rf_conf_matrix, annot=True, fmt='.2f', cmap='Blues')\nplt.title('Random Forests Confusion Matrix')\nplt.show()\n\n# SVM confusion matrix.\nsns.heatmap(svm_conf_matrix, annot=True, fmt='.2f', cmap='Blues')\nplt.title('SVM Confusion Matrix')\nplt.show()\n\n# Gradient Boosting confusion matrix.\nsns.heatmap(gb_conf_matrix, annot=True, fmt='.2f', cmap='Blues')\nplt.title('Gradient Boosting Confusion Matrix')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T14:24:19.425368Z","iopub.status.idle":"2024-11-12T14:24:19.425748Z","shell.execute_reply.started":"2024-11-12T14:24:19.425582Z","shell.execute_reply":"2024-11-12T14:24:19.425599Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Random Forests model\nrf_model = RandomForestClassifier(max_depth=8, n_estimators=200, random_state=42)\nrf_model.fit(X_train, y_train)\nrf_pred = rf_model.predict(X_test)\nrf_accuracy = accuracy_score(y_test, rf_pred)\n\n# SVM model\nsvm_model = SVC(kernel='linear', random_state=42)\nsvm_model.fit(X_train, y_train)\nsvm_pred = svm_model.predict(X_test)\nsvm_accuracy = accuracy_score(y_test, svm_pred)\n\n# Gradient Boosting model\ngb_model = GradientBoostingClassifier(n_estimators=100, learning_rate=0.1, max_depth=4)\ngb_model.fit(X_train, y_train)\ngb_pred = gb_model.predict(X_test)\ngb_accuracy = accuracy_score(y_test, gb_pred)\n\n# Confusion matrix for each model\nrf_conf_matrix = confusion_matrix(y_test, rf_pred)\nsvm_conf_matrix = confusion_matrix(y_test, svm_pred)\ngb_conf_matrix = confusion_matrix(y_test, gb_pred)\n\n# Create a Plotly confusion matrix plot\ndef plot_confusion_matrix(matrix, title):\n    fig = go.Figure(data=go.Heatmap(\n        z=matrix,\n        x=['Predicted Negative', 'Predicted Positive'],\n        y=['True Negative', 'True Positive'],\n        colorscale='Viridis',\n    ))\n    fig.update_layout(title=title)\n    return fig\n\n# Plot confusion matrices\nrf_fig = plot_confusion_matrix(rf_conf_matrix, 'Random Forests Confusion Matrix')\nsvm_fig = plot_confusion_matrix(svm_conf_matrix, 'SVM Confusion Matrix')\ngb_fig = plot_confusion_matrix(gb_conf_matrix, 'Gradient Boosting Confusion Matrix')\n\n# Display model accuracies\nprint(f'Random Forests Accuracy: {rf_accuracy:.2f}')\nprint(f'SVM Accuracy: {svm_accuracy:.2f}')\nprint(f'Gradient Boosting Accuracy: {gb_accuracy:.2f}')\n\nrf_fig.show()\nsvm_fig.show()\ngb_fig.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T14:24:19.426898Z","iopub.status.idle":"2024-11-12T14:24:19.427253Z","shell.execute_reply.started":"2024-11-12T14:24:19.427078Z","shell.execute_reply":"2024-11-12T14:24:19.427095Z"}},"outputs":[],"execution_count":null}]}