{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":52254,"databundleVersionId":9674523,"sourceType":"competition"},{"sourceId":6211844,"sourceType":"datasetVersion","datasetId":3567114}],"dockerImageVersionId":30554,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#import necessary libs\nimport os\n\nimport keras_cv\nimport keras_core as keras\nfrom keras_core import layers\n\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom matplotlib import pyplot as plt\nfrom sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2024-11-12T18:08:34.735241Z","iopub.execute_input":"2024-11-12T18:08:34.735619Z","iopub.status.idle":"2024-11-12T18:08:34.741042Z","shell.execute_reply.started":"2024-11-12T18:08:34.735589Z","shell.execute_reply":"2024-11-12T18:08:34.740026Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Configuration","metadata":{}},{"cell_type":"markdown","source":"refer this for EDA - https://www.kaggle.com/code/aritrag/eda-train-csv","metadata":{}},{"cell_type":"code","source":"class Config:\n    SEED = 69\n    IMAGE_SIZE = [256,256]\n    BATCH_SIZE=8\n    EPOCHS = 2\n    TARGET_COLS = [\n        \"bowel_injury\", \"extravasation_injury\",\n        \"kidney_healthy\", \"kidney_low\", \"kidney_high\",\n        \"liver_healthy\", \"liver_low\", \"liver_high\",\n        \"spleen_healthy\", \"spleen_low\", \"spleen_high\"\n    ]\n    AUTOTUNE = tf.data.AUTOTUNE\n    \nconfig = Config()","metadata":{"execution":{"iopub.status.busy":"2024-11-12T18:08:34.76495Z","iopub.execute_input":"2024-11-12T18:08:34.765783Z","iopub.status.idle":"2024-11-12T18:08:34.771118Z","shell.execute_reply.started":"2024-11-12T18:08:34.765751Z","shell.execute_reply":"2024-11-12T18:08:34.770101Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Reproducibility\n\nWe would want this notebook to have reproducible results. Here we set the seed for all the random algorithms so that we can reproduce the experiments each time exactly the same way.\n\n","metadata":{}},{"cell_type":"code","source":"keras.utils.set_random_seed(seed=config.SEED)","metadata":{"execution":{"iopub.status.busy":"2024-11-12T18:08:34.773471Z","iopub.execute_input":"2024-11-12T18:08:34.77376Z","iopub.status.idle":"2024-11-12T18:08:34.801395Z","shell.execute_reply.started":"2024-11-12T18:08:34.773736Z","shell.execute_reply":"2024-11-12T18:08:34.800326Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The dataset provided in the competition consists of DICOM images. We will not be training on the DICOM images, rather would work on PNG image which are extracted from the DICOM format.\n\nA helpful resource on the conversion of DICOM to PNG - https://www.kaggle.com/code/radek1/how-to-process-dicom-images-to-pngs","metadata":{}},{"cell_type":"code","source":"BASE_PATH = f'/kaggle/input/rsna-atd-512x512-png-v2-dataset'","metadata":{"execution":{"iopub.status.busy":"2024-11-12T18:08:34.810072Z","iopub.execute_input":"2024-11-12T18:08:34.810367Z","iopub.status.idle":"2024-11-12T18:08:34.816821Z","shell.execute_reply.started":"2024-11-12T18:08:34.810343Z","shell.execute_reply":"2024-11-12T18:08:34.81584Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"train.csv contains metadata","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(f'{BASE_PATH}/train.csv')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2024-11-12T18:08:34.823717Z","iopub.execute_input":"2024-11-12T18:08:34.824041Z","iopub.status.idle":"2024-11-12T18:08:34.880408Z","shell.execute_reply.started":"2024-11-12T18:08:34.824016Z","shell.execute_reply":"2024-11-12T18:08:34.879447Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df[\"image_path\"] = f\"{BASE_PATH}/train_images\"\\\n                    + \"/\" + df.patient_id.astype(str)\\\n                    + \"/\" + df.series_id.astype(str)\\\n                    + \"/\" + df.instance_number.astype(str) +\".png\"\n\ndf = df.drop_duplicates()","metadata":{"execution":{"iopub.status.busy":"2024-11-12T18:08:34.882349Z","iopub.execute_input":"2024-11-12T18:08:34.882658Z","iopub.status.idle":"2024-11-12T18:08:34.939804Z","shell.execute_reply.started":"2024-11-12T18:08:34.882632Z","shell.execute_reply":"2024-11-12T18:08:34.938943Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2024-11-12T18:08:34.941012Z","iopub.execute_input":"2024-11-12T18:08:34.94134Z","iopub.status.idle":"2024-11-12T18:08:34.967497Z","shell.execute_reply.started":"2024-11-12T18:08:34.941312Z","shell.execute_reply":"2024-11-12T18:08:34.966528Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"We split the training dataset into train and validation. This is a common practise in the Machine Learning pipelines. We not only want to train our model, but also want to validate it's training.\n\nA small catch here is that the training and validation data should have an aligned data distribution. Here we handle that by grouping the lables and then splitting the dataset. This ensures an aligned data distribution between the training and the validation splits.","metadata":{}},{"cell_type":"code","source":"#Split_group function \"\"\ndef split_group(group,test_size=0.2):\n    if len(group)==1:\n        return (group, pd.DataFrame()) if np.random.rand() < test_size else (pd.DataFrame()\n                                                                             ,group)\n    else:\n        return train_test_split(group,test_size=test_size,random_state=69)\n    \n","metadata":{"execution":{"iopub.status.busy":"2024-11-12T18:08:34.969695Z","iopub.execute_input":"2024-11-12T18:08:34.970005Z","iopub.status.idle":"2024-11-12T18:08:34.97598Z","shell.execute_reply.started":"2024-11-12T18:08:34.969977Z","shell.execute_reply":"2024-11-12T18:08:34.974978Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#initialize 2 empty dfs for train and val\n\ntrain_data = pd.DataFrame()\nval_data = pd.DataFrame()","metadata":{"execution":{"iopub.status.busy":"2024-11-12T18:08:34.977253Z","iopub.execute_input":"2024-11-12T18:08:34.977619Z","iopub.status.idle":"2024-11-12T18:08:34.987009Z","shell.execute_reply.started":"2024-11-12T18:08:34.977583Z","shell.execute_reply":"2024-11-12T18:08:34.986097Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Iterate through the groups and split them, handling single-sample groups\nfor _, group in df.groupby(config.TARGET_COLS):\n    train_group, val_group = split_group(group)\n    train_data = pd.concat([train_data,train_group],ignore_index=True)\n    val_data = pd.concat([val_data,val_group],ignore_index=True)\n    \nprint(train_data.shape)\nprint(val_data.shape)","metadata":{"execution":{"iopub.status.busy":"2024-11-12T18:08:34.988211Z","iopub.execute_input":"2024-11-12T18:08:34.988496Z","iopub.status.idle":"2024-11-12T18:08:35.088802Z","shell.execute_reply.started":"2024-11-12T18:08:34.988472Z","shell.execute_reply":"2024-11-12T18:08:35.087872Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Data Pipeline /w tf.data\n'\n\nHere we build the data pipeline using tf.data. Using tf.data we can map out data to an augmentation pipeline simple by using the map API.\n\nAdding augmentations to the data pipeline is as simple as adding a layer into the list of layers that the Augmenter processes.\n\nReference: https://keras.io/api/keras_cv/layers/augmentation/","metadata":{}},{"cell_type":"code","source":"#from keras_cv.augment.preprocess.augment import RandomFlip\nfrom tensorflow.keras.layers.experimental import preprocessing\n\ndef decode_image_and_label(image_path, label):\n    file_bytes = tf.io.read_file(image_path)\n    image = tf.io.decode_png(file_bytes, channels=3, dtype=tf.uint8)\n    image = tf.image.resize(image, config.IMAGE_SIZE, method=\"bilinear\")\n    image = tf.cast(image, tf.float32) / 255.0\n    \n    label = tf.cast(label, tf.float32)\n    #         bowel       fluid       kidney      liver       spleen\n    labels = (label[0:1], label[1:2], label[2:5], label[5:8], label[8:11])\n    \n    return (image, labels)\n\n\ndef apply_augmentation(images, labels):\n    augmenter = keras.Sequential(\n        \n        layers=[\n            keras_cv.layers.RandomFlip(mode=\"horizontal_and_vertical\"),\n            keras_cv.layers.RandomCutout(height_factor=0.2, width_factor=0.2),\n            \n        ]\n    )\n    aug = augmenter(images)\n    return (aug, labels)\n\n# Độ nhiễu - Gaussian Noise\ndef add_noise(image, noise_factor=0.1):\n    noise = tf.random.normal(shape=tf.shape(image), mean=0.0, stddev=noise_factor, dtype=tf.float32)\n    image = image + noise\n    image = tf.clip_by_value(image, 0.0, 1.0)\n    return image\n\ndef aug(image, labels):\n    # Thêm độ nhiễu vào ảnh\n    image = add_noise(image)  # <-- Thêm độ nhiễu ở đây\n    # Các phép biến đổi khác\n    image = preprocessing.RandomFlip(mode=\"horizontal_and_vertical\")(image)\n    image = preprocessing.RandomRotation(factor=0.2)(image)\n    image = preprocessing.RandomTranslation(height_factor=0.2, width_factor=0.2)(image)\n    return (image, labels)\n","metadata":{"execution":{"iopub.status.busy":"2024-11-12T18:08:35.090304Z","iopub.execute_input":"2024-11-12T18:08:35.090689Z","iopub.status.idle":"2024-11-12T18:08:35.102302Z","shell.execute_reply.started":"2024-11-12T18:08:35.090637Z","shell.execute_reply":"2024-11-12T18:08:35.101336Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def build_dataset(image_paths, labels):\n    ds = (\n        tf.data.Dataset.from_tensor_slices((image_paths, labels))\n        .map(decode_image_and_label, num_parallel_calls=config.AUTOTUNE)\n        .shuffle(config.BATCH_SIZE * 10)\n        .batch(config.BATCH_SIZE)\n        .map(aug, num_parallel_calls=config.AUTOTUNE)\n        .prefetch(config.AUTOTUNE)\n    )\n    return ds","metadata":{"execution":{"iopub.status.busy":"2024-11-12T18:08:35.103535Z","iopub.execute_input":"2024-11-12T18:08:35.103818Z","iopub.status.idle":"2024-11-12T18:08:35.115309Z","shell.execute_reply.started":"2024-11-12T18:08:35.103794Z","shell.execute_reply":"2024-11-12T18:08:35.114449Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"paths  = train_data.image_path.tolist()\nlabels = train_data[config.TARGET_COLS].values\n\nds = build_dataset(image_paths=paths, labels=labels)\nimages, labels = next(iter(ds))\nimages.shape, [label.shape for label in labels]","metadata":{"execution":{"iopub.status.busy":"2024-11-12T18:08:35.116391Z","iopub.execute_input":"2024-11-12T18:08:35.116716Z","iopub.status.idle":"2024-11-12T18:08:36.100237Z","shell.execute_reply.started":"2024-11-12T18:08:35.11669Z","shell.execute_reply":"2024-11-12T18:08:36.099254Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"keras_cv.visualization.plot_image_gallery(\n    images=images,\n    value_range=(0,1),\n    rows=2,\n    cols=2\n)","metadata":{"execution":{"iopub.status.busy":"2024-11-12T18:08:36.104737Z","iopub.execute_input":"2024-11-12T18:08:36.105026Z","iopub.status.idle":"2024-11-12T18:08:36.408483Z","shell.execute_reply.started":"2024-11-12T18:08:36.105001Z","shell.execute_reply":"2024-11-12T18:08:36.407426Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"BUILD MODEL","metadata":{}},{"cell_type":"code","source":"# get image_paths and labels\nprint(\"[INFO] Building the dataset...\")\ntrain_paths = train_data.image_path.values; train_labels = train_data[config.TARGET_COLS].values.astype(np.float32)\nvalid_paths = val_data.image_path.values; valid_labels = val_data[config.TARGET_COLS].values.astype(np.float32)\n\n# train and valid dataset\ntrain_ds = build_dataset(image_paths=train_paths, labels=train_labels)\nval_ds = build_dataset(image_paths=valid_paths, labels=valid_labels)\n\ntotal_train_steps = train_ds.cardinality().numpy() * config.BATCH_SIZE * config.EPOCHS\nwarmup_steps = int(total_train_steps * 0.10)\ndecay_steps = total_train_steps - warmup_steps\n\nprint(f\"{total_train_steps=}\")\nprint(f\"{warmup_steps=}\")\nprint(f\"{decay_steps=}\")","metadata":{"execution":{"iopub.status.busy":"2024-11-12T18:08:36.40979Z","iopub.execute_input":"2024-11-12T18:08:36.411325Z","iopub.status.idle":"2024-11-12T18:08:36.99484Z","shell.execute_reply.started":"2024-11-12T18:08:36.411281Z","shell.execute_reply":"2024-11-12T18:08:36.993893Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# build the model\n#print(\"[INFO] Building the model...\")\n#model = build_model(warmup_steps, decay_steps)\n\n# train\n#print(\"[INFO] Training...\")\n#history = model.fit(\n#    train_ds,\n#    epochs=config.EPOCHS,\n#    validation_data=val_ds,\n#)","metadata":{"execution":{"iopub.status.busy":"2024-11-12T18:08:36.996141Z","iopub.execute_input":"2024-11-12T18:08:36.996542Z","iopub.status.idle":"2024-11-12T18:08:37.001268Z","shell.execute_reply.started":"2024-11-12T18:08:36.996505Z","shell.execute_reply":"2024-11-12T18:08:37.000331Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.metrics import Precision, Recall\n\ndef build_model(warmup_steps, decay_steps):\n    # Define Input\n    inputs = keras.Input(shape=config.IMAGE_SIZE + [3,], batch_size=config.BATCH_SIZE)\n    \n    # Define Backbone\n    backbone = keras_cv.models.YOLOV8Backbone.from_preset('yolo_v8_xl_backbone_coco')\n    include_rescaling = False\n    x = inputs\n    \n    # GAP to get the activation maps\n    gap = keras.layers.GlobalAveragePooling2D()\n    x = gap(x)\n\n    # Define 'necks' for each head\n    x_bowel = keras.layers.Dense(32, activation='silu')(x)\n    x_extra = keras.layers.Dense(32, activation='silu')(x)\n    x_liver = keras.layers.Dense(32, activation='silu')(x)\n    x_kidney = keras.layers.Dense(32, activation='silu')(x)\n    x_spleen = keras.layers.Dense(32, activation='silu')(x)\n\n    # Define heads\n    out_bowel = keras.layers.Dense(1, name='bowel', activation='sigmoid')(x_bowel) # use sigmoid to convert predictions to [0-1]\n    out_extra = keras.layers.Dense(1, name='extra', activation='sigmoid')(x_extra) # use sigmoid to convert predictions to [0-1]\n    out_liver = keras.layers.Dense(3, name='liver', activation='softmax')(x_liver) # use softmax for the liver head\n    out_kidney = keras.layers.Dense(3, name='kidney', activation='softmax')(x_kidney) # use softmax for the kidney head\n    out_spleen = keras.layers.Dense(3, name='spleen', activation='softmax')(x_spleen) # use softmax for the spleen head\n    \n    # Concatenate the outputs\n    outputs = [out_bowel, out_extra, out_liver, out_kidney, out_spleen]\n\n    # Create model\n    print(\"[INFO] Building the model...\")\n    model = keras.Model(inputs=inputs, outputs=outputs)\n    \n    # Cosine Decay\n    cosine_decay = keras.optimizers.schedules.CosineDecay(\n        initial_learning_rate=1e-4,\n        decay_steps=decay_steps,\n        alpha=0.0,\n        warmup_target=1e-3,\n        warmup_steps=warmup_steps,\n    )\n\n    # Define metrics\n    precision = Precision()\n    recall = Recall()\n\n    # Compile the model\n    optimizer = keras.optimizers.Adam(learning_rate=cosine_decay)\n    loss = {\n        \"bowel\":keras.losses.BinaryCrossentropy(),\n        \"extra\":keras.losses.BinaryCrossentropy(),\n        \"liver\":keras.losses.CategoricalCrossentropy(),\n        \"kidney\":keras.losses.CategoricalCrossentropy(),\n        \"spleen\":keras.losses.CategoricalCrossentropy(),\n    }\n    metrics = {\n        \"bowel\": [precision, recall, \"accuracy\"],\n        \"extra\": [precision, recall, \"accuracy\"],\n        \"liver\": [\"accuracy\"],\n        \"kidney\": [\"accuracy\"],\n        \"spleen\": [\"accuracy\"],\n    }\n    print(\"[INFO] Compiling the model...\")\n    model.compile(\n        optimizer=optimizer,\n        loss=loss,\n        metrics=metrics\n    )\n    \n    return model\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T18:08:37.002588Z","iopub.execute_input":"2024-11-12T18:08:37.002881Z","iopub.status.idle":"2024-11-12T18:08:37.01879Z","shell.execute_reply.started":"2024-11-12T18:08:37.002855Z","shell.execute_reply":"2024-11-12T18:08:37.017833Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Sau khi huấn luyện mô hình, in ra các keys có trong history\nprint(history.history.keys())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T18:08:37.02027Z","iopub.execute_input":"2024-11-12T18:08:37.020625Z","iopub.status.idle":"2024-11-12T18:08:37.031536Z","shell.execute_reply.started":"2024-11-12T18:08:37.020598Z","shell.execute_reply":"2024-11-12T18:08:37.030665Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import KFold  # Import KFold\n\n# Số fold bạn muốn (0 đến 4 -> tức là 5 fold)\nn_splits = 5\nkf = KFold(n_splits=n_splits, shuffle=True, random_state=config.SEED)\n\n# Khởi tạo các danh sách lưu trữ kết quả cho mỗi fold\nall_f1_scores = []\nall_acc_scores = []\nall_precision_scores = []\nall_recall_scores = []\n\n# Lặp qua từng fold (từ fold 0 đến fold 4)\nfor fold, (train_idx, val_idx) in enumerate(kf.split(train_data)):\n    # Kiểm tra fold đang được sử dụng (ở đây chúng ta bắt đầu từ fold 0)\n    print(f\"[INFO] Training fold {fold}/{n_splits - 1}...\")  # In từ 0 đến 4\n    \n    # Chia dữ liệu cho mỗi fold\n    train_fold_data = train_data.iloc[train_idx]\n    val_fold_data = train_data.iloc[val_idx]\n    \n    # Tạo datasets cho mỗi fold\n    train_paths = train_fold_data.image_path.values\n    train_labels = train_fold_data[config.TARGET_COLS].values.astype(np.float32)\n    val_paths = val_fold_data.image_path.values\n    val_labels = val_fold_data[config.TARGET_COLS].values.astype(np.float32)\n\n    train_ds = build_dataset(image_paths=train_paths, labels=train_labels)\n    val_ds = build_dataset(image_paths=val_paths, labels=val_labels)\n\n    total_train_steps = train_ds.cardinality().numpy() * config.BATCH_SIZE * config.EPOCHS\n    warmup_steps = int(total_train_steps * 0.10)\n    decay_steps = total_train_steps - warmup_steps\n\n    # Xây dựng mô hình cho mỗi fold\n    model = build_model(warmup_steps, decay_steps)\n\n    # Huấn luyện mô hình và ghi lại kết quả mỗi epoch\n    history = model.fit(\n        train_ds,\n        epochs=config.EPOCHS,\n        validation_data=val_ds,\n    )\n\n    # Lưu kết quả cho mỗi fold\n    fold_f1_scores = []\n    fold_acc_scores = []\n    fold_precision_scores = []\n    fold_recall_scores = []\n\n  # Duyệt qua các epoch và lưu lại các phép đo (Accuracy, Precision, Recall, F1 Score)\nfor epoch in range(config.EPOCHS):\n    # Lấy giá trị các phép đo tại epoch này\n    epoch_bowel_acc = history.history['bowel_accuracy'][epoch]\n    epoch_extra_acc = history.history['extra_accuracy'][epoch]\n    epoch_liver_acc = history.history['liver_accuracy'][epoch]\n    epoch_kidney_acc = history.history['kidney_accuracy'][epoch]\n    epoch_spleen_acc = history.history['spleen_accuracy'][epoch]\n\n    # Dùng đúng tên metric precision và recall với hậu tố \"_7\"\n    epoch_bowel_precision = history.history['bowel_precision_7'][epoch]\n    epoch_extra_precision = history.history['extra_precision_7'][epoch]\n    epoch_liver_precision = history.history['liver_precision_7'][epoch]\n    epoch_kidney_precision = history.history['kidney_precision_7'][epoch]\n    epoch_spleen_precision = history.history['spleen_precision_7'][epoch]\n\n    epoch_bowel_recall = history.history['bowel_recall_7'][epoch]\n    epoch_extra_recall = history.history['extra_recall_7'][epoch]\n    epoch_liver_recall = history.history['liver_recall_7'][epoch]\n    epoch_kidney_recall = history.history['kidney_recall_7'][epoch]\n    epoch_spleen_recall = history.history['spleen_recall_7'][epoch]\n\n    # Tính F1 Score cho từng lớp\n    bowel_f1 = 2 * (epoch_bowel_precision * epoch_bowel_recall) / (epoch_bowel_precision + epoch_bowel_recall + tf.keras.backend.epsilon())\n    extra_f1 = 2 * (epoch_extra_precision * epoch_extra_recall) / (epoch_extra_precision + epoch_extra_recall + tf.keras.backend.epsilon())\n    liver_f1 = 2 * (epoch_liver_precision * epoch_liver_recall) / (epoch_liver_precision + epoch_liver_recall + tf.keras.backend.epsilon())\n    kidney_f1 = 2 * (epoch_kidney_precision * epoch_kidney_recall) / (epoch_kidney_precision + epoch_kidney_recall + tf.keras.backend.epsilon())\n    spleen_f1 = 2 * (epoch_spleen_precision * epoch_spleen_recall) / (epoch_spleen_precision + epoch_spleen_recall + tf.keras.backend.epsilon())\n\n    # Lưu các phép đo của mỗi epoch\n    fold_f1_scores.append(np.mean([bowel_f1, extra_f1, liver_f1, kidney_f1, spleen_f1]))\n    fold_acc_scores.append(np.mean([epoch_bowel_acc, epoch_extra_acc, epoch_liver_acc, epoch_kidney_acc, epoch_spleen_acc]))\n    fold_precision_scores.append(np.mean([epoch_bowel_precision, epoch_extra_precision, epoch_liver_precision, epoch_kidney_precision, epoch_spleen_precision]))\n    fold_recall_scores.append(np.mean([epoch_bowel_recall, epoch_extra_recall, epoch_liver_recall, epoch_kidney_recall, epoch_spleen_recall]))\n\n    # Lưu trữ kết quả\n    all_f1_scores.append(fold_f1_scores)\n    all_acc_scores.append(fold_acc_scores)\n    all_precision_scores.append(fold_precision_scores)\n    all_recall_scores.append(fold_recall_scores)\n\n# Tính toán kết quả trung bình cho các fold\navg_f1_scores = np.mean(all_f1_scores, axis=0)\navg_acc_scores = np.mean(all_acc_scores, axis=0)\navg_precision_scores = np.mean(all_precision_scores, axis=0)\navg_recall_scores = np.mean(all_recall_scores, axis=0)\n\n# In kết quả trung bình của các fold\nprint(\"[INFO] K-fold training results (mean over folds):\")\nfor epoch in range(config.EPOCHS):\n    print(f\"Epoch {epoch}:\")\n    print(f\"  Mean Accuracy: {avg_acc_scores[epoch]:.4f}\")\n    print(f\"  Mean Precision: {avg_precision_scores[epoch]:.4f}\")\n    print(f\"  Mean Recall: {avg_recall_scores[epoch]:.4f}\")\n    print(f\"  Mean F1 Score: {avg_f1_scores[epoch]:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T18:24:55.491266Z","iopub.execute_input":"2024-11-12T18:24:55.491646Z","iopub.status.idle":"2024-11-12T18:25:04.921325Z","shell.execute_reply.started":"2024-11-12T18:24:55.491611Z","shell.execute_reply":"2024-11-12T18:25:04.918017Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Hàm tính F1 Score\ndef calculate_f1(precision, recall):\n    return 2 * (precision * recall) / (precision + recall + tf.keras.backend.epsilon())\n\n# Chỉ số training\nmetrics = {\n    \"Bowel\": {\n        \"Accuracy\": history.history['bowel_accuracy'][-1],\n        \"Precision\": history.history['bowel_precision'][-1],\n        \"Recall\": history.history['bowel_recall'][-1],\n        \"F1 Score\": calculate_f1(history.history['bowel_precision'][-1], history.history['bowel_recall'][-1])\n    },\n    \"Extravasation\": {\n        \"Accuracy\": history.history['extra_accuracy'][-1],\n        \"Precision\": history.history['extra_precision'][-1],\n        \"Recall\": history.history['extra_recall'][-1],\n        \"F1 Score\": calculate_f1(history.history['extra_precision'][-1], history.history['extra_recall'][-1])\n    },\n    \"Kidney\": {\n        \"Accuracy\": history.history['kidney_accuracy'][-1],\n        \"Precision\": None,  # Precision cho Kidney không có (vì sử dụng softmax cho 3 class)\n        \"Recall\": None,  # Recall cho Kidney không có (vì sử dụng softmax cho 3 class)\n        \"F1 Score\": None  # F1 không có cho Kidney nếu không có Precision và Recall\n    },\n    \"Liver\": {\n        \"Accuracy\": history.history['liver_accuracy'][-1],\n        \"Precision\": None,  # Precision cho Liver không có (vì sử dụng softmax cho 3 class)\n        \"Recall\": None,  # Recall cho Liver không có (vì sử dụng softmax cho 3 class)\n        \"F1 Score\": None  # F1 không có cho Liver nếu không có Precision và Recall\n    },\n    \"Spleen\": {\n        \"Accuracy\": history.history['spleen_accuracy'][-1],\n        \"Precision\": None,  # Precision cho Spleen không có (vì sử dụng softmax cho 3 class)\n        \"Recall\": None,  # Recall cho Spleen không có (vì sử dụng softmax cho 3 class)\n        \"F1 Score\": None  # F1 không có cho Spleen nếu không có Precision và Recall\n    },\n}\n\n# In ra bảng\nimport pandas as pd\n\n# Tạo DataFrame\nmetrics_df = pd.DataFrame(metrics).T\nmetrics_df = metrics_df.round(4)\n\n# In bảng\nprint(\"Model Performance (Final Epoch):\")\nprint(metrics_df)\n\n# In ra kết quả cuối cùng của từng class (organ)\nfor organ in metrics:\n    print(f\"{organ} - Accuracy: {metrics[organ]['Accuracy']}, Precision: {metrics[organ]['Precision']}, Recall: {metrics[organ]['Recall']}, F1 Score: {metrics[organ]['F1 Score']}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T18:23:49.434994Z","iopub.status.idle":"2024-11-12T18:23:49.435577Z","shell.execute_reply.started":"2024-11-12T18:23:49.435288Z","shell.execute_reply":"2024-11-12T18:23:49.435315Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import f1_score\nimport tensorflow_addons as tfa\n\n# Hàm tính độ nhạy\ndef sensitivity(y_true, y_pred):\n    true_positives = tf.reduce_sum(y_true * y_pred)\n    possible_positives = tf.reduce_sum(y_true)\n    return true_positives / (possible_positives + tf.keras.backend.epsilon())\n\n# F1 Score\ndef f1(y_true, y_pred):\n    precision = tf.keras.metrics.Precision()(y_true, y_pred)\n    recall = sensitivity(y_true, y_pred)\n    return 2 * (precision * recall) / (precision + recall + tf.keras.backend.epsilon())\n\n# Độ nhiễu (Noise) - Gaussian Noise\ndef add_noise(image, noise_factor=0.1):\n    noise = tf.random.normal(shape=tf.shape(image), mean=0.0, stddev=noise_factor, dtype=tf.float32)\n    image = image + noise\n    image = tf.clip_by_value(image, 0.0, 1.0)\n    return image\n\ndef aug(image, labels):\n    # Thêm độ nhiễu vào ảnh\n    image = add_noise(image)\n    # Các phép biến đổi khác\n    image = preprocessing.RandomFlip(mode=\"horizontal_and_vertical\")(image)\n    image = preprocessing.RandomRotation(factor=0.2)(image)\n    image = preprocessing.RandomTranslation(height_factor=0.2, width_factor=0.2)(image)\n    return (image, labels)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T18:23:49.437293Z","iopub.status.idle":"2024-11-12T18:23:49.437658Z","shell.execute_reply.started":"2024-11-12T18:23:49.43748Z","shell.execute_reply":"2024-11-12T18:23:49.437497Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create a 3x2 grid for the subplots\nfig, axes = plt.subplots(5, 1, figsize=(5, 15))\n\n# Flatten axes to iterate through them\naxes = axes.flatten()\n\n# Iterate through the metrics and plot them\nfor i, name in enumerate([\"bowel\", \"extra\", \"kidney\", \"liver\", \"spleen\"]):\n    # Plot training accuracy\n    axes[i].plot(history.history[name + '_accuracy'], label='Training ' + name)\n    # Plot validation accuracy\n    axes[i].plot(history.history['val_' + name + '_accuracy'], label='Validation ' + name)\n    axes[i].set_title(name)\n    axes[i].set_xlabel('Epoch')\n    axes[i].set_ylabel('Accuracy')\n    axes[i].legend()\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-12T18:23:49.438731Z","iopub.status.idle":"2024-11-12T18:23:49.43927Z","shell.execute_reply.started":"2024-11-12T18:23:49.438967Z","shell.execute_reply":"2024-11-12T18:23:49.438993Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.plot(history.history[\"loss\"], label=\"loss\")\nplt.plot(history.history[\"val_loss\"], label=\"val loss\")\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-12T18:23:49.440761Z","iopub.status.idle":"2024-11-12T18:23:49.441167Z","shell.execute_reply.started":"2024-11-12T18:23:49.440951Z","shell.execute_reply":"2024-11-12T18:23:49.440969Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# store best results\nbest_epoch = np.argmin(history.history['val_loss'])\nbest_loss = history.history['val_loss'][best_epoch]\nbest_acc_bowel = history.history['val_bowel_accuracy'][best_epoch]\nbest_acc_extra = history.history['val_extra_accuracy'][best_epoch]\nbest_acc_liver = history.history['val_liver_accuracy'][best_epoch]\nbest_acc_kidney = history.history['val_kidney_accuracy'][best_epoch]\nbest_acc_spleen = history.history['val_spleen_accuracy'][best_epoch]\n\n# Find mean accuracy\nbest_acc = np.mean(\n    [best_acc_bowel,\n     best_acc_extra,\n     best_acc_liver,\n     best_acc_kidney,\n     best_acc_spleen\n])\n\n\nprint(f'>>>> BEST Loss  : {best_loss:.4f}\\n>>>> BEST Acc   : {best_acc:.4f}\\n>>>> BEST Epoch : {best_epoch}\\n')\nprint('ORGAN Acc:')\nprint(f'  >>>> {\"Bowel\".ljust(15)} : {best_acc_bowel:.4f}')\nprint(f'  >>>> {\"Extravasation\".ljust(15)} : {best_acc_extra:.4f}')\nprint(f'  >>>> {\"Liver\".ljust(15)} : {best_acc_liver:.4f}')\nprint(f'  >>>> {\"Kidney\".ljust(15)} : {best_acc_kidney:.4f}')\nprint(f'  >>>> {\"Spleen\".ljust(15)} : {best_acc_spleen:.4f}')","metadata":{"execution":{"iopub.status.busy":"2024-11-12T18:23:49.442668Z","iopub.status.idle":"2024-11-12T18:23:49.443225Z","shell.execute_reply.started":"2024-11-12T18:23:49.442927Z","shell.execute_reply":"2024-11-12T18:23:49.442953Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2024-11-12T18:23:49.445326Z","iopub.status.idle":"2024-11-12T18:23:49.445666Z","shell.execute_reply.started":"2024-11-12T18:23:49.445494Z","shell.execute_reply":"2024-11-12T18:23:49.445509Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Save the model\n#import keras.models\nmodel.save(\"yolov8_xl_atd.keras\")","metadata":{"execution":{"iopub.status.busy":"2024-11-12T18:23:49.446884Z","iopub.status.idle":"2024-11-12T18:23:49.447256Z","shell.execute_reply.started":"2024-11-12T18:23:49.447046Z","shell.execute_reply":"2024-11-12T18:23:49.447086Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}