{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":52254,"databundleVersionId":8756537,"sourceType":"competition"}],"dockerImageVersionId":30762,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install kaggle\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nimport pandas as pd\nimport numpy as np\nimport os\nimport cv2  # OpenCV cho xử lý ảnh\nfrom sklearn.model_selection import KFold\nfrom sklearn.metrics import accuracy_score, f1_score\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Flatten, Conv2D, MaxPooling2D, Dropout\nfrom tensorflow.keras.optimizers import Adam\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Đọc file nhãn\nlabels_df = pd.read_csv('/kaggle/input/rsna-2023-abdominal-trauma-detection/train_labels.csv')\n\n# Hàm để tải hình ảnh và nhãn từ DataFrame\ndef load_images_and_labels(image_dir, dataframe, image_size=(128, 128)):\n    images = []\n    labels = []\n    \n    for index, row in dataframe.iterrows():\n        image_path = os.path.join(image_dir, row['filename'])\n        image = cv2.imread(image_path, cv2.IMREAD_GRAYSCALE)  # Đọc ảnh xám\n        image = cv2.resize(image, image_size)\n        images.append(image)\n        labels.append(row['label'])\n        \n    return np.array(images), np.array(labels)\n\n# Load dữ liệu hình ảnh và nhãn\nX, y = load_images_and_labels('/kaggle/input/rsna-2023-abdominal-trauma-detection/train_images', labels_df)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Chuyển đổi nhãn thành dạng nhị phân nếu cần thiết\ny = np.array(y).astype(int)\n\n# Thay đổi kích thước dữ liệu hình ảnh thành định dạng mong muốn của Keras (số lượng mẫu, chiều cao, chiều rộng, số kênh)\nX = X.reshape(X.shape[0], 128, 128, 1)  # 1 kênh vì ảnh xám\nX = X / 255.0  # Chuẩn hóa dữ liệu hình ảnh\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_model(input_shape):\n    model = Sequential([\n        Conv2D(32, (3, 3), activation='relu', input_shape=input_shape),\n        MaxPooling2D((2, 2)),\n        Conv2D(64, (3, 3), activation='relu'),\n        MaxPooling2D((2, 2)),\n        Flatten(),\n        Dense(128, activation='relu'),\n        Dropout(0.5),\n        Dense(1, activation='sigmoid')  # Đầu ra cho bài toán nhị phân\n    ])\n    \n    model.compile(optimizer=Adam(learning_rate=0.001),\n                  loss='binary_crossentropy',\n                  metrics=['accuracy'])\n    return model\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"kf = KFold(n_splits=4, shuffle=True, random_state=42)\nfold_no = 1\naccuracy_scores = []\nf1_scores = []\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for train_index, val_index in kf.split(X):\n    print(f'Fold {fold_no}...')\n\n    # Chia dữ liệu thành tập huấn luyện và tập kiểm tra\n    X_train, X_val = X[train_index], X[val_index]\n    y_train, y_val = y[train_index], y[val_index]\n\n    # Tạo mô hình\n    model = create_model(X_train.shape[1:])\n\n    # Huấn luyện mô hình\n    model.fit(X_train, y_train, epochs=10, batch_size=32, validation_data=(X_val, y_val), verbose=1)\n\n    # Dự đoán trên tập kiểm tra\n    y_pred = model.predict(X_val)\n    y_pred = (y_pred > 0.5).astype(int)\n\n    # Tính toán độ chính xác và F1 score\n    accuracy = accuracy_score(y_val, y_pred)\n    f1 = f1_score(y_val, y_pred)\n\n    accuracy_scores.append(accuracy)\n    f1_scores.append(f1)\n\n    print(f'Accuracy for fold {fold_no}: {accuracy}')\n    print(f'F1 Score for fold {fold_no}: {f1}')\n\n    fold_no += 1\n\nprint(f'Mean Accuracy: {np.mean(accuracy_scores)}')\nprint(f'Mean F1 Score: {np.mean(f1_scores)}')\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}