{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\n# Use the kagglehub client library to attach Kaggle resources like competitions, datasets, and models to your session\n# Learn more about kagglehub: https://github.com/Kaggle/kagglehub/blob/main/README.md\n\nimport kagglehub\n# kagglehub.dataset_download('<owner>/<dataset-slug>')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-06-23T07:45:36.646869Z","iopub.execute_input":"2026-06-23T07:45:36.647382Z","iopub.status.idle":"2026-06-23T07:46:14.67917Z","shell.execute_reply.started":"2026-06-23T07:45:36.647349Z","shell.execute_reply":"2026-06-23T07:46:14.678334Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nINPUT_DIR = \"/kaggle/input/competitions/prostate-cancer-grade-assessment\"\nWORKING_DIR = \"/kaggle/working/panda_selected_subset\"\n\n# Tạo thư mục gốc cho tập dữ liệu đã lọc\nos.makedirs(WORKING_DIR, exist_ok=True)\n\n# Tạo thư mục con chứa các file ảnh .tiff gốc phục vụ cho phần xử lý sau\nos.makedirs(os.path.join(WORKING_DIR, \"train_images\"), exist_ok=True)\nos.makedirs(os.path.join(WORKING_DIR, \"train_label_masks\"), exist_ok=True)\n\nprint(f\"Đã khởi tạo cấu trúc thư mục làm việc tại: {WORKING_DIR}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-23T07:47:44.876472Z","iopub.execute_input":"2026-06-23T07:47:44.876813Z","iopub.status.idle":"2026-06-23T07:47:44.883894Z","shell.execute_reply.started":"2026-06-23T07:47:44.87678Z","shell.execute_reply":"2026-06-23T07:47:44.883032Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport os\n\n# 1. Đọc file metadata tổng\ndf_total = pd.read_csv(os.path.join(INPUT_DIR, \"train.csv\"))\n\n# 2. Bộ lọc 1: Chỉ lấy từ nguồn Radboud\ndf_filtered = df_total[df_total['data_provider'] == 'radboud'].copy()\n\n# 3. Bộ lọc 2: Chỉ lấy các Slide có điểm Gleason chuẩn xác, loại bỏ nhãn nhiễu\nvalid_scores = ['0+0', '3+3', '3+4', '4+3', '4+4', '4+5', '5+4', '5+5']\ndf_filtered = df_filtered[df_filtered['gleason_score'].isin(valid_scores)]\n\n# --- BỔ SUNG: KIỂM TRA SỰ TỒN TẠI CỦA FILE MASK VẬT LÝ ---\ndef check_mask_exists(image_id):\n    mask_path = os.path.join(INPUT_DIR, \"train_label_masks\", f\"{image_id}_mask.tiff\")\n    image_path = os.path.join(INPUT_DIR, \"train_images\", f\"{image_id}.tiff\")\n    # Chỉ giữ lại mẫu nếu đồng thời tồn tại cả file ảnh và file mask\n    return os.path.exists(mask_path) and os.path.exists(image_path)\n\n# Thêm một cột tạm để đánh dấu các mẫu có đầy đủ file\ndf_filtered['has_both_files'] = df_filtered['image_id'].apply(check_mask_exists)\n# Lọc bỏ hoàn toàn các mẫu bị thiếu file mask hoặc file ảnh\ndf_filtered = df_filtered[df_filtered['has_both_files'] == True]\n# --- HẾT PHẦN BỔ SUNG ---\n\n# 4. Bộ lọc 3: Giới hạn số lượng mẫu trên mỗi nhóm điểm (Cân bằng dữ liệu)\nSAMPLES_PER_SCORE = 50  \ndf_selected = df_filtered.groupby('gleason_score').apply(\n    lambda x: x.sample(n=min(len(x), SAMPLES_PER_SCORE), random_state=42)\n).reset_index(drop=True)\n\n# Loại bỏ cột tạm trước khi lưu\ndf_selected = df_selected.drop(columns=['has_both_files'])\n\n# 5. Lưu lại file Metadata mới dành riêng cho tập con này\ndf_selected.to_csv(os.path.join(WORKING_DIR, \"selected_train.csv\"), index=False)\n\nprint(\"--- KẾT QUẢ LỌC DỮ LIỆU ĐÃ ĐẢM BẢO CÓ MASK ---\")\nprint(f\"Tổng số Slide được chọn: {len(df_selected)}\")\nprint(\"\\nPhân phối chi tiết theo từng Gleason Score:\")\nprint(df_selected['gleason_score'].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-23T07:47:53.706795Z","iopub.execute_input":"2026-06-23T07:47:53.707522Z","iopub.status.idle":"2026-06-23T07:47:58.517775Z","shell.execute_reply.started":"2026-06-23T07:47:53.707486Z","shell.execute_reply":"2026-06-23T07:47:58.516675Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport shutil\nimport pandas as pd\nfrom tqdm import tqdm\n\n# 1. Định nghĩa các đường dẫn gốc của Kaggle và thư mục làm việc của bạn\nINPUT_DIR = \"/kaggle/input/competitions/prostate-cancer-grade-assessment\"\nWORKING_DIR = \"/kaggle/working/panda_selected_subset\"  # Đảm bảo trùng với tên thư mục bạn lưu file CSV trước đó\n\ncsv_path = os.path.join(WORKING_DIR, \"selected_train.csv\")\n\n# Kiểm tra xem file CSV có tồn tại không trước khi đọc\nif not os.path.exists(csv_path):\n    print(f\"❌ Không tìm thấy file {csv_path}. Bạn hãy kiểm tra lại đường dẫn lưu file CSV ở bước trước!\")\nelse:\n    # --- PHẦN ĐỌC FILE SELECTED_TRAIN ---\n    selected_train = pd.read_csv(csv_path)\n    \n    # Giới hạn lấy đúng 350 slide từ danh sách đã đọc\n    df_to_download = selected_train.head(350) \n    \n    # Định nghĩa và khởi tạo thư mục đích train_images và train_label_masks\n    dst_img_dir = os.path.join(WORKING_DIR, \"train_images\")\n    dst_mask_dir = os.path.join(WORKING_DIR, \"train_label_masks\")\n    os.makedirs(dst_img_dir, exist_ok=True)\n    os.makedirs(dst_mask_dir, exist_ok=True)\n\n    print(f\"🚀 Đã đọc file CSV thành công. Bắt đầu sao chép vật lý {len(df_to_download)} cặp (Ảnh + Mask) gốc...\")\n    success_count = 0\n\n    # 2. Tiến hành sao chép vật lý trực tiếp\n    for idx, row in tqdm(df_to_download.iterrows(), total=len(df_to_download)):\n        image_id = row['image_id']\n        \n        # Đường dẫn nguồn (Dữ liệu gốc từ Kaggle Input)\n        src_img = os.path.join(INPUT_DIR, \"train_images\", f\"{image_id}.tiff\")\n        src_mask = os.path.join(INPUT_DIR, \"train_label_masks\", f\"{image_id}_mask.tiff\")\n        \n        # Đường dẫn đích (Thư mục làm việc trên Notebook để tải về)\n        dst_img = os.path.join(dst_img_dir, f\"{image_id}.tiff\")\n        dst_mask = os.path.join(dst_mask_dir, f\"{image_id}_mask.tiff\")\n        \n        # Sao chép nếu file nguồn tồn tại và file đích chưa có\n        if os.path.exists(src_img) and os.path.exists(src_mask):\n            if not os.path.exists(dst_img):\n                shutil.copy(src_img, dst_img)\n            if not os.path.exists(dst_mask):\n                shutil.copy(src_mask, dst_mask)\n            success_count += 1\n\n    print(\"\\n--- 🎉 HOÀN THÀNH ĐƯA ẢNH VÀ MASK VÀO THƯ MỤC SÀNG LỌC ---\")\n    print(f\"Đã sao chép thành công: {success_count}/{len(df_to_download)} slide.\")\n    print(f\"Vị trí ảnh (.tiff): {dst_img_dir}\")\n    print(f\"Vị trí mask (_mask.tiff): {dst_mask_dir}\")\n\n    # 3. Kiểm tra tổng dung lượng thư mục hiện tại để đảm bảo an toàn\n    print(\"\\n📊 Dung lượng thư mục đã sàng lọc trên Notebook của bạn:\")\n    !du -sh {WORKING_DIR}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-23T07:48:03.613669Z","iopub.execute_input":"2026-06-23T07:48:03.614033Z","iopub.status.idle":"2026-06-23T07:50:08.673313Z","shell.execute_reply.started":"2026-06-23T07:48:03.614003Z","shell.execute_reply":"2026-06-23T07:50:08.671724Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1. Nén toàn bộ thư mục panda_selected_subset thành file zip duy nhất bên ngoài thư mục làm việc\n!zip -r -q /kaggle/working/panda_filtered_data.zip /kaggle/working/panda_selected_subset\n\n# 2. Kiểm tra lại dung lượng thực tế của file zip sau khi nén\n!ls -lh /kaggle/working/panda_filtered_data.zip","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-23T07:50:29.759542Z","iopub.execute_input":"2026-06-23T07:50:29.760022Z","iopub.status.idle":"2026-06-23T07:56:59.282627Z","shell.execute_reply.started":"2026-06-23T07:50:29.759978Z","shell.execute_reply":"2026-06-23T07:56:59.281361Z"}},"outputs":[],"execution_count":null}]}