{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    print(os.path.join(dirname))\n    # for filename in filenames:\n        # print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\n# Use the kagglehub client library to attach Kaggle resources like competitions, datasets, and models to your session\n# Learn more about kagglehub: https://github.com/Kaggle/kagglehub/blob/main/README.md\n\nimport kagglehub\n# kagglehub.dataset_download('<owner>/<dataset-slug>')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-08-06T13:19:43.632986Z","iopub.execute_input":"2026-08-06T13:19:43.633378Z","iopub.status.idle":"2026-08-06T13:19:58.919977Z","shell.execute_reply.started":"2026-08-06T13:19:43.633342Z","shell.execute_reply":"2026-08-06T13:19:58.919056Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nINPUT_DIR = \"/kaggle/input/competitions/prostate-cancer-grade-assessment\"\nWORKING_DIR = \"/kaggle/working/panda_selected_subset\"\n\n# Create the root directory for the filtered dataset\nos.makedirs(WORKING_DIR, exist_ok=True)\n\n# Create subdirectories to store the original .tiff images and label masks for downstream processing\nos.makedirs(os.path.join(WORKING_DIR, \"train_images\"), exist_ok=True)\nos.makedirs(os.path.join(WORKING_DIR, \"train_label_masks\"), exist_ok=True)\n\nprint(f\"Initialized working directory structure at: {WORKING_DIR}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T13:20:02.215986Z","iopub.execute_input":"2026-08-06T13:20:02.21655Z","iopub.status.idle":"2026-08-06T13:20:02.22464Z","shell.execute_reply.started":"2026-08-06T13:20:02.216513Z","shell.execute_reply":"2026-08-06T13:20:02.223609Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport os\n\n# 1. Read the main metadata file\ndf_total = pd.read_csv(os.path.join(INPUT_DIR, \"train.csv\"))\n\n# 2. Filter 1: Keep only samples from the Radboud data provider\ndf_filtered = df_total[df_total['data_provider'] == 'radboud'].copy()\n\n# 3. Filter 2: Keep only slides with valid Gleason scores, filtering out noisy labels\nvalid_scores = ['0+0', '3+3', '3+4', '4+3', '4+4', '4+5', '5+4', '5+5']\ndf_filtered = df_filtered[df_filtered['gleason_score'].isin(valid_scores)]\n\n# --- ADDED: CHECK FOR PHYSICAL EXISTENCE OF THE MASK AND IMAGE FILES ---\ndef check_mask_exists(image_id):\n    mask_path = os.path.join(INPUT_DIR, \"train_label_masks\", f\"{image_id}_mask.tiff\")\n    image_path = os.path.join(INPUT_DIR, \"train_images\", f\"{image_id}.tiff\")\n    # Keep the sample only if both the image and mask files physically exist\n    return os.path.exists(mask_path) and os.path.exists(image_path)\n\n# Add a temporary column to flag samples that have both files\ndf_filtered['has_both_files'] = df_filtered['image_id'].apply(check_mask_exists)\n# Filter out any samples missing either the mask or image file\ndf_filtered = df_filtered[df_filtered['has_both_files'] == True]\n# --- END OF ADDED SECTION ---\n\n# 4. Filter 3: Limit the number of samples per score group (Data Balancing)\nSAMPLES_PER_SCORE = 50  \ndf_selected = df_filtered.groupby('gleason_score').apply(\n    lambda x: x.sample(n=min(len(x), SAMPLES_PER_SCORE), random_state=42)\n).reset_index(drop=True)\n\n# Drop the temporary flag column before saving\ndf_selected = df_selected.drop(columns=['has_both_files'])\n\n# 5. Save the new metadata file specifically for this subset\ndf_selected.to_csv(os.path.join(WORKING_DIR, \"selected_train.csv\"), index=False)\n\nprint(\"--- FILTERING RESULTS (VALIDATED WITH MASK AVAILABILITY) ---\")\nprint(f\"Total slides selected: {len(df_selected)}\")\nprint(\"\\nDetailed distribution per Gleason Score:\")\nprint(df_selected['gleason_score'].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T13:20:04.804314Z","iopub.execute_input":"2026-08-06T13:20:04.804644Z","iopub.status.idle":"2026-08-06T13:20:09.12566Z","shell.execute_reply.started":"2026-08-06T13:20:04.804617Z","shell.execute_reply":"2026-08-06T13:20:09.124617Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import pandas as pd\n# import os\n\n# # 1. Đọc file metadata tổng\n# df_total = pd.read_csv(os.path.join(INPUT_DIR, \"train.csv\"))\n\n# # 2. Bộ lọc 1: Chỉ lấy từ nguồn Radboud\n# df_filtered = df_total[df_total['data_provider'] == 'radboud'].copy()\n\n# # 3. Bộ lọc 2: Chỉ lấy các Slide có điểm Gleason chuẩn xác, loại bỏ nhãn nhiễu\n# valid_scores = ['0+0', '3+3', '3+4', '4+3', '4+4', '4+5', '5+4', '5+5']\n# df_filtered = df_filtered[df_filtered['gleason_score'].isin(valid_scores)]\n\n# # --- BỔ SUNG: KIỂM TRA SỰ TỒN TẠI CỦA FILE MASK VẬT LÝ ---\n# def check_mask_exists(image_id):\n#     mask_path = os.path.join(INPUT_DIR, \"train_label_masks\", f\"{image_id}_mask.tiff\")\n#     image_path = os.path.join(INPUT_DIR, \"train_images\", f\"{image_id}.tiff\")\n#     # Chỉ giữ lại mẫu nếu đồng thời tồn tại cả file ảnh và file mask\n#     return os.path.exists(mask_path) and os.path.exists(image_path)\n\n# # Thêm một cột tạm để đánh dấu các mẫu có đầy đủ file\n# df_filtered['has_both_files'] = df_filtered['image_id'].apply(check_mask_exists)\n# # Lọc bỏ hoàn toàn các mẫu bị thiếu file mask hoặc file ảnh\n# df_filtered = df_filtered[df_filtered['has_both_files'] == True]\n# # --- HẾT PHẦN BỔ SUNG ---\n\n# # 4. Bộ lọc 3: Giới hạn số lượng mẫu trên mỗi nhóm điểm (Cân bằng dữ liệu)\n# SAMPLES_PER_SCORE = 50  \n# df_selected = df_filtered.groupby('gleason_score').apply(\n#     lambda x: x.sample(n=min(len(x), SAMPLES_PER_SCORE), random_state=42)\n# ).reset_index(drop=True)\n\n# # Loại bỏ cột tạm trước khi lưu\n# df_selected = df_selected.drop(columns=['has_both_files'])\n\n# # 5. Lưu lại file Metadata mới dành riêng cho tập con này\n# df_selected.to_csv(os.path.join(WORKING_DIR, \"selected_train.csv\"), index=False)\n\n# print(\"--- KẾT QUẢ LỌC DỮ LIỆU ĐÃ ĐẢM BẢO CÓ MASK ---\")\n# print(f\"Tổng số Slide được chọn: {len(df_selected)}\")\n# print(\"\\nPhân phối chi tiết theo từng Gleason Score:\")\n# print(df_selected['gleason_score'].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-21T16:28:55.256467Z","iopub.execute_input":"2026-07-21T16:28:55.256722Z","iopub.status.idle":"2026-07-21T16:28:55.261646Z","shell.execute_reply.started":"2026-07-21T16:28:55.256698Z","shell.execute_reply":"2026-07-21T16:28:55.26089Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import os\n# import shutil\n# import pandas as pd\n# from tqdm import tqdm\n\n# # 1. Define Kaggle source paths and your working directory\n# INPUT_DIR = \"/kaggle/input/competitions/prostate-cancer-grade-assessment\"\n# WORKING_DIR = \"/kaggle/working/panda_selected_subset\"  # Ensure this matches the directory name where you previously saved the CSV file\n\n# csv_path = os.path.join(WORKING_DIR, \"selected_train.csv\")\n\n# # Check if the CSV file exists before reading\n# if not os.path.exists(csv_path):\n#     print(f\"❌ File {csv_path} not found. Please double-check the CSV save path from the previous step!\")\n# else:\n#     # --- SECTION: READ SELECTED_TRAIN ---\n#     selected_train = pd.read_csv(csv_path)\n    \n#     # Limit to exactly 350 slides from the loaded list\n#     df_to_download = selected_train.head(3) \n    \n#     # Define and initialize the destination directories for train_images and train_label_masks\n#     dst_img_dir = os.path.join(WORKING_DIR, \"train_images\")\n#     dst_mask_dir = os.path.join(WORKING_DIR, \"train_label_masks\")\n#     os.makedirs(dst_img_dir, exist_ok=True)\n#     os.makedirs(dst_mask_dir, exist_ok=True)\n\n#     print(f\" CSV file read successfully. Starting the physical copy of {len(df_to_download)} original pairs (Image + Mask)...\")\n#     success_count = 0\n\n#     # 2. Perform the physical copy directly\n#     for idx, row in tqdm(df_to_download.iterrows(), total=len(df_to_download)):\n#         image_id = row['image_id']\n        \n#         # Source paths (Original dataset from Kaggle Input)\n#         src_img = os.path.join(INPUT_DIR, \"train_images\", f\"{image_id}.tiff\")\n#         src_mask = os.path.join(INPUT_DIR, \"train_label_masks\", f\"{image_id}_mask.tiff\")\n        \n#         # Destination paths (Working directory on the Notebook for download)\n#         dst_img = os.path.join(dst_img_dir, f\"{image_id}.tiff\")\n#         dst_mask = os.path.join(dst_mask_dir, f\"{image_id}_mask.tiff\")\n        \n#         # Copy only if the source files exist and the destination files do not already exist\n#         if os.path.exists(src_img) and os.path.exists(src_mask):\n#             if not os.path.exists(dst_img):\n#                 shutil.copy(src_img, dst_img)\n#             if not os.path.exists(dst_mask):\n#                 shutil.copy(src_mask, dst_mask)\n#             success_count += 1\n\n#     print(\"\\n---  FINISHED SAVING IMAGES AND MASKS TO FILTERED DIRECTORY ---\")\n#     print(f\"Successfully copied: {success_count}/{len(df_to_download)} slides.\")\n#     print(f\"Image location (.tiff): {dst_img_dir}\")\n#     print(f\"Mask location (_mask.tiff): {dst_mask_dir}\")\n\n#     # 3. Check the total directory size to ensure safety\n#     print(\"\\n Current size of the filtered directory on your Notebook:\")\n#     !du -sh {WORKING_DIR}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-21T16:28:55.262683Z","iopub.execute_input":"2026-07-21T16:28:55.262929Z","iopub.status.idle":"2026-07-21T16:28:56.732455Z","shell.execute_reply.started":"2026-07-21T16:28:55.262906Z","shell.execute_reply":"2026-07-21T16:28:56.731401Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\n\n# 1. Define paths\nINPUT_DIR = \"/kaggle/input/competitions/prostate-cancer-grade-assessment\"\nWORKING_DIR = \"/kaggle/working/panda_selected_subset\"\nos.makedirs(WORKING_DIR, exist_ok=True)\n\ntrain_csv_path = os.path.join(INPUT_DIR, \"train.csv\")\ncsv_path = os.path.join(WORKING_DIR, \"selected_train.csv\")\n\n# 2. Load full training metadata\ntrain_df = pd.read_csv(train_csv_path)\nprint(f\"Total slides in train.csv: {len(train_df)}\")\nprint(train_df['data_provider'].value_counts())\n\n# 3. Filter to only Radboud provider\nradboud_df = train_df[train_df['data_provider'] == 'radboud'].copy()\nprint(f\"\\nTotal Radboud slides available: {len(radboud_df)}\")\n\n# 4. Randomly sample 150 (set random_state for reproducibility)\nN_SAMPLES = 150\nif len(radboud_df) < N_SAMPLES:\n    print(f\"⚠️ Only {len(radboud_df)} Radboud slides available, using all of them.\")\n    selected_train = radboud_df\nelse:\n    selected_train = radboud_df.sample(n=N_SAMPLES, random_state=42).reset_index(drop=True)\n\nprint(f\"\\n✅ Selected {len(selected_train)} Radboud slides.\")\nprint(selected_train['isup_grade'].value_counts().sort_index())  # sanity check on grade distribution\n\n# 5. Save to CSV for the copy step\nselected_train.to_csv(csv_path, index=False)\nprint(f\"\\nSaved selection to: {csv_path}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T13:20:36.547146Z","iopub.execute_input":"2026-08-06T13:20:36.547769Z","iopub.status.idle":"2026-08-06T13:20:36.586291Z","shell.execute_reply.started":"2026-08-06T13:20:36.547724Z","shell.execute_reply":"2026-08-06T13:20:36.585307Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport shutil\nimport pandas as pd\nfrom tqdm import tqdm\n\n# 1. Define Kaggle source paths and your working directory\nINPUT_DIR = \"/kaggle/input/competitions/prostate-cancer-grade-assessment\"\nWORKING_DIR = \"/kaggle/working/panda_selected_subset\"\ncsv_path = os.path.join(WORKING_DIR, \"selected_train.csv\")\n\n# Check if the CSV file exists before reading\nif not os.path.exists(csv_path):\n    print(f\"❌ File {csv_path} not found. Please double-check the CSV save path from the previous step!\")\nelse:\n    # --- SECTION: READ SELECTED_TRAIN ---\n    selected_train = pd.read_csv(csv_path)\n\n    # Use ALL 150 selected Radboud slides (no more .head(3))\n    df_to_download = selected_train.copy()\n\n    # Define and initialize the destination directories for train_images and train_label_masks\n    dst_img_dir = os.path.join(WORKING_DIR, \"train_images\")\n    dst_mask_dir = os.path.join(WORKING_DIR, \"train_label_masks\")\n    os.makedirs(dst_img_dir, exist_ok=True)\n    os.makedirs(dst_mask_dir, exist_ok=True)\n\n    print(f\" CSV file read successfully. Starting the physical copy of {len(df_to_download)} original pairs (Image + Mask)...\")\n\n    success_count = 0\n    missing_list = []\n\n    # 2. Perform the physical copy directly\n    for idx, row in tqdm(df_to_download.iterrows(), total=len(df_to_download)):\n        image_id = row['image_id']\n\n        # Source paths (Original dataset from Kaggle Input)\n        src_img = os.path.join(INPUT_DIR, \"train_images\", f\"{image_id}.tiff\")\n        src_mask = os.path.join(INPUT_DIR, \"train_label_masks\", f\"{image_id}_mask.tiff\")\n\n        # Destination paths (Working directory on the Notebook for download)\n        dst_img = os.path.join(dst_img_dir, f\"{image_id}.tiff\")\n        dst_mask = os.path.join(dst_mask_dir, f\"{image_id}_mask.tiff\")\n\n        # Copy only if the source files exist and the destination files do not already exist\n        if os.path.exists(src_img) and os.path.exists(src_mask):\n            if not os.path.exists(dst_img):\n                shutil.copy(src_img, dst_img)\n            if not os.path.exists(dst_mask):\n                shutil.copy(src_mask, dst_mask)\n            success_count += 1\n        else:\n            missing_list.append(image_id)\n\n    print(\"\\n---  FINISHED SAVING IMAGES AND MASKS TO FILTERED DIRECTORY ---\")\n    print(f\"Successfully copied: {success_count}/{len(df_to_download)} slides.\")\n    print(f\"Image location (.tiff): {dst_img_dir}\")\n    print(f\"Mask location (_mask.tiff): {dst_mask_dir}\")\n\n    if missing_list:\n        print(f\"\\n⚠️ {len(missing_list)} slides had missing image/mask files and were skipped:\")\n        print(missing_list[:10], \"...\" if len(missing_list) > 10 else \"\")\n\n    # 3. Check the total directory size to ensure safety\n    print(\"\\n Current size of the filtered directory on your Notebook:\")\n    !du -sh {WORKING_DIR}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T13:24:51.319509Z","iopub.execute_input":"2026-08-06T13:24:51.319923Z","iopub.status.idle":"2026-08-06T13:25:23.866417Z","shell.execute_reply.started":"2026-08-06T13:24:51.319889Z","shell.execute_reply":"2026-08-06T13:25:23.864907Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import os\n# import shutil\n# import pandas as pd\n# from tqdm import tqdm\n\n# # 1. Định nghĩa các đường dẫn gốc của Kaggle và thư mục làm việc của bạn\n# INPUT_DIR = \"/kaggle/input/competitions/prostate-cancer-grade-assessment\"\n# WORKING_DIR = \"/kaggle/working/panda_selected_subset\"  # Đảm bảo trùng với tên thư mục bạn lưu file CSV trước đó\n\n# csv_path = os.path.join(WORKING_DIR, \"selected_train.csv\")\n\n# # Kiểm tra xem file CSV có tồn tại không trước khi đọc\n# if not os.path.exists(csv_path):\n#     print(f\"❌ Không tìm thấy file {csv_path}. Bạn hãy kiểm tra lại đường dẫn lưu file CSV ở bước trước!\")\n# else:\n#     # --- PHẦN ĐỌC FILE SELECTED_TRAIN ---\n#     selected_train = pd.read_csv(csv_path)\n    \n#     # Giới hạn lấy đúng 350 slide từ danh sách đã đọc\n#     df_to_download = selected_train.head(350) \n    \n#     # Định nghĩa và khởi tạo thư mục đích train_images và train_label_masks\n#     dst_img_dir = os.path.join(WORKING_DIR, \"train_images\")\n#     dst_mask_dir = os.path.join(WORKING_DIR, \"train_label_masks\")\n#     os.makedirs(dst_img_dir, exist_ok=True)\n#     os.makedirs(dst_mask_dir, exist_ok=True)\n\n#     print(f\"🚀 Đã đọc file CSV thành công. Bắt đầu sao chép vật lý {len(df_to_download)} cặp (Ảnh + Mask) gốc...\")\n#     success_count = 0\n\n#     # 2. Tiến hành sao chép vật lý trực tiếp\n#     for idx, row in tqdm(df_to_download.iterrows(), total=len(df_to_download)):\n#         image_id = row['image_id']\n        \n#         # Đường dẫn nguồn (Dữ liệu gốc từ Kaggle Input)\n#         src_img = os.path.join(INPUT_DIR, \"train_images\", f\"{image_id}.tiff\")\n#         src_mask = os.path.join(INPUT_DIR, \"train_label_masks\", f\"{image_id}_mask.tiff\")\n        \n#         # Đường dẫn đích (Thư mục làm việc trên Notebook để tải về)\n#         dst_img = os.path.join(dst_img_dir, f\"{image_id}.tiff\")\n#         dst_mask = os.path.join(dst_mask_dir, f\"{image_id}_mask.tiff\")\n        \n#         # Sao chép nếu file nguồn tồn tại và file đích chưa có\n#         if os.path.exists(src_img) and os.path.exists(src_mask):\n#             if not os.path.exists(dst_img):\n#                 shutil.copy(src_img, dst_img)\n#             if not os.path.exists(dst_mask):\n#                 shutil.copy(src_mask, dst_mask)\n#             success_count += 1\n\n#     print(\"\\n--- 🎉 HOÀN THÀNH ĐƯA ẢNH VÀ MASK VÀO THƯ MỤC SÀNG LỌC ---\")\n#     print(f\"Đã sao chép thành công: {success_count}/{len(df_to_download)} slide.\")\n#     print(f\"Vị trí ảnh (.tiff): {dst_img_dir}\")\n#     print(f\"Vị trí mask (_mask.tiff): {dst_mask_dir}\")\n\n#     # 3. Kiểm tra tổng dung lượng thư mục hiện tại để đảm bảo an toàn\n#     print(\"\\n📊 Dung lượng thư mục đã sàng lọc trên Notebook của bạn:\")\n#     !du -sh {WORKING_DIR}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-21T16:28:56.734009Z","iopub.execute_input":"2026-07-21T16:28:56.734361Z","iopub.status.idle":"2026-07-21T16:28:56.740517Z","shell.execute_reply.started":"2026-07-21T16:28:56.734328Z","shell.execute_reply":"2026-07-21T16:28:56.73971Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1. Compress the entire panda_selected_subset directory into a single zip file outside the target folder\n!zip -r -q /kaggle/working/panda_filtered_data.zip /kaggle/working/panda_selected_subset\n\n# 2. Verify the actual file size of the zip file after compression\n!ls -lh /kaggle/working/panda_filtered_data.zip","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T13:26:52.750531Z","iopub.execute_input":"2026-08-06T13:26:52.753142Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # 1. Nén toàn bộ thư mục panda_selected_subset thành file zip duy nhất bên ngoài thư mục làm việc\n# !zip -r -q /kaggle/working/panda_filtered_data.zip /kaggle/working/panda_selected_subset\n\n# # 2. Kiểm tra lại dung lượng thực tế của file zip sau khi nén\n# !ls -lh /kaggle/working/panda_filtered_data.zip","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-21T16:28:59.674445Z","iopub.execute_input":"2026-07-21T16:28:59.675348Z","iopub.status.idle":"2026-07-21T16:28:59.679612Z","shell.execute_reply.started":"2026-07-21T16:28:59.67531Z","shell.execute_reply":"2026-07-21T16:28:59.678781Z"}},"outputs":[],"execution_count":null}]}