{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":13089871,"sourceType":"datasetVersion","datasetId":8291159},{"sourceId":13090191,"sourceType":"datasetVersion","datasetId":8291364},{"sourceId":13116672,"sourceType":"datasetVersion","datasetId":8309026},{"sourceId":13169780,"sourceType":"datasetVersion","datasetId":8345323}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!cp -r /kaggle/input/code-focus/FOCUS/ /kaggle/working/\nimport os\n!cd /kaggle/working","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-25T13:26:37.794635Z","iopub.execute_input":"2025-09-25T13:26:37.795476Z","iopub.status.idle":"2025-09-25T13:26:38.298237Z","shell.execute_reply.started":"2025-09-25T13:26:37.795432Z","shell.execute_reply":"2025-09-25T13:26:38.297189Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# !rm -rf /kaggle/working/FOCUS","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-25T12:12:11.190143Z","iopub.execute_input":"2025-09-25T12:12:11.190392Z","iopub.status.idle":"2025-09-25T12:12:11.194566Z","shell.execute_reply.started":"2025-09-25T12:12:11.190364Z","shell.execute_reply":"2025-09-25T12:12:11.193822Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!apt-get update -qq\n!apt-get install openjdk-11-jdk-headless -y\n\n!pip uninstall -y pyspark\n!pip install pyspark==3.5.1\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-25T12:12:11.195372Z","iopub.execute_input":"2025-09-25T12:12:11.195583Z","iopub.status.idle":"2025-09-25T12:13:23.714025Z","shell.execute_reply.started":"2025-09-25T12:12:11.195568Z","shell.execute_reply":"2025-09-25T12:13:23.713052Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # 5. Thiết lập biến môi trường\n# import os\n# os.environ[\"JAVA_HOME\"] = \"/usr/lib/jvm/java-11-openjdk-amd64\"\n# os.environ[\"SPARK_HOME\"] = \"/opt/spark\"\n# os.environ[\"PATH\"] += \":/opt/spark/bin\"\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-25T12:13:23.716386Z","iopub.execute_input":"2025-09-25T12:13:23.716643Z","iopub.status.idle":"2025-09-25T12:13:23.720964Z","shell.execute_reply.started":"2025-09-25T12:13:23.716617Z","shell.execute_reply":"2025-09-25T12:13:23.720378Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nos.environ[\"JAVA_HOME\"] = \"/usr/lib/jvm/java-11-openjdk-amd64\"\nif \"SPARK_HOME\" in os.environ:\n    del os.environ[\"SPARK_HOME\"]\n\nfrom pyspark.sql import SparkSession\n\nspark = SparkSession.builder \\\n    .appName(\"SparkTestNew\") \\\n    .master(\"local[*]\") \\\n    .getOrCreate()\n\nprint(\"Spark version:\", spark.version)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-25T12:13:23.721807Z","iopub.execute_input":"2025-09-25T12:13:23.72212Z","iopub.status.idle":"2025-09-25T12:13:27.755249Z","shell.execute_reply.started":"2025-09-25T12:13:23.722094Z","shell.execute_reply":"2025-09-25T12:13:27.754414Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Thêm cột case_id và sile_id: case_id: mã người, silde_id: mã ảnh (data không có)\nimport pandas as pd\nimport os\ncsv_src = \"/kaggle/input/ubc-ocean/UBC-OCEAN/train.csv\"\ncsv_fixed = \"/kaggle/working/FOCUS/train.csv\"\nif not os.path.exists(csv_fixed):  \n    df = pd.read_csv(csv_src)\n    df[\"case_id\"] = df[\"image_id\"].astype(str)\n    df[\"slide_id\"] = df[\"image_id\"].astype(str)\n    df.to_csv(csv_fixed, index=False)\nprint(\"✅ CSV fixed saved at:\", csv_fixed)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-25T12:13:27.756291Z","iopub.execute_input":"2025-09-25T12:13:27.756852Z","iopub.status.idle":"2025-09-25T12:13:28.175558Z","shell.execute_reply.started":"2025-09-25T12:13:27.756824Z","shell.execute_reply":"2025-09-25T12:13:28.174643Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Bắt buộc\n# !cp -r /kaggle/input/conch-source-code/CONCH-main/conch /kaggle/working/","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-25T12:13:28.176491Z","iopub.execute_input":"2025-09-25T12:13:28.17678Z","iopub.status.idle":"2025-09-25T12:13:28.18028Z","shell.execute_reply.started":"2025-09-25T12:13:28.176753Z","shell.execute_reply":"2025-09-25T12:13:28.17942Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Bắt buộc\n\n!mkdir -p /kaggle/working/FOCUS/ckpts\n!mkdir -p /kaggle/working/FOCUS/features\n# !mkdir -p /kaggle/working/FOCUS/features/features.csv\n\n!cp /kaggle/input/conch-ckpts/conch.pth /kaggle/working/FOCUS/ckpts/conch.pth\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-25T12:13:28.18117Z","iopub.execute_input":"2025-09-25T12:13:28.181444Z","iopub.status.idle":"2025-09-25T12:13:34.236271Z","shell.execute_reply.started":"2025-09-25T12:13:28.181414Z","shell.execute_reply":"2025-09-25T12:13:34.235388Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# #Trích xuất đặc trưng ảnh\n!cp -rn /kaggle/input/code-focus/FOCUS/ /kaggle/working/\nproject_dir = '/kaggle/working/FOCUS'\nos.chdir(project_dir)\n!mkdir -p results/FOCUS/conch/\n\n# !apt-get install -y openslide-tools\n# !pip install openslide-python\n\n\n# !python /kaggle/working/FOCUS/map_reduce_features.py \\\n#   --csv_path /kaggle/working/FOCUS/train.csv \\\n#   --source_folder /kaggle/input/ubc-ocean/UBC-OCEAN/train_thumbnails \\\n#   --output_csv /kaggle/working/FOCUS/features/features.csv \\\n#   --features_dir /kaggle/working/FOCUS/features \\\n#   --ckpt_path /kaggle/working/FOCUS/ckpts/conch.pth \\\n#   2>&1 | tee spark_output.log\n\n!spark-submit \\\n  --master local[*] \\\n  --driver-memory 8g \\\n  --executor-memory 6g \\\n  /kaggle/working/FOCUS/map_reduce_features.py \\\n  --csv_path /kaggle/working/FOCUS/train.csv \\\n  --source_folder /kaggle/input/ubc-ocean/UBC-OCEAN/train_thumbnails \\\n  --output_csv /kaggle/working/FOCUS/features/features.csv \\\n  --features_dir /kaggle/working/FOCUS/features \\\n  --ckpt_path /kaggle/working/FOCUS/ckpts/conch.pth \\\n  2>&1 | tee spark_output.log\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-25T12:13:34.23751Z","iopub.execute_input":"2025-09-25T12:13:34.237757Z","iopub.status.idle":"2025-09-25T13:00:07.959186Z","shell.execute_reply.started":"2025-09-25T12:13:34.237732Z","shell.execute_reply":"2025-09-25T13:00:07.958383Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# #Thay đổi (tạo file csv mới chỉ chứa các id đã trích xuất)\n\n# import pandas as pd\n# from sklearn.model_selection import StratifiedKFold, train_test_split\n\n# # --- 1. CẤU HÌNH ---\n# # Đường dẫn đến file CSV chứa danh sách các ảnh đã được trích xuất đặc trưng\n# # (Nếu bạn đã tạo train_cleaned.csv thì dùng nó, nếu không thì dùng train.csv gốc)\n# csv_path = '/kaggle/input/UBC-OCEAN/train.csv'\n\n# # Thư mục để lưu 8 file split CSV\n# output_dir = '/kaggle/working/FOCUS/splits/UBC-OCEAN_debug_8folds/'\n\n# # Số lượng fold cần tạo\n# n_splits = 8\n\n# # Tỉ lệ dữ liệu dành cho tập validation (ví dụ: 15% từ tập train)\n# val_size = 0.15 \n\n# # --- 2. TẠO THƯ MỤC LƯU TRỮ ---\n# os.makedirs(output_dir, exist_ok=True)\n# print(f\"Các file split sẽ được lưu tại: {output_dir}\")\n\n# # --- 3. ĐỌC VÀ CHUẨN BỊ DỮ LIỆU ---\n# df = pd.read_csv(csv_path)\n\n# # Lọc ra danh sách các ảnh thực sự tồn tại (để chắc chắn)\n# existing_features_dir = '/kaggle/working/FOCUS/features/'\n# if os.path.exists(existing_features_dir):\n#     existing_ids = {int(f.split('.')[0]) for f in os.listdir(existing_features_dir)}\n#     df = df[df['image_id'].isin(existing_ids)].reset_index(drop=True)\n#     print(f\"Đã lọc, chỉ sử dụng {len(df)} ảnh có file đặc trưng tồn tại.\")\n\n# # Lấy ra slide_id và nhãn\n# slide_ids = df['image_id']\n# labels = df['label']\n\n# # --- 4. THỰC HIỆN CHIA DỮ LIỆU ---\n# skf = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=42)\n\n# # Lặp qua 8 fold\n# for i, (train_val_indices, test_indices) in enumerate(skf.split(slide_ids, labels)):\n    \n#     # Lấy ra tập train+val và tập test cho fold hiện tại\n#     train_val_ids = slide_ids.iloc[train_val_indices]\n#     train_val_labels = labels.iloc[train_val_indices]\n#     test_ids = slide_ids.iloc[test_indices]\n    \n#     # Tiếp tục chia tập train+val thành tập train và tập val\n#     train_ids, val_ids = train_test_split(train_val_ids, \n#                                           test_size=val_size, \n#                                           stratify=train_val_labels, \n#                                           random_state=42)\n    \n#     # Tạo DataFrame theo đúng định dạng yêu cầu\n#     # Dùng pd.Series để xử lý các list có độ dài khác nhau\n#     split_df = pd.DataFrame({\n#         'train': pd.Series(train_ids.tolist()),\n#         'val': pd.Series(val_ids.tolist()),\n#         'test': pd.Series(test_ids.tolist())\n#     })\n    \n#     # Lưu file CSV\n#     output_filename = os.path.join(output_dir, f'splits_{i}.csv')\n#     split_df.to_csv(output_filename, index=False)\n    \n#     print(f\"Đã tạo file: splits_{i}.csv (Train: {len(train_ids)}, Val: {len(val_ids)}, Test: {len(test_ids)})\")\n\n# print(f\"\\n--- Hoàn tất! Đã tạo thành công {n_splits} file split. ---\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-25T13:00:07.961657Z","iopub.execute_input":"2025-09-25T13:00:07.96188Z","iopub.status.idle":"2025-09-25T13:00:07.967286Z","shell.execute_reply.started":"2025-09-25T13:00:07.961858Z","shell.execute_reply":"2025-09-25T13:00:07.96636Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\n\nfeat = torch.load(\"/kaggle/working/FOCUS/features/10077.pt\")\nprint(type(feat))\nprint(feat.shape if isinstance(feat, torch.Tensor) else \"not tensor\")\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-25T13:00:07.96834Z","iopub.execute_input":"2025-09-25T13:00:07.96859Z","iopub.status.idle":"2025-09-25T13:00:10.391224Z","shell.execute_reply.started":"2025-09-25T13:00:07.968562Z","shell.execute_reply":"2025-09-25T13:00:10.390401Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import subprocess\nfrom joblib import Parallel, delayed\nimport glob\nimport pandas as pd\nimport torch\n\n\nos.chdir('/kaggle/working/FOCUS/')\n\n\n!pip install --quiet tensorboardX openai-clip faiss-cpu\n\n\nCORRECT_CSV_PATH = \"/kaggle/working/FOCUS/train.csv\"\n\n!sed -i -e \"s|csv_path = '.*'|csv_path = '{CORRECT_CSV_PATH}'|g\" main.py\nprint(f\"Đã cập nhật đường dẫn trong main.py thành: {CORRECT_CSV_PATH}\")\n\nprint(\"Cập nhật đường dẫn thành công.\")\n\n# ===================== MAP =====================\n# def run_fold(fold_id, gpu_id, total_folds):\n#     \"\"\"\n#     Hàm này chạy một fold duy nhất trên một GPU cụ thể.\n#     \"\"\"\n#     log_dir = \"/kaggle/working/FOCUS/results/FOCUS/conch\"\n#     os.makedirs(log_dir, exist_ok=True)\n#     log_file = os.path.join(log_dir, f\"fold_{fold_id}.log\")\n    \n#     env = os.environ.copy()\n#     env[\"CUDA_VISIBLE_DEVICES\"] = str(gpu_id)\n\n#     # === BỔ SUNG k, k_start, k_end VÀO ĐÂY ===\n#     cmd = [\n#         \"python\", \"main.py\",\n#         \"--k\", str(total_folds),         # Tổng số fold của toàn bộ thí nghiệm\n#         \"--k_start\", str(fold_id),       # Fold bắt đầu (chính là fold hiện tại)\n#         \"--k_end\", str(fold_id + 1),     # Fold kết thúc (chỉ chạy 1 fold)\n#         # Các tham số còn lại giữ nguyên\n#         \"--seed\", \"1\",\n#         \"--drop_out\",\n#         \"--early_stopping\",\n#         \"--lr\", \"1e-4\",\n#         \"--label_frac\", \"1\",\n#         \"--bag_loss\", \"ce\",\n#         \"--task\", \"task_UBC-OCEAN_subtyping\",\n#         \"--results_dir\", \"results/FOCUS/conch/\",\n#         \"--exp_code\", \"UBC-OCEAN_4shots_10folds\",\n#         \"--model_type\", \"FOCUS\",   #ViLa_MIL\n#         \"--mode\", \"transformer\",\n#         \"--max_epochs\", \"4\",\n#         \"--log_data\",\n#         \"--data_root_dir\", \"/kaggle/input/ubc-ovarian-cancer-subtype-classification/\",\n#         \"--data_folder_s\", \"/kaggle/working/FOCUS/features/\",\n#         \"--data_folder_l\", \"/kaggle/working/FOCUS/features/\",\n#         \"--split_dir\", \"UBC-OCEAN_4shots_10folds\",\n#         \"--text_prompt_path\", \"text_prompt/UBC-OCEAN_two_scale_text_prompt.csv\",\n#     ]\n\n#     print(f\"Bắt đầu chạy Fold {fold_id} trên GPU {gpu_id}...\")\n#     with open(log_file, \"w\") as f:\n#         subprocess.run(cmd, env=env, stdout=f, stderr=subprocess.STDOUT)\n#     print(f\"Hoàn thành Fold {fold_id}.\")\n#     return log_file\n\ndef run_fold(fold_id, gpu_id, total_folds):\n\n    import sys\n    \n    log_dir = \"/kaggle/working/FOCUS/results/FOCUS/conch\"\n    os.makedirs(log_dir, exist_ok=True)\n    log_file = os.path.join(log_dir, f\"fold_{fold_id}.log\")\n    \n    env = os.environ.copy()\n    env[\"CUDA_VISIBLE_DEVICES\"] = str(gpu_id)\n\n    cmd = [\n        \"python\", \"main.py\",\n        \"--k\", str(total_folds),\n        \"--k_start\", str(fold_id),\n        \"--k_end\", str(fold_id + 1),\n        \"--seed\", \"1\",\n        \"--drop_out\",\n        \"--early_stopping\",\n        \"--lr\", \"1e-4\",\n        \"--label_frac\", \"1\",\n        \"--bag_loss\", \"ce\",\n        \"--task\", \"task_UBC-OCEAN_subtyping\",\n        \"--results_dir\", \"results/FOCUS/conch/\",\n        \"--exp_code\", \"UBC-OCEAN_16shots_10folds\",\n        \"--model_type\", \"FOCUS\",\n        \"--mode\", \"transformer\",\n        \"--max_epochs\", \"4\",\n        \"--log_data\",\n        \"--data_root_dir\", \"/kaggle/input/ubc-ovarian-cancer-subtype-classification/\",\n        \"--data_folder_s\", \"/kaggle/working/FOCUS/features/\",\n        \"--data_folder_l\", \"/kaggle/working/FOCUS/features/\",\n        \"--split_dir\", \"UBC-OCEAN_16shots_10folds\",\n        \"--text_prompt_path\", \"text_prompt/UBC-OCEAN_two_scale_text_prompt.csv\",\n    ]\n\n    print(f\"🚀 Bắt đầu chạy Fold {fold_id} trên GPU {gpu_id}...\")\n    with open(log_file, \"w\") as f:\n        process = subprocess.Popen(\n            cmd,\n            env=env,\n            stdout=subprocess.PIPE,\n            stderr=subprocess.STDOUT,\n            text=True,\n            bufsize=1\n        )\n        # Đọc từng dòng output\n        for line in process.stdout:\n            sys.stdout.write(f\"[Fold {fold_id} | GPU {gpu_id}] {line}\")\n            sys.stdout.flush()\n            f.write(line)\n        process.wait()\n\n    print(f\"✅ Hoàn thành Fold {fold_id}, log lưu tại {log_file}\")\n    return log_file\n\n\ndef reduce_results(results_dir):\n\n\n    result_files = glob.glob(os.path.join(results_dir, \"result_partial_*.csv\"))\n    print(f\"\\n🔎 Giai đoạn Reduce: Tìm thấy {len(result_files)} file kết quả của các fold.\")\n\n    df_all = pd.concat([pd.read_csv(f) for f in result_files], ignore_index=True)\n    summary = df_all.drop(columns=['folds']).agg(['mean', 'std']).T\n    summary.reset_index(inplace=True)\n    summary.rename(columns={'index': 'metric'}, inplace=True)\n    \n    output_file = os.path.join(results_dir, \"summary.csv\")\n    summary.to_csv(output_file, index=False)\n    print(f\"✅ Đã lưu summary cuối cùng vào {output_file}\")\n    return summary\n\nif __name__ == \"__main__\":\n    num_gpus = torch.cuda.device_count()\n    # Danh sách công việc: (fold_id, gpu_id)\n    jobs = [(0, 0), (1, 1), (2, 0), (3, 1)] \n    total_folds_in_experiment = 2 # Tổng số fold bạn muốn chạy\n\n    print(f\"\\n--- Bắt đầu giai đoạn MAP: Chạy {len(jobs)} fold song song trên {num_gpus} GPU... ---\")\n    logs = Parallel(n_jobs=num_gpus)(delayed(run_fold)(fid, gid, total_folds_in_experiment) for fid, gid in jobs)\n    print(\"✅ Hoàn thành giai đoạn MAP.\")\n    \n    results_dir = \"/kaggle/working/FOCUS/results/FOCUS/conch/UBC-OCEAN_16shots_10folds\"\n    summary_df = reduce_results(results_dir)\n\n    print(\"\\n--- KẾT QUẢ CUỐI CÙNG ---\")\n    if summary_df is not None:\n        print(summary_df.to_string())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-25T13:26:48.317235Z","iopub.execute_input":"2025-09-25T13:26:48.317934Z"}},"outputs":[],"execution_count":null}]}