{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":13451,"datasetId":654585,"databundleVersionId":1188070},{"sourceType":"datasetVersion","sourceId":16210215,"datasetId":10370975,"databundleVersionId":17190086}],"dockerImageVersionId":31328,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# base code","metadata":{}},{"cell_type":"code","source":"%cd /kaggle/working\n!rm -rf RSNA2019_Intracranial-Hemorrhage-Detection\n\n!git clone https://github.com/SeuTao/RSNA2019_Intracranial-Hemorrhage-Detection.git","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-05-11T14:22:47.568126Z","iopub.execute_input":"2026-05-11T14:22:47.568746Z","iopub.status.idle":"2026-05-11T14:22:51.418119Z","shell.execute_reply.started":"2026-05-11T14:22:47.568717Z","shell.execute_reply":"2026-05-11T14:22:51.417052Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%cd RSNA2019_Intracranial-Hemorrhage-Detection","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-11T14:18:11.526175Z","iopub.execute_input":"2026-05-11T14:18:11.526495Z","iopub.status.idle":"2026-05-11T14:18:11.534117Z","shell.execute_reply.started":"2026-05-11T14:18:11.526459Z","shell.execute_reply":"2026-05-11T14:18:11.533317Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!ls -l","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-11T14:18:11.535301Z","iopub.execute_input":"2026-05-11T14:18:11.535632Z","iopub.status.idle":"2026-05-11T14:18:12.003236Z","shell.execute_reply.started":"2026-05-11T14:18:11.53561Z","shell.execute_reply":"2026-05-11T14:18:12.002439Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nreq_content = \"\"\"\nnumpy\npandas\ntorch\ntorchvision\nalbumentations>=1.4.0\ntqdm\nopencv-contrib-python\npathlib\nscikit-image\njoblib\npydicom\npylibjpeg\n\"\"\"\n\nwith open(\"requirements.txt\", \"w\") as f:\n    f.write(req_content)\n\nprint(\"Đã tạo lại file requirements.txt mới!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-11T14:18:12.006339Z","iopub.execute_input":"2026-05-11T14:18:12.006582Z","iopub.status.idle":"2026-05-11T14:18:12.012617Z","shell.execute_reply.started":"2026-05-11T14:18:12.006556Z","shell.execute_reply":"2026-05-11T14:18:12.011712Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install --upgrade pip setuptools wheel\n!pip install -r requirements.txt","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-11T14:18:12.013606Z","iopub.execute_input":"2026-05-11T14:18:12.014336Z","iopub.status.idle":"2026-05-11T14:18:18.077727Z","shell.execute_reply.started":"2026-05-11T14:18:12.014313Z","shell.execute_reply":"2026-05-11T14:18:18.076964Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Copy file cần thiết từ input vào mục working","metadata":{}},{"cell_type":"code","source":"import os, glob, shutil\n\nDATASET_ROOT = \"/kaggle/input/datasets/ren4gg/rsna2019\"\n\ndst_base = \"/kaggle/working/features/stage2_finetune\"\nos.makedirs(dst_base, exist_ok=True)\n\nprob_csvs = glob.glob(os.path.join(DATASET_ROOT, \"**\", \"stage2_finetune\", \"*\", \"*_prob_*.csv\"), recursive=True)\n\nprint(\"Found prob csv:\", len(prob_csvs))\nif len(prob_csvs) == 0:\n    raise RuntimeError(\"Không tìm thấy *_prob_*.csv trong dataset. Kiểm tra lại DATASET_ROOT.\")\n\nfor src in prob_csvs:\n    model_name = os.path.basename(os.path.dirname(src))  # st_se101_256_fine, dsn121_512_fine\n    dst_dir = os.path.join(dst_base, model_name)\n    os.makedirs(dst_dir, exist_ok=True)\n    dst = os.path.join(dst_dir, os.path.basename(src))\n    shutil.copy(src, dst)\n    print(\"Copied:\", src, \"->\", dst)\n\nprint(\"\\nDone. Now list /kaggle/working/features/stage2_finetune:\")\nfor m in sorted(os.listdir(dst_base)):\n    print(\" -\", m, \":\", os.listdir(os.path.join(dst_base, m)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-11T14:22:55.064355Z","iopub.execute_input":"2026-05-11T14:22:55.064839Z","iopub.status.idle":"2026-05-11T14:23:17.33265Z","shell.execute_reply.started":"2026-05-11T14:22:55.064803Z","shell.execute_reply":"2026-05-11T14:23:17.331684Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Setup**","metadata":{}},{"cell_type":"code","source":"# =========================\n# RSNA2019 Master Cell (Kaggle) \n# =========================\nimport os, sys, shutil, subprocess, textwrap\nfrom pathlib import Path\n\n# --------- USER CONFIG ----------\nDATASET_ROOT = \"/kaggle/input/datasets/ren4gg/rsna2019\"  \nREPO_URL = \"https://github.com/SeuTao/RSNA2019_Intracranial-Hemorrhage-Detection.git\"\nREPO_DIR = \"/kaggle/working/RSNA2019_Intracranial-Hemorrhage-Detection\"\n\nRUN_SEQUENCE_MAIN = False  \n# --------------------------------\n\ndef sh(cmd):\n    print(f\"\\n$ {cmd}\")\n    r = subprocess.run(cmd, shell=True, text=True, capture_output=True)\n    if r.stdout.strip(): print(r.stdout)\n    if r.returncode != 0:\n        if r.stderr.strip(): print(r.stderr)\n        raise RuntimeError(f\"Command failed ({r.returncode}): {cmd}\")\n    return r.stdout\n\n# 0) Basic checks + folders\nassert os.path.exists(DATASET_ROOT), f\"DATASET_ROOT not found: {DATASET_ROOT}\"\nos.makedirs(\"/kaggle/working/csv\", exist_ok=True)\nos.makedirs(\"/kaggle/working/features\", exist_ok=True)\nos.makedirs(\"/kaggle/working/FinalSubmission\", exist_ok=True)\n\n# 1) Clone repo\nif not os.path.exists(REPO_DIR):\n    sh(f\"git clone --depth 1 {REPO_URL} {REPO_DIR}\")\nelse:\n    print(f\"Repo already exists: {REPO_DIR}\")\n\n# 2) requirements.txt + pip install\nreq_content = \"\"\"\nnumpy\npandas\ntorch\ntorchvision\nalbumentations>=1.4.0\ntqdm\nopencv-contrib-python\npathlib\nscikit-image\njoblib\npydicom\npylibjpeg\n\"\"\".strip() + \"\\n\"\n\n\nreq_path = \"/kaggle/working/requirements_rsna2019.txt\"\nPath(req_path).write_text(req_content)\nprint(f\"Wrote requirements to: {req_path}\\n---\\n{req_content}---\")\n\n# Upgrade tooling \nsh(\"pip -q install --upgrade pip setuptools wheel\")\nsh(f\"pip -q install -r {req_path}\")\n\n# 3) Scan dataset \nall_files = []\nfor root, dirs, files in os.walk(DATASET_ROOT):\n    for f in files:\n        all_files.append(os.path.join(root, f))\n\nby_base = {}\nfor f in all_files:\n    by_base.setdefault(os.path.basename(f), []).append(f)\n\n# 4) Copy CSVs toiws /kaggle/working/csv\ncsv_candidates = [f for f in all_files if f.lower().endswith(\".csv\")]\nif not csv_candidates:\n    raise RuntimeError(\"No CSV files found in dataset. Please check dataset contents.\")\nprint(f\"Found {len(csv_candidates)} CSV files; copying to /kaggle/working/csv ...\")\n\nfor f in csv_candidates:\n    dst = os.path.join(\"/kaggle/working/csv\", os.path.basename(f))\n    if not os.path.exists(dst):\n        shutil.copy(f, dst)\n\n# 5) Copy study_csv dir (if exists)\nstudy_csv_src = None\nfor root, dirs, files in os.walk(DATASET_ROOT):\n    if \"study_csv\" in dirs:\n        study_csv_src = os.path.join(root, \"study_csv\")\n        break\n\ndst_study_csv = \"/kaggle/working/csv/study_csv\"\nos.makedirs(dst_study_csv, exist_ok=True)\n\nif study_csv_src:\n    print(f\"Copying study_csv: {study_csv_src} -> {dst_study_csv}\")\n    sh(f\"cp -r {study_csv_src}/* {dst_study_csv}/\")\n    print(f\"study_csv files: {len(os.listdir(dst_study_csv))}\")\nelse:\n    print(\"WARNING: study_csv directory not found in dataset. Sequence model may fail if it expects it.\")\n\n# 6) Prepare feature dirs + copy/rename npy \nstage2_dir = \"/kaggle/working/features/stage2_finetune\"\nos.makedirs(stage2_dir, exist_ok=True)\n\nsubfolders = [\"dsn121_512_fine\", \"st_se101_256_fine\"]\nfor sf in subfolders:\n    os.makedirs(os.path.join(stage2_dir, sf), exist_ok=True)\n\nnpy_mapping = [\n    (\"dsn121_512_fine_test_feature_TTA_stage2_finetune.npy\", \"dsn121_512_fine\", \"dsn121_512_fine_test_oof_feature_TTA.npy\"),\n    (\"dsn121_512_fine_val_oof_feature_TTA_stage2_finetune.npy\",  \"dsn121_512_fine\", \"dsn121_512_fine_val_oof_feature_TTA.npy\"),\n    (\"st_se101_256_fine_test_feature_TTA_stage2_finetune.npy\",   \"st_se101_256_fine\", \"st_se101_256_fine_test_oof_feature_TTA.npy\"),\n    (\"st_se101_256_fine_val_oof_feature_TTA_stage2_finetune.npy\",\"st_se101_256_fine\", \"st_se101_256_fine_val_oof_feature_TTA.npy\"),\n]\n\ncopied = 0\nmissing = []\nfor old_name, subfolder, new_name in npy_mapping:\n    src_list = by_base.get(old_name, [])\n    if not src_list:\n        missing.append(old_name)\n        continue\n    src = src_list[0]\n    dst = os.path.join(stage2_dir, subfolder, new_name)\n    shutil.copy(src, dst)\n    print(f\"OK: {old_name} -> {dst}\")\n    copied += 1\n\nif missing:\n    print(\"\\nWARNING: missing .npy files (not found in dataset):\")\n    for m in missing:\n        print(\" -\", m)\n\n# 7) (Optional) copy 2D checkpoint into repo\n# search for model_epoch_best_*.pth.\nmodel_ckpt = None\nfor base in [\"model_epoch_best_4.pth\", \"model_epoch_best.pth\"]:\n    if base in by_base:\n        model_ckpt = by_base[base][0]\n        break\n\nif model_ckpt:\n    dst_2d_model = os.path.join(REPO_DIR, \"2DNet\", \"model_save_dir\")\n    os.makedirs(dst_2d_model, exist_ok=True)\n    shutil.copy(model_ckpt, os.path.join(dst_2d_model, \"seresnext101_best.pth\"))\n    print(f\"2D model copied: {model_ckpt} -> {dst_2d_model}/seresnext101_best.pth\")\nelse:\n    print(\"NOTE: No model_epoch_best_*.pth found in dataset.\")\n\n# 8)  SequenceModel settings \nseq_dir = os.path.join(REPO_DIR, \"SequenceModel\")\ncandidate_files = [\n    os.path.join(seq_dir, \"settings.py\"),\n    os.path.join(seq_dir, \"setting.py\"),\n]\ntarget_settings = None\nfor c in candidate_files:\n    if os.path.exists(c):\n        target_settings = c\n        break\nif target_settings is None:\n    target_settings = os.path.join(seq_dir, \"settings.py\")  # create it\n\nsettings_content = textwrap.dedent(\"\"\"\\\n    csv_root = r'/kaggle/working/csv'\n    feature_path = r'/kaggle/working/features'\n    final_output_path = r'/kaggle/working/FinalSubmission'\n\"\"\")\nPath(target_settings).write_text(settings_content)\nprint(f\"\\nWrote SequenceModel settings to: {target_settings}\\n---\\n{settings_content}---\")\n\n# 9) Sanity checks\nprint(\"\\nSanity check:\")\nprint(\"csv_root sample:\", sorted(os.listdir(\"/kaggle/working/csv\"))[:15])\nprint(\"features:\", os.listdir(\"/kaggle/working/features\"))\nprint(\"FinalSubmission exists:\", os.path.exists(\"/kaggle/working/FinalSubmission\"))\n\n# 10) (Optional) run SequenceModel/main.py\nif RUN_SEQUENCE_MAIN:\n    sh(f\"cd {seq_dir} && python main.py\")\n    print(\"Done. Check /kaggle/working/FinalSubmission for submission csv.\")\nelse:\n    print(\"\\nRUN_SEQUENCE_MAIN=False.\")\n    print(\"To run sequence model manually:\")\n    print(f\"cd {seq_dir} && python main.py\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-11T14:18:18.079258Z","iopub.execute_input":"2026-05-11T14:18:18.079592Z","iopub.status.idle":"2026-05-11T14:20:42.082036Z","shell.execute_reply.started":"2026-05-11T14:18:18.079554Z","shell.execute_reply":"2026-05-11T14:20:42.081183Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os, shutil\n\nBASE = \"/kaggle/working/features/stage2_finetune\"\n\npairs = [\n    (\"st_se101_256_fine\", \"st_se101_256_fine_test_oof_feature_TTA.npy\", \"st_se101_256_fine_test_feature_TTA_stage2_finetune.npy\"),\n    (\"dsn121_512_fine\",   \"dsn121_512_fine_test_oof_feature_TTA.npy\",   \"dsn121_512_fine_test_feature_TTA_stage2_finetune.npy\"),\n]\n\nfor model, src_name, dst_name in pairs:\n    src = os.path.join(BASE, model, src_name)\n    dst = os.path.join(BASE, model, dst_name)\n    if os.path.exists(src) and not os.path.exists(dst):\n        shutil.copy(src, dst)\n        print(\"Copied:\", src_name, \"->\", dst_name)\n    else:\n        print(\"Skip (missing src or dst exists):\", src, dst)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-11T14:20:42.083199Z","iopub.execute_input":"2026-05-11T14:20:42.08353Z","iopub.status.idle":"2026-05-11T14:20:42.089756Z","shell.execute_reply.started":"2026-05-11T14:20:42.083497Z","shell.execute_reply":"2026-05-11T14:20:42.08915Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\n\nBASE = \"/kaggle/working/features/stage2_finetune\"\nmodels = [\"st_se101_256_fine\", \"dsn121_512_fine\"]\n\ndef fix_filename(csv_path):\n    df = pd.read_csv(csv_path)\n    df[\"filename\"] = df[\"filename\"].astype(str).str.replace(\".dcm\", \"\", regex=False).str.replace(\".png\", \"\", regex=False)\n    df.to_csv(csv_path, index=False)\n\nfor m in models:\n    d = os.path.join(BASE, m)\n\n    files = [\n        os.path.join(d, f\"{m}_val_prob_TTA_stage2_finetune.csv\"),\n        os.path.join(d, f\"{m}_test_prob_TTA_stage2_finetune.csv\"),\n    ]\n\n    for f in files:\n        if os.path.exists(f):\n            print(\"Fixing:\", f)\n            fix_filename(f)\n        else:\n            print(\"Missing:\", f)\n\nprint(\"Done.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-11T14:20:42.091671Z","iopub.execute_input":"2026-05-11T14:20:42.091873Z","iopub.status.idle":"2026-05-11T14:20:55.25486Z","shell.execute_reply.started":"2026-05-11T14:20:42.091853Z","shell.execute_reply":"2026-05-11T14:20:55.253984Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd, os\n\ncsv_root = \"/kaggle/working/csv\"\nfeat_root = \"/kaggle/working/features/stage2_finetune\"\nstd_test = pd.read_csv(f\"{csv_root}/standard_test.csv\")\n\nfor model in os.listdir(feat_root):\n    d = os.path.join(feat_root, model)\n    test_path = os.path.join(d, f\"{model}_test_prob_TTA_stage2_finetune.csv\")\n    pred = pd.read_csv(test_path)\n    merged = std_test.merge(pred, how=\"left\", on=\"filename\", indicator=True)\n    miss = (merged[\"_merge\"] == \"left_only\").mean()\n    print(model, \"miss-rate:\", miss)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-11T14:20:55.255968Z","iopub.execute_input":"2026-05-11T14:20:55.256516Z","iopub.status.idle":"2026-05-11T14:20:55.760094Z","shell.execute_reply.started":"2026-05-11T14:20:55.256489Z","shell.execute_reply":"2026-05-11T14:20:55.759248Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd, os\n\ncsv_root = \"/kaggle/working/csv\"\nfeat_root = \"/kaggle/working/features/stage2_finetune\"\n\nstd_test = pd.read_csv(f\"{csv_root}/standard_test.csv\")\nprint(\"standard_test rows:\", len(std_test))\nprint(\"standard_test columns:\", std_test.columns.tolist()[:20])\n\nfor model in os.listdir(feat_root):\n    d = os.path.join(feat_root, model)\n    test_path = os.path.join(d, f\"{model}_test_prob_TTA_stage2_finetune.csv\")\n    if not os.path.exists(test_path):\n        print(\"\\n[Missing test prob]:\", test_path)\n        continue\n\n    pred = pd.read_csv(test_path)\n    print(\"\\nModel:\", model)\n    print(\"test_prob rows:\", len(pred), \"cols:\", pred.columns.tolist()[:20])\n\n    # check common key\n    if \"filename\" not in pred.columns:\n        print(\"  ERROR: no filename column in test_prob\")\n        continue\n\n    merged = std_test.merge(pred, how=\"left\", on=\"filename\", indicator=True)\n    miss = (merged[\"_merge\"] == \"left_only\").mean()\n    print(\"  merge miss-rate:\", miss)\n    \n    if miss > 0:\n        print(\"  examples missing:\", merged.loc[merged[\"_merge\"]==\"left_only\", \"filename\"].head(5).tolist())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-11T14:20:56.252242Z","iopub.execute_input":"2026-05-11T14:20:56.252705Z","iopub.status.idle":"2026-05-11T14:20:56.85001Z","shell.execute_reply.started":"2026-05-11T14:20:56.252671Z","shell.execute_reply":"2026-05-11T14:20:56.849285Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Sequence Model run**","metadata":{}},{"cell_type":"code","source":"%cd /kaggle/working/RSNA2019_Intracranial-Hemorrhage-Detection/SequenceModel\n!python main.py","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-11T14:20:55.761247Z","iopub.execute_input":"2026-05-11T14:20:55.761567Z","iopub.status.idle":"2026-05-11T14:20:55.767038Z","shell.execute_reply.started":"2026-05-11T14:20:55.761545Z","shell.execute_reply":"2026-05-11T14:20:55.766366Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ls -R /kaggle/working/features/stage2_finetune","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-11T14:20:55.768121Z","iopub.execute_input":"2026-05-11T14:20:55.768673Z","iopub.status.idle":"2026-05-11T14:20:56.250796Z","shell.execute_reply.started":"2026-05-11T14:20:55.768637Z","shell.execute_reply":"2026-05-11T14:20:56.250056Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\np = \"/kaggle/working/features/stage2_finetune/st_se101_256_fine/st_se101_256_fine_test_prob_TTA_stage2_finetune.csv\"\ndf = pd.read_csv(p)\nprint(df[\"filename\"].head(10).tolist())\nprint(df[\"filename\"].tail(10).tolist())\nprint(\"unique count:\", df[\"filename\"].nunique(), \"rows:\", len(df))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-11T14:20:56.853129Z","iopub.execute_input":"2026-05-11T14:20:56.853453Z","iopub.status.idle":"2026-05-11T14:20:56.996455Z","shell.execute_reply.started":"2026-05-11T14:20:56.853429Z","shell.execute_reply":"2026-05-11T14:20:56.995712Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\ns = pd.read_csv(\"/kaggle/working/csv/standard_test.csv\")\nprint(s[\"filename\"].head(10).tolist())\nprint(s[\"filename\"].tail(10).tolist())\nprint(\"unique count:\", s[\"filename\"].nunique(), \"rows:\", len(s))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-11T14:20:56.997361Z","iopub.execute_input":"2026-05-11T14:20:56.997676Z","iopub.status.idle":"2026-05-11T14:20:57.111159Z","shell.execute_reply.started":"2026-05-11T14:20:56.997643Z","shell.execute_reply":"2026-05-11T14:20:57.110375Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!cp /kaggle/input/datasets/ren4gg/rsna2019/fold_4/fold_4.pt /kaggle/working/RSNA2019_Intracranial-Hemorrhage-Detection/SequenceModel/fold_4.pt","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-11T14:20:57.112272Z","iopub.execute_input":"2026-05-11T14:20:57.112564Z","iopub.status.idle":"2026-05-11T14:20:57.770639Z","shell.execute_reply.started":"2026-05-11T14:20:57.112534Z","shell.execute_reply":"2026-05-11T14:20:57.769828Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os, sys, importlib, torch\nfrom torch.utils.data import DataLoader\n\nSEQ_DIR = \"/kaggle/working/RSNA2019_Intracranial-Hemorrhage-Detection/SequenceModel\"\nos.chdir(SEQ_DIR)\nsys.path.insert(0, SEQ_DIR)\n\nimport check_oof, seq_dataset, main\n\n# reload lấy state mới\nimportlib.reload(check_oof)\nimportlib.reload(seq_dataset)\nimportlib.reload(main)\n\n# set các global cần cho inference\nmain.fold_index = -1\nmain.fold_num = 5\nmain.Add_position = True\nmain.lstm_layers = 2\nmain.seq_len = 24\nmain.hidden = 96\nmain.drop_out = 0.5\nmain.train_epoch = 40\nmain.class_num = 6\n\norig_DataLoader = main.DataLoader\ndef DataLoader_no_workers(*args, **kwargs):\n    kwargs[\"num_workers\"] = 0\n    kwargs[\"pin_memory\"] = False\n    return orig_DataLoader(*args, **kwargs)\nmain.DataLoader = DataLoader_no_workers\n\n# chạy inference\nmain.inference()\nprint(\"Done. Output at:\", main.model_save_dir)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os, sys, importlib\n\nSEQ_DIR = \"/kaggle/working/RSNA2019_Intracranial-Hemorrhage-Detection/SequenceModel\"\nos.chdir(SEQ_DIR); sys.path.insert(0, SEQ_DIR)\n\nimport check_feature\nimportlib.reload(check_feature)\n\nprint(\"train_fea type:\", type(getattr(check_feature, \"train_fea\", None)))\nprint(\"test_fea type:\", type(getattr(check_feature, \"test_fea\", None)))\n\ntf = getattr(check_feature, \"test_fea\", None)\nif hasattr(tf, \"shape\"):\n    print(\"test_fea shape:\", tf.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-11T14:21:11.871356Z","iopub.status.idle":"2026-05-11T14:21:11.871601Z","shell.execute_reply.started":"2026-05-11T14:21:11.871482Z","shell.execute_reply":"2026-05-11T14:21:11.871495Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Report biểu đồ + chỉ số","metadata":{}},{"cell_type":"code","source":"# =========================\n# RSNA2019 Report Plots Cell (Kaggle)\n# =========================\nimport os, glob, re\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\n# Optional: nicer plots\nplt.style.use(\"seaborn-v0_8-whitegrid\")\n\n# -------------------------\n# Paths (edit if needed)\n# -------------------------\nCSV_ROOT = \"/kaggle/working/csv\"\nFEATURE_ROOT = \"/kaggle/working/features/stage2_finetune\"\nSUBMISSION_PATH = \"/kaggle/working/FinalSubmission/version3_debug/submission_tta.csv\"\nLOG_PATH = \"/kaggle/working/FinalSubmission/version3_debug/log.txt\"\n\nCLASSES = [\"any\", \"epidural\", \"intraparenchymal\", \"intraventricular\", \"subarachnoid\", \"subdural\"]\n\ndef display_df(df, n=5, name=\"df\"):\n    print(f\"\\n{name}: shape={df.shape}\")\n    display(df.head(n))\n\ndef safe_read_csv(path):\n    if not os.path.exists(path):\n        print(f\"[WARN] missing: {path}\")\n        return None\n    return pd.read_csv(path)\n\n# -------------------------\n# 1) Load submission + pivot to wide\n# -------------------------\nsub = safe_read_csv(SUBMISSION_PATH)\nassert sub is not None, f\"Submission not found: {SUBMISSION_PATH}\"\nassert set([\"ID\",\"Label\"]).issubset(sub.columns), \"submission must have columns: ID, Label\"\n\n# submission: long format (ID = <image>_<class>)\nsub[\"filename\"] = sub[\"ID\"].apply(lambda x: x.split(\"_\")[1]).apply(lambda x: \"ID_\"+x if not x.startswith(\"ID_\") else x)\nsub[\"type\"] = sub[\"ID\"].apply(lambda x: x.split(\"_\")[2])\nsub_wide = sub.pivot_table(index=\"filename\", columns=\"type\", values=\"Label\", aggfunc=\"mean\").reset_index()\n\n# Ensure all classes exist\nfor c in CLASSES:\n    if c not in sub_wide.columns:\n        sub_wide[c] = np.nan\nsub_wide = sub_wide[[\"filename\"] + CLASSES]\n\nprint(\"Loaded submission:\", SUBMISSION_PATH)\nprint(\"submission wide shape:\", sub_wide.shape)\nprint(\"submission label stats (min/mean/max):\")\nprint(sub_wide[CLASSES].agg([\"min\",\"mean\",\"max\"]).T)\n\n# -------------------------\n# 2) Plot: submission probability distributions (histograms)\n# -------------------------\nfig, axes = plt.subplots(2, 3, figsize=(16, 8))\naxes = axes.ravel()\nfor i, c in enumerate(CLASSES):\n    ax = axes[i]\n    x = sub_wide[c].astype(float).clip(0,1)\n    ax.hist(x, bins=50, color=\"steelblue\", alpha=0.85)\n    ax.set_title(f\"Submission prob distribution: {c}\")\n    ax.set_xlabel(\"Predicted probability\")\n    ax.set_ylabel(\"Count\")\nplt.tight_layout()\nplt.show()\n\n# Plot: boxplot across classes\nfig, ax = plt.subplots(figsize=(12, 5))\nax.boxplot([sub_wide[c].astype(float).values for c in CLASSES], labels=CLASSES, showfliers=False)\nax.set_title(\"Submission probability boxplot by class\")\nax.set_ylabel(\"Predicted probability\")\nplt.xticks(rotation=25)\nplt.tight_layout()\nplt.show()\n\n# Plot: correlation heatmap (submission)\ncorr = sub_wide[CLASSES].astype(float).corr()\nfig, ax = plt.subplots(figsize=(7, 6))\nim = ax.imshow(corr.values, cmap=\"coolwarm\", vmin=-1, vmax=1)\nax.set_xticks(range(len(CLASSES))); ax.set_xticklabels(CLASSES, rotation=45, ha=\"right\")\nax.set_yticks(range(len(CLASSES))); ax.set_yticklabels(CLASSES)\nax.set_title(\"Submission class correlation (Pearson)\")\nfig.colorbar(im, ax=ax, fraction=0.046, pad=0.04)\nplt.tight_layout()\nplt.show()\n\n# -------------------------\n# 3) Load train labels (standard.csv) to report label distribution\n# -------------------------\ntrain_std = safe_read_csv(f\"{CSV_ROOT}/standard.csv\")\nif train_std is not None:\n    # standard.csv in repo typically has filename + 6 labels\n    # normalize filename to no extension\n    train_std[\"filename\"] = train_std[\"filename\"].astype(str).str.replace(\".dcm\",\"\", regex=False).str.replace(\".png\",\"\", regex=False)\n    # keep needed columns\n    keep = [\"filename\"] + [c for c in CLASSES if c in train_std.columns]\n    train_std = train_std[keep].dropna(subset=[c for c in CLASSES if c in train_std.columns], how=\"any\")\n\n    print(\"\\nTrain label prevalence (mean label per class):\")\n    prev = train_std[CLASSES].mean().sort_values(ascending=False)\n    print(prev)\n\n    fig, ax = plt.subplots(figsize=(10, 4))\n    ax.bar(prev.index, prev.values, color=\"darkorange\", alpha=0.85)\n    ax.set_title(\"Train label prevalence (fraction positive) - standard.csv\")\n    ax.set_ylabel(\"Positive fraction\")\n    plt.xticks(rotation=25)\n    plt.tight_layout()\n    plt.show()\nelse:\n    print(\"\\n[SKIP] standard.csv not found => skip train label prevalence plots.\")\n\n# -------------------------\n# 4) Compare 2D model val OOF/prob files (if present) + ensemble diagnostics\n#    We will compute:\n#      - per-class histogram for each model (val)\n#      - correlation between model predictions\n#      - simple weighted logloss on train (if mergeable)\n# -------------------------\ndef weighted_logloss_multi_label(y_true, y_pred, eps=1e-7, weights=None):\n    \"\"\"\n    y_true, y_pred shape: (N,6) with columns in CLASSES order.\n    weights: list of 6 weights (default repo uses [2,1,1,1,1,1])\n    \"\"\"\n    if weights is None:\n        weights = np.array([2,1,1,1,1,1], dtype=float)\n    else:\n        weights = np.array(weights, dtype=float)\n\n    y_true = np.asarray(y_true, dtype=float)\n    y_pred = np.asarray(y_pred, dtype=float)\n    y_pred = np.clip(y_pred, eps, 1 - eps)\n\n    losses = -(y_true*np.log(y_pred) + (1-y_true)*np.log(1-y_pred))  # (N,6)\n    per_class = losses.mean(axis=0)  # (6,)\n    weighted = (per_class * weights).sum() / weights.sum()\n    return weighted, pd.Series(per_class, index=CLASSES)\n\nval_prob_files = []\nfor model_dir in sorted(glob.glob(os.path.join(FEATURE_ROOT, \"*\"))):\n    model = os.path.basename(model_dir)\n    f = os.path.join(model_dir, f\"{model}_val_prob_TTA_stage2_finetune.csv\")\n    if os.path.exists(f):\n        val_prob_files.append((model, f))\n\nif len(val_prob_files) == 0:\n    print(\"\\n[SKIP] No *_val_prob_TTA_stage2_finetune.csv found => skip 2D OOF comparison.\")\nelse:\n    print(\"\\nFound val prob files:\")\n    for model, f in val_prob_files:\n        print(\" -\", model, \"=>\", f)\n\n    # Load all model val probs\n    model_val = {}\n    for model, f in val_prob_files:\n        df = pd.read_csv(f)\n        df[\"filename\"] = df[\"filename\"].astype(str).str.replace(\".dcm\",\"\", regex=False).str.replace(\".png\",\"\", regex=False)\n        df = df[[\"filename\"] + CLASSES]\n        model_val[model] = df\n\n    # 4.1 Plot: histogram per model per class (val probs)\n    for c in CLASSES:\n        fig, ax = plt.subplots(figsize=(10,4))\n        for model, df in model_val.items():\n            ax.hist(df[c].astype(float).clip(0,1), bins=60, alpha=0.35, label=model)\n        ax.set_title(f\"VAL prob distributions per model: {c}\")\n        ax.set_xlabel(\"Predicted probability\")\n        ax.set_ylabel(\"Count\")\n        ax.legend()\n        plt.tight_layout()\n        plt.show()\n\n    # 4.2 Model-to-model correlation (val) per class\n    # Build merged frame on filename\n    merged = None\n    for model, df in model_val.items():\n        tmp = df.rename(columns={c: f\"{model}__{c}\" for c in CLASSES})\n        merged = tmp if merged is None else merged.merge(tmp, on=\"filename\", how=\"inner\")\n    print(\"\\nMerged val predictions shape (inner join):\", merged.shape)\n\n    for c in CLASSES:\n        cols = [f\"{m}__{c}\" for m in model_val.keys()]\n        corr_m = merged[cols].corr()\n        fig, ax = plt.subplots(figsize=(6,5))\n        im = ax.imshow(corr_m.values, cmap=\"viridis\", vmin=0, vmax=1)\n        ax.set_xticks(range(len(cols))); ax.set_xticklabels(list(model_val.keys()), rotation=45, ha=\"right\")\n        ax.set_yticks(range(len(cols))); ax.set_yticklabels(list(model_val.keys()))\n        ax.set_title(f\"Model correlation (VAL) for class: {c}\")\n        fig.colorbar(im, ax=ax, fraction=0.046, pad=0.04)\n        plt.tight_layout()\n        plt.show()\n\n    # 4.3 Compute weighted logloss for each model on standard.csv (if possible)\n    if train_std is None:\n        print(\"\\n[SKIP] standard.csv missing => skip val logloss computation.\")\n    else:\n        # For each model: merge with train labels and compute logloss\n        results = []\n        for model, df in model_val.items():\n            mrg = train_std[[\"filename\"] + CLASSES].merge(df, on=\"filename\", how=\"inner\", suffixes=(\"_true\",\"_pred\"))\n            # ensure right columns\n            y_true = mrg[[f\"{c}_true\" for c in CLASSES]].values\n            y_pred = mrg[[f\"{c}_pred\" for c in CLASSES]].values\n            wll, per_class = weighted_logloss_multi_label(y_true, y_pred)\n            results.append({\n                \"model\": model,\n                \"N_merged\": len(mrg),\n                \"weighted_logloss\": wll,\n                **{f\"logloss_{c}\": per_class[c] for c in CLASSES}\n            })\n\n        res_df = pd.DataFrame(results).sort_values(\"weighted_logloss\")\n        print(\"\\nWeighted logloss on merged train (lower is better):\")\n        display(res_df)\n\n        # Bar plot weighted logloss\n        fig, ax = plt.subplots(figsize=(8,4))\n        ax.bar(res_df[\"model\"], res_df[\"weighted_logloss\"], color=\"crimson\", alpha=0.85)\n        ax.set_title(\"2D model OOF/VAL weighted logloss (from standard.csv merge)\")\n        ax.set_ylabel(\"Weighted logloss\")\n        plt.xticks(rotation=20)\n        plt.tight_layout()\n        plt.show()\n\n# -------------------------\n# 5) (Optional) Plot training loss over epochs from log.txt\n# -------------------------\nif os.path.exists(LOG_PATH):\n    with open(LOG_PATH, \"r\") as f:\n        lines = [ln.strip() for ln in f.readlines() if ln.strip()]\n\n    # parse lines like:\n    # fold: 0 33 train_loss:... val_loss:... score:...\n    rows = []\n    for ln in lines:\n        m = re.search(r\"fold:\\s*(\\d+)\\s+(\\d+)\\s+train_loss:([0-9\\.eE+-]+)\\s+val_loss:([0-9\\.eE+-]+)\", ln)\n        if m:\n            rows.append({\n                \"fold\": int(m.group(1)),\n                \"epoch\": int(m.group(2)),\n                \"train_loss\": float(m.group(3)),\n                \"val_loss\": float(m.group(4)),\n            })\n    if len(rows) == 0:\n        print(\"\\n[WARN] log.txt exists but could not parse expected pattern.\")\n    else:\n        logdf = pd.DataFrame(rows).sort_values([\"fold\",\"epoch\"])\n        print(\"\\nParsed training log rows:\", logdf.shape[0])\n        display(logdf.head())\n\n        # plot per-fold curves\n        fig, axes = plt.subplots(1, 2, figsize=(14,4))\n        for fold, g in logdf.groupby(\"fold\"):\n            axes[0].plot(g[\"epoch\"], g[\"train_loss\"], label=f\"fold {fold}\")\n            axes[1].plot(g[\"epoch\"], g[\"val_loss\"], label=f\"fold {fold}\")\n        axes[0].set_title(\"Train loss by epoch (per fold)\")\n        axes[1].set_title(\"Val loss by epoch (per fold)\")\n        axes[0].set_xlabel(\"epoch\"); axes[1].set_xlabel(\"epoch\")\n        axes[0].set_ylabel(\"loss\"); axes[1].set_ylabel(\"loss\")\n        axes[0].legend(); axes[1].legend()\n        plt.tight_layout()\n        plt.show()\n\n        # plot average curves across folds\n        avg = logdf.groupby(\"epoch\")[[\"train_loss\",\"val_loss\"]].mean().reset_index()\n        fig, ax = plt.subplots(figsize=(8,4))\n        ax.plot(avg[\"epoch\"], avg[\"train_loss\"], label=\"train_loss (avg)\")\n        ax.plot(avg[\"epoch\"], avg[\"val_loss\"], label=\"val_loss (avg)\")\n        ax.set_title(\"Average loss across folds\")\n        ax.set_xlabel(\"epoch\"); ax.set_ylabel(\"loss\")\n        ax.legend()\n        plt.tight_layout()\n        plt.show()\nelse:\n    print(\"\\n[SKIP] log.txt not found => skip loss curves.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-11T14:33:26.256004Z","iopub.execute_input":"2026-05-11T14:33:26.257005Z","iopub.status.idle":"2026-05-11T14:33:37.291748Z","shell.execute_reply.started":"2026-05-11T14:33:26.256971Z","shell.execute_reply":"2026-05-11T14:33:37.291049Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os, glob\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.calibration import calibration_curve\n\nplt.style.use(\"seaborn-v0_8-whitegrid\")\n\nCSV_ROOT = \"/kaggle/working/csv\"\nFEATURE_ROOT = \"/kaggle/working/features/stage2_finetune\"\nCLASSES = [\"any\", \"epidural\", \"intraparenchymal\", \"intraventricular\", \"subarachnoid\", \"subdural\"]\nWEIGHTS = np.array([2,1,1,1,1,1], dtype=float)\n\ndef norm_fn(s: pd.Series) -> pd.Series:\n    # normalize filenames used across repo\n    return (s.astype(str)\n             .str.replace(\".dcm\",\"\", regex=False)\n             .str.replace(\".png\",\"\", regex=False)\n             .str.strip())\n\ndef weighted_logloss(y_true, y_pred, eps=1e-7):\n    y_true = np.asarray(y_true, float)\n    y_pred = np.clip(np.asarray(y_pred, float), eps, 1-eps)\n    loss = -(y_true*np.log(y_pred) + (1-y_true)*np.log(1-y_pred))   # (N,6)\n    per_class = loss.mean(axis=0)\n    w = (per_class * WEIGHTS).sum() / WEIGHTS.sum()\n    return w, pd.Series(per_class, index=CLASSES)\n\n# 1) Load ground truth (train)\ngt = pd.read_csv(f\"{CSV_ROOT}/standard.csv\")\ngt[\"filename\"] = norm_fn(gt[\"filename\"])\ngt = gt[[\"filename\"] + CLASSES]\nprint(\"GT standard.csv:\", gt.shape, \"unique filenames:\", gt[\"filename\"].nunique())\nprint(\"GT positive rate:\", gt[CLASSES].mean().to_dict())\n\n# 2) Find model val prob files\nval_prob_files = []\nfor model_dir in sorted(glob.glob(os.path.join(FEATURE_ROOT, \"*\"))):\n    model = os.path.basename(model_dir)\n    f = os.path.join(model_dir, f\"{model}_val_prob_TTA_stage2_finetune.csv\")\n    if os.path.exists(f):\n        val_prob_files.append((model, f))\n\nassert len(val_prob_files) > 0, \"No val prob csv found in features/stage2_finetune/*\"\n\nprint(\"\\nFound VAL prob files:\")\nfor m,f in val_prob_files:\n    print(\" -\", m, \"=>\", f)\n\nall_metrics = []\n\nfor model, path in val_prob_files:\n    pred = pd.read_csv(path)\n    pred[\"filename\"] = norm_fn(pred[\"filename\"])\n    pred = pred[[\"filename\"] + CLASSES]\n\n    # 3) Merge check\n    merged = gt.merge(pred, on=\"filename\", how=\"inner\", suffixes=(\"_true\",\"_pred\"))\n    match_rate = len(merged) / len(gt)\n    print(f\"\\n[{model}] merged rows: {len(merged)} / {len(gt)} (match_rate={match_rate:.3f})\")\n\n    if len(merged) == 0:\n        # diagnose why\n        gt_set = set(gt[\"filename\"].head(2000))\n        pred_set = set(pred[\"filename\"].head(2000))\n        print(\"Example gt filenames:\", list(gt_set)[:5])\n        print(\"Example pred filenames:\", list(pred_set)[:5])\n        continue\n\n    y_true = merged[[f\"{c}_true\" for c in CLASSES]].values\n    y_pred = merged[[f\"{c}_pred\" for c in CLASSES]].values\n\n    # 4) Metrics\n    wll, per_class = weighted_logloss(y_true, y_pred)\n\n    aucs = {}\n    for i,c in enumerate(CLASSES):\n        yt = y_true[:, i]\n        yp = y_pred[:, i]\n        aucs[c] = roc_auc_score(yt, yp) if len(np.unique(yt)) == 2 else np.nan\n\n    row = {\n        \"model\": model,\n        \"N\": len(merged),\n        \"match_rate\": match_rate,\n        \"weighted_logloss\": wll,\n        **{f\"logloss_{c}\": float(per_class[c]) for c in CLASSES},\n        **{f\"auc_{c}\": float(aucs[c]) for c in CLASSES},\n    }\n    all_metrics.append(row)\n\n    print(f\"  weighted_logloss: {wll:.6f}\")\n    print(\"  per-class logloss:\", {c: float(per_class[c]) for c in CLASSES})\n    print(\"  per-class auc:\", {c: float(aucs[c]) for c in CLASSES})\n\n    # 5) Plots: distributions by label (positive vs negative) for each class\n    fig, axes = plt.subplots(2, 3, figsize=(16, 8))\n    axes = axes.ravel()\n    for i,c in enumerate(CLASSES):\n        ax = axes[i]\n        yt = merged[f\"{c}_true\"].values.astype(int)\n        yp = merged[f\"{c}_pred\"].values.astype(float)\n\n        ax.hist(yp[yt==0], bins=50, alpha=0.6, label=\"neg (y=0)\")\n        ax.hist(yp[yt==1], bins=50, alpha=0.6, label=\"pos (y=1)\")\n        ax.set_title(f\"{model} VAL probs by label: {c}\")\n        ax.set_xlabel(\"pred prob\")\n        ax.set_ylabel(\"count\")\n        ax.legend()\n    plt.tight_layout()\n    plt.show()\n\n    # 6) Calibration curve for 'any' (often most important)\n    c = \"any\"\n    prob_true, prob_pred = calibration_curve(\n        merged[f\"{c}_true\"].values.astype(int),\n        merged[f\"{c}_pred\"].values.astype(float),\n        n_bins=15,\n        strategy=\"quantile\"\n    )\n    fig, ax = plt.subplots(figsize=(6,6))\n    ax.plot([0,1],[0,1], \"--\", color=\"gray\", label=\"perfect\")\n    ax.plot(prob_pred, prob_true, marker=\"o\", label=f\"{model} ({c})\")\n    ax.set_title(f\"Calibration curve (VAL) - class '{c}'\")\n    ax.set_xlabel(\"mean predicted probability\")\n    ax.set_ylabel(\"fraction of positives\")\n    ax.legend()\n    plt.tight_layout()\n    plt.show()\n\n# 7) Summary table\nmetrics_df = pd.DataFrame(all_metrics).sort_values(\"weighted_logloss\")\nprint(\"\\n=== SUMMARY METRICS (VAL/OOF) ===\")\ndisplay(metrics_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-11T14:35:29.164453Z","iopub.execute_input":"2026-05-11T14:35:29.165124Z","iopub.status.idle":"2026-05-11T14:35:38.896958Z","shell.execute_reply.started":"2026-05-11T14:35:29.165095Z","shell.execute_reply":"2026-05-11T14:35:38.895968Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os, re\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nLOG_DIR = \"/kaggle/working/FinalSubmission/version3_debug\"\n\nprint(\"Listing:\", LOG_DIR)\nfiles = sorted(os.listdir(LOG_DIR)) if os.path.exists(LOG_DIR) else []\nprint(\"\\n\".join(files[:200]))\n\n# tìm các file text tiềm năng\ncandidates = [f for f in files if f.lower().endswith((\".txt\", \".log\"))]\nprint(\"\\nLog candidates:\", candidates)\n\n# chọn file log: ưu tiên log.txt nếu có, không thì file txt/log đầu tiên\nlog_path = \"/kaggle/input/datasets/ren4gg/rsna2019/log.txt\"\n# if \"log.txt\" in files:\n#     log_path = os.path.join(LOG_DIR, \"log.txt\")\n# elif len(candidates) > 0:\n#     log_path = os.path.join(LOG_DIR, candidates[0])\n\n# if log_path is None or (not os.path.exists(log_path)):\n#     raise FileNotFoundError(f\"Không tìm thấy file log (.txt/.log) trong {LOG_DIR}. \"\n#                             \"Bạn hãy tạo log bằng cách redirect stdout khi chạy main.py, hoặc paste tên file trong thư mục.\")\n\nprint(\"\\nUsing log file:\", log_path)\n\n# xem thử 80 dòng đầu\nwith open(log_path, \"r\", encoding=\"utf-8\", errors=\"ignore\") as f:\n    head_lines = [next(f).rstrip(\"\\n\") for _ in range(80)]\nprint(\"\\n--- log head (first 80 lines) ---\")\nprint(\"\\n\".join(head_lines))\n\n# parse linh hoạt\nlines = []\nwith open(log_path, \"r\", encoding=\"utf-8\", errors=\"ignore\") as f:\n    lines = [ln.strip() for ln in f if ln.strip()]\n\nrows = []\n\n# Pattern A: \"33 train_loss:... val_loss:... score:...\"\npatA = re.compile(r\"^\\s*(\\d+)\\s+train_loss:([0-9\\.eE+-]+)\\s+val_loss:([0-9\\.eE+-]+)\\s+score:([0-9\\.eE+-]+)\")\n# Pattern B: \"fold: 0 33 train_loss:... val_loss:... score:...\"\npatB = re.compile(r\"fold:\\s*(\\d+)\\s+(\\d+)\\s+train_loss:([0-9\\.eE+-]+)\\s+val_loss:([0-9\\.eE+-]+)\\s+score:([0-9\\.eE+-]+)\")\n# Pattern C: your console style might not include \"score:\" (fallback)\npatC = re.compile(r\"^\\s*(\\d+)\\s+train_loss:([0-9\\.eE+-]+)\\s+val_loss:([0-9\\.eE+-]+)\")\n\nfor ln in lines:\n    m = patB.search(ln)\n    if m:\n        rows.append({\"fold\": int(m.group(1)), \"epoch\": int(m.group(2)),\n                     \"train_loss\": float(m.group(3)), \"val_loss\": float(m.group(4)), \"score\": float(m.group(5))})\n        continue\n    m = patA.search(ln)\n    if m:\n        rows.append({\"fold\": -1, \"epoch\": int(m.group(1)),\n                     \"train_loss\": float(m.group(2)), \"val_loss\": float(m.group(3)), \"score\": float(m.group(4))})\n        continue\n    m = patC.search(ln)\n    if m:\n        rows.append({\"fold\": -1, \"epoch\": int(m.group(1)),\n                     \"train_loss\": float(m.group(2)), \"val_loss\": float(m.group(3)), \"score\": float(m.group(3))})\n        continue\n\nif len(rows) == 0:\n    raise RuntimeError(\n        \"Không parse được epoch/loss từ log. \"\n        \"Hãy paste vài dòng log tiêu biểu (có 'train_loss'/'val_loss') để mình viết regex đúng.\"\n    )\n\ndf = pd.DataFrame(rows).sort_values([\"fold\",\"epoch\"]).reset_index(drop=True)\nprint(\"\\nParsed rows:\", df.shape)\ndisplay(df.tail(15))\n\n# best score: score nhỏ nhất\nbest = df.loc[df[\"score\"].idxmin()]\nprint(\"\\nBest:\")\nprint(best)\n\n# plot curves: nếu có nhiều fold thì vẽ từng fold + average\nplt.figure(figsize=(10,4))\nfor fold, g in df.groupby(\"fold\"):\n    label = f\"fold {fold}\" if fold != -1 else \"all\"\n    plt.plot(g[\"epoch\"], g[\"val_loss\"], alpha=0.6, label=label)\nplt.title(\"Val loss / score over epochs (parsed log)\")\nplt.xlabel(\"epoch\"); plt.ylabel(\"val_loss/score\")\nplt.legend()\nplt.tight_layout()\nplt.show()\n\navg = df.groupby(\"epoch\")[[\"train_loss\",\"val_loss\",\"score\"]].mean().reset_index()\nplt.figure(figsize=(10,4))\nplt.plot(avg[\"epoch\"], avg[\"train_loss\"], label=\"train_loss (avg)\")\nplt.plot(avg[\"epoch\"], avg[\"val_loss\"], label=\"val_loss (avg)\")\nplt.title(\"Average loss over epochs\")\nplt.xlabel(\"epoch\"); plt.ylabel(\"loss\")\nplt.legend()\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-11T14:40:56.528761Z","iopub.execute_input":"2026-05-11T14:40:56.529374Z","iopub.status.idle":"2026-05-11T14:40:56.92397Z","shell.execute_reply.started":"2026-05-11T14:40:56.529346Z","shell.execute_reply":"2026-05-11T14:40:56.923105Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# df là dataframe bạn vừa parse được (fold, epoch, train_loss, val_loss, score)\n# Nếu chưa còn df trong kernel, bạn chạy lại cell parse log trước.\n\nbest_by_fold = df.loc[df.groupby(\"fold\")[\"score\"].idxmin()].sort_values(\"fold\").reset_index(drop=True)\ndisplay(best_by_fold)\n\nprint(\"Mean best score:\", best_by_fold[\"score\"].mean())\nprint(\"Std  best score:\", best_by_fold[\"score\"].std())","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**TEST**","metadata":{}},{"cell_type":"code","source":"import os, main\n\nprint(\"model_save_dir:\", main.model_save_dir)\nprint(\"has fold_4:\", os.path.exists(os.path.join(main.model_save_dir, \"fold_4.pt\")))\n\n# các CSV metadata quan trọng\nfrom settings import csv_root\nprint(\"csv_root:\", csv_root)\nprint(\"has test meta:\", os.path.exists(os.path.join(csv_root, \"test_meta_id_seriser_stage2.csv\")))\nprint(\"has study_csv dir:\", os.path.isdir(os.path.join(csv_root, \"study_csv\")))\n\nstudy_dir = os.path.join(csv_root, \"study_csv\")\nif os.path.isdir(study_dir):\n    print(\"study_csv sample:\", sorted(os.listdir(study_dir))[:5])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-11T14:21:11.88606Z","iopub.status.idle":"2026-05-11T14:21:11.886465Z","shell.execute_reply.started":"2026-05-11T14:21:11.88631Z","shell.execute_reply":"2026-05-11T14:21:11.886332Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os, shutil, sys, importlib\n\nSEQ_DIR = \"/kaggle/working/RSNA2019_Intracranial-Hemorrhage-Detection/SequenceModel\"\nos.chdir(SEQ_DIR)\nsys.path.insert(0, SEQ_DIR)\n\nimport main\nimportlib.reload(main)\n\nsrc = \"/kaggle/input/datasets/ren4gg/rsna2019/fold_4/fold_4.pt\"\ndst_dir = main.model_save_dir                         # /kaggle/working/FinalSubmission/version3_debug\ndst = os.path.join(dst_dir, \"fold_0.pt\")              # đổi tên để inference load được\n\nos.makedirs(dst_dir, exist_ok=True)\nshutil.copy(src, dst)\n\nprint(\"Copied to:\", dst)\nprint(\"Exists:\", os.path.exists(dst))\nimport importlib\nimportlib.reload(main)\n\n# cấu hình inference (giống bạn đã set)\nmain.fold_num = 1\nmain.Add_position = True\nmain.lstm_layers = 2\nmain.seq_len = 24\nmain.hidden = 96\nmain.drop_out = 0.5\nmain.class_num = 6\n\n# giảm worker cho Kaggle\norig_DataLoader = main.DataLoader\ndef DataLoader_no_workers(*args, **kwargs):\n    kwargs[\"num_workers\"] = 0\n    kwargs[\"pin_memory\"] = False\n    return orig_DataLoader(*args, **kwargs)\nmain.DataLoader = DataLoader_no_workers\n\nmain.inference()\n\nprint(\"Submission at:\", os.path.join(main.model_save_dir, \"submission_tta.csv\"))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os, shutil\nimport main\n\nsrc = os.path.join(main.model_save_dir, \"fold_0.pt\")\nfor i in range(1, 5):\n    dst = os.path.join(main.model_save_dir, f\"fold_{i}.pt\")\n    if not os.path.exists(dst):\n        shutil.copy(src, dst)\n        print(\"copied:\", dst)\n\nprint(\"done\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-11T14:28:03.923276Z","iopub.execute_input":"2026-05-11T14:28:03.924021Z","iopub.status.idle":"2026-05-11T14:28:03.96212Z","shell.execute_reply.started":"2026-05-11T14:28:03.923988Z","shell.execute_reply":"2026-05-11T14:28:03.961151Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"main.inference()\nprint(\"Submission at:\", os.path.join(main.model_save_dir, \"submission_tta.csv\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-11T14:28:12.3749Z","iopub.execute_input":"2026-05-11T14:28:12.375746Z","iopub.status.idle":"2026-05-11T14:31:12.823421Z","shell.execute_reply.started":"2026-05-11T14:28:12.375716Z","shell.execute_reply":"2026-05-11T14:31:12.822565Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\ndef print_tree(base_dir, stop_threshold=10):\n    for root, dirs, files in os.walk(base_dir):\n        level = root.replace(base_dir, \"\").count(os.sep)\n        indent = \" \" * 4 * level\n        print(f\"{indent}{os.path.basename(root)}/\")\n\n        # Nếu số lượng file trong thư mục vượt ngưỡng thì dừng lại\n        if len(files) > stop_threshold:\n            print(f\"{indent}    ... {len(files)} files (stopped here)\")\n            continue\n\n        subindent = \" \" * 4 * (level + 1)\n        for f in files:\n            print(f\"{subindent}{f}\")\n\n# In cây thư mục Kaggle working\nprint_tree(\"/kaggle/working\", stop_threshold=10)\n\n# In cây thư mục dataset Kaggle input\nprint_tree(\"/kaggle/input\", stop_threshold=10)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-11T14:23:17.33432Z","iopub.execute_input":"2026-05-11T14:23:17.334613Z","iopub.status.idle":"2026-05-11T14:24:31.02627Z","shell.execute_reply.started":"2026-05-11T14:23:17.33459Z","shell.execute_reply":"2026-05-11T14:24:31.025192Z"}},"outputs":[],"execution_count":null}]}