{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":99552,"databundleVersionId":13441085},{"sourceType":"datasetVersion","sourceId":12925487,"datasetId":8178911,"databundleVersionId":13579111},{"sourceType":"datasetVersion","sourceId":12893106,"datasetId":7979194,"databundleVersionId":13542196},{"sourceType":"kernelVersion","sourceId":226368929}],"dockerImageVersionId":30919,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install ultralytics","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T05:59:20.786657Z","iopub.execute_input":"2025-09-01T05:59:20.786988Z","iopub.status.idle":"2025-09-01T05:59:28.328368Z","shell.execute_reply.started":"2025-09-01T05:59:20.78696Z","shell.execute_reply":"2025-09-01T05:59:28.327293Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport torch\nimport numpy as np\nimport random\nimport glob\nfrom pathlib import Path\nfrom PIL import Image\nimport matplotlib.pyplot as plt\nfrom matplotlib.patches import Rectangle\nfrom ultralytics import YOLO\nfrom torchvision.ops import nms\nimport yaml\nimport pandas as pd\nimport json\n\n# Set random seeds for reproducibility\nnp.random.seed(42)\nrandom.seed(42)\ntorch.manual_seed(42)\n\n# Define paths for Kaggle environment\nyolo_dataset_dir = \"/kaggle/input/parse-data/yolo_dataset\"\nyolo_weights_dir = \"/kaggle/working/yolo_weights\"\nyolo_pretrained_weights = \"yolov8n.pt\"  # Path to pre-downloaded weights\n\n# Create weights directory if it doesn't exist\nos.makedirs(yolo_weights_dir, exist_ok=True)\n\ndef prepare_dataset(yolo_dataset_dir: str, class_names: dict | None = None) -> str:\n    \"\"\"\n    既存の dataset.yaml を優先。無ければ最小構成で作成。\n    \"\"\"\n    yolo_dataset_dir = str(yolo_dataset_dir)\n    images_train = Path(yolo_dataset_dir) / \"images\" / \"train\"\n    images_val   = Path(yolo_dataset_dir) / \"images\" / \"val\"\n    labels_train = Path(yolo_dataset_dir) / \"labels\" / \"train\"\n    labels_val   = Path(yolo_dataset_dir) / \"labels\" / \"val\"\n\n    print(\"Directory status:\")\n    print(f\"- {images_train} exists: {images_train.exists()}\")\n    print(f\"- {images_val}   exists: {images_val.exists()}\")\n    print(f\"- {labels_train} exists: {labels_train.exists()}\")\n    print(f\"- {labels_val}   exists: {labels_val.exists()}\")\n\n    yaml_path = Path(yolo_dataset_dir) / \"dataset.yaml\"\n    if yaml_path.exists():\n        print(f\"Found dataset.yaml at {yaml_path}\")\n        # 中身の path/train/val が相対表記かチェックし、相対に修正（YOLOが推奨）\n        with open(yaml_path, \"r\") as f:\n            data = yaml.safe_load(f)\n        changed = False\n        if str(data.get(\"path\", \"\")) != yolo_dataset_dir:\n            data[\"path\"] = yolo_dataset_dir; changed = True\n        if data.get(\"train\", \"\") not in (\"images/train\", str(images_train)):\n            data[\"train\"] = \"images/train\"; changed = True\n        if data.get(\"val\", \"\") not in (\"images/val\", str(images_val)):\n            data[\"val\"] = \"images/val\"; changed = True\n        if \"names\" not in data:\n            data[\"names\"] = class_names or {0: \"aneurysm\"}; changed = True\n        if changed:\n            with open(yaml_path, \"w\") as f:\n                yaml.safe_dump(data, f, sort_keys=False, allow_unicode=True)\n            print(\"dataset.yaml was normalized (path/train/val/names).\")\n        return str(yaml_path)\n\n    print(\"dataset.yaml not found. Creating a new one.\")\n    if class_names is None:\n        class_names = {0: \"aneurysm\"}\n    data = {\n        \"path\": yolo_dataset_dir,\n        \"train\": \"images/train\",\n        \"val\": \"images/val\",\n        \"names\": class_names,\n    }\n    with open(yaml_path, \"w\") as f:\n        yaml.safe_dump(data, f, sort_keys=False, allow_unicode=True)\n    print(f\"Created dataset.yaml at: {yaml_path}\")\n    return str(yaml_path)\n\n\ndef train_yolo_model(\n    yaml_path: str,\n    pretrained_weights_path: str,\n    project_dir: str,\n    run_name: str,\n    *,\n    epochs: int = 30,\n    batch_size: int = 16,\n    img_size: int = 640,\n    rect: bool = True,         # ← アスペクト比を保つ\n    # augmentation knobs (medical想定の控えめ設定)\n    mosaic: float = 0.0,       # 医療画像では通常OFFを推奨\n    mixup: float = 0.0,\n    copy_paste: float = 0.0,\n    degrees: float = 5.0,\n    translate: float = 0.05,   # 0.0～0.2程度で局所クロップに近い効果\n    scale: float = 0.15,       # ズームイン(=実質的なクロップ)も起きる\n    shear: float = 0.0,\n    perspective: float = 0.0,\n    fliplr: float = 0.5,\n    flipud: float = 0.0,\n    patience: int = 8,\n    save_period: int = 5,\n    seed: int = 42,\n    device: str | int | None = None,\n):\n    \"\"\"Ultralytics YOLOv8 の学習。rect=True でアスペクト比維持。\"\"\"\n    model = YOLO(pretrained_weights_path)\n    results = model.train(\n        data=yaml_path,\n        epochs=epochs,\n        batch=batch_size,\n        imgsz=img_size,\n        project=project_dir,\n        name=run_name,\n        exist_ok=True,\n        device=device,\n        seed=seed,\n        patience=patience,\n        save_period=save_period,\n        rect=rect,                      # ← これが“アスペクト比維持”の要\n        # built-in augmentation\n        mosaic=mosaic,\n        mixup=mixup,\n        copy_paste=copy_paste,\n        degrees=degrees,\n        translate=translate,\n        scale=scale,\n        shear=shear,\n        perspective=perspective,\n        fliplr=fliplr,\n        flipud=flipud,\n        workers=2,\n        verbose=True,\n    )\n    run_dir = str(Path(project_dir) / run_name)\n    return model, results, run_dir\n\n\ndef _count_split_images_and_boxes(yolo_dataset_dir: str, split: str = \"val\"):\n    \"\"\"\n    画像枚数と GT ボックス数（YOLO txt の行数合計）を数える\n    \"\"\"\n    img_paths = list(Path(yolo_dataset_dir, \"images\", split).rglob(\"*.png\"))\n    lbl_dir = Path(yolo_dataset_dir, \"labels\", split)\n    box_count = 0\n    for img_p in img_paths:\n        rel = img_p.relative_to(Path(yolo_dataset_dir) / \"images\" / split)\n        txt_p = lbl_dir / rel.with_suffix(\".txt\")\n        if txt_p.exists():\n            try:\n                with open(txt_p, \"r\") as f:\n                    box_count += sum(1 for _ in f if _.strip())\n            except Exception:\n                pass\n    return len(img_paths), box_count\n\n\ndef _safe_scalar(x):\n    \"\"\"\n    Ultralytics の metrics から取り出した値を float に丸めるための安全関数\n    \"\"\"\n    if x is None:\n        return np.nan\n    try:\n        if hasattr(x, \"item\"):\n            return float(x.item())\n        if isinstance(x, (list, tuple, np.ndarray)):\n            if len(x) == 0:\n                return np.nan\n            return float(np.nanmean(x))\n        return float(x)\n    except Exception:\n        return np.nan\n\n\ndef _extract_val_metrics(metrics_obj):\n    \"\"\"\n    model.val(...) の戻り値 (metrics) から主なスカラーを抽出\n    \"\"\"\n    m = metrics_obj\n    out = {\n        \"mAP50-95\": _safe_scalar(getattr(getattr(m, \"box\", None), \"map\", None)),\n        \"mAP50\":    _safe_scalar(getattr(getattr(m, \"box\", None), \"map50\", None)),\n        \"precision\":_safe_scalar(\n            getattr(getattr(m, \"box\", None), \"mp\",\n                    getattr(getattr(m, \"box\", None), \"p\", None))\n        ),\n        \"recall\":   _safe_scalar(\n            getattr(getattr(m, \"box\", None), \"mr\",\n                    getattr(getattr(m, \"box\", None), \"r\", None))\n        ),\n    }\n    return out\n\n\ndef _read_last_results_row(run_dir: str):\n    \"\"\"\n    runs/<name>/results.csv の最終行を dict で返す（無ければ空）\n    \"\"\"\n    csv_path = Path(run_dir) / \"results.csv\"\n    if not csv_path.exists():\n        return {}\n    df = pd.read_csv(csv_path)\n    if df.empty:\n        return {}\n    last = df.iloc[-1].to_dict()\n    # 欲しい列だけ薄めに拾う（存在しない場合は自動で無視）\n    keys = [\n        \"val/box_loss\", \"val/cls_loss\", \"val/dfl_loss\",\n        \"metrics/mAP50-95(B)\", \"metrics/mAP50(B)\", \"metrics/precision(B)\", \"metrics/recall(B)\"\n    ]\n    return {k: float(last[k]) for k in keys if k in last}\n\n\ndef _summarize_cv(rows, out_dir: str):\n    \"\"\"\n    rows: 各 fold の dict を集めたリスト\n    - mean / std を計算\n    - 画像枚数・GT数での重み付き平均も出力\n    \"\"\"\n    df = pd.DataFrame(rows)\n    num_cols = df.select_dtypes(include=[np.number]).columns.tolist()\n\n    # 平均・標準偏差\n    mean = df[num_cols].mean(numeric_only=True)\n    std  = df[num_cols].std(numeric_only=True, ddof=1)\n\n    # 重み（画像枚数 / GT数）\n    w_img = df[\"val_images\"].to_numpy(dtype=float)\n    w_box = df[\"val_boxes\"].to_numpy(dtype=float)\n    def wavg(col, w):\n        x = df[col].to_numpy(dtype=float)\n        mask = ~np.isnan(x) & (w > 0)\n        if mask.sum() == 0:\n            return np.nan\n        return float(np.average(x[mask], weights=w[mask]))\n\n    weighted_img = {col: wavg(col, w_img) for col in num_cols}\n    weighted_box = {col: wavg(col, w_box) for col in num_cols}\n\n    # まとめ\n    summary = pd.DataFrame({\n        \"mean\": mean,\n        \"std\": std,\n        \"wavg_by_images\": pd.Series(weighted_img),\n        \"wavg_by_boxes\": pd.Series(weighted_box),\n    }).sort_index()\n\n    # 保存\n    out_dir = Path(out_dir)\n    out_dir.mkdir(parents=True, exist_ok=True)\n    df.to_csv(out_dir / \"cv_folds_raw.csv\", index=False)\n    summary.to_csv(out_dir / \"cv_summary.csv\")\n\n    print(\"\\n===== Cross-Validation Summary =====\")\n    print(summary.loc[[\"mAP50-95\", \"mAP50\", \"precision\", \"recall\",\n                       \"val/box_loss\", \"val/cls_loss\", \"val/dfl_loss\"], :].dropna(how=\"all\"))\n    print(f\"\\nSaved CV raw per-fold metrics => {out_dir / 'cv_folds_raw.csv'}\")\n    print(f\"Saved CV summary            => {out_dir / 'cv_summary.csv'}\")\n\n    return df, summary\n\n\ndef run_5fold_training_with_cv(\n    df: pd.DataFrame,\n    base_out_dir: str,\n    pretrained_weights_path: str,\n    *,\n    epochs: int = 30,\n    batch_size: int = 16,\n    img_size: int = 640,\n    folds: int = 5,\n    include_unlabeled: bool = True,\n    class_names: dict = None,\n    rect: bool = True,   # アスペクト比維持\n):\n    \"\"\"\n    5-fold を回して、foldごとのメトリクス＋CV集計を出力\n    \"\"\"\n    if class_names is None:\n        class_names = {0: \"aneurysm\"}\n\n    per_fold_rows = []\n    for k in range(folds):\n        print(f\"\\n===== Fold {k} / {folds} =====\")\n        yolo_dataset_dir = str(Path(base_out_dir) / f\"fold{k}\")\n        # 1) データを fold で振分\n        _yaml, _manifest = yolo_distribute_by_fold(\n            df, yolo_dataset_dir=yolo_dataset_dir,\n            val_fold=k, include_unlabeled=include_unlabeled,\n            preserve_subdirs=True, class_names=class_names\n        )\n        # 2) dataset.yaml\n        yaml_path = prepare_dataset(yolo_dataset_dir, class_names=class_names)\n\n        # 3) 学習\n        project_dir = str(Path(base_out_dir) / \"runs\")\n        run_name = f\"fold{k}\"\n        model, results, run_dir = train_yolo_model(\n            yaml_path=yaml_path,\n            pretrained_weights_path=pretrained_weights_path,\n            project_dir=project_dir,\n            run_name=run_name,\n            epochs=epochs, batch_size=batch_size, img_size=img_size,\n            rect=rect,\n            mosaic=0.5, mixup=0.5, copy_paste=0.0,\n            degrees=5.0, translate=0.05, scale=0.15, shear=0.0, perspective=0.0,\n            fliplr=0.5, flipud=0.5,\n            patience=8, save_period=5, seed=42, device=None,\n        )\n\n        # 4) 検証メトリクス\n        try:\n            metrics = model.val(data=yaml_path, imgsz=img_size, device=None, split=\"val\", max_det=300)\n            m = _extract_val_metrics(metrics)\n        except Exception as e:\n            print(f\"[fold{k}] val failed: {e}\")\n            m = {\"mAP50-95\": np.nan, \"mAP50\": np.nan, \"precision\": np.nan, \"recall\": np.nan}\n\n        # 5) results.csv の最終行から val 損失などを拾う\n        last = _read_last_results_row(run_dir)\n\n        # 6) 画像枚数・GT数\n        val_images, val_boxes = _count_split_images_and_boxes(yolo_dataset_dir, split=\"val\")\n\n        row = {\n            \"fold\": k,\n            \"run_dir\": run_dir,\n            \"dataset_dir\": yolo_dataset_dir,\n            \"val_images\": val_images,\n            \"val_boxes\": val_boxes,\n            **m,\n            **last,   # 例: val/box_loss など\n        }\n        per_fold_rows.append(row)\n\n        # 7) （任意）可視化\n        try:\n            predict_on_samples_with_tta(\n                model, yolo_dataset_dir=yolo_dataset_dir, split=\"val\",\n                num_samples=8, conf=0.25, iou=0.7, img_size=img_size,\n                out_png=str(Path(run_dir) / \"predictions_grid.png\"),\n                angles=(0,90,180,270),\n                only_one=True\n            )\n        except Exception as e:\n            print(f\"[fold{k}] preview failed: {e}\")\n\n    # 8) CV 集計\n    cv_dir = str(Path(base_out_dir) / \"cv_summary\")\n    df_raw, df_summary = _summarize_cv(per_fold_rows, out_dir=cv_dir)\n\n    return {\n        \"per_fold\": df_raw,\n        \"summary\": df_summary,\n        \"runs_dir\": str(Path(base_out_dir) / \"runs\"),\n        \"cv_dir\": cv_dir,\n    }\n\n\ndef plot_loss_curves(run_dir: str, out_path: str):\n    \"\"\"\n    Ultralytics の results.csv から box_loss / cls_loss / dfl_loss / mAP50-95 を描画。\n    \"\"\"\n    import pandas as pd\n    csv_path = Path(run_dir) / \"results.csv\"\n    if not csv_path.exists():\n        raise FileNotFoundError(f\"{csv_path} not found.\")\n    df = pd.read_csv(csv_path)\n    plt.figure(figsize=(10, 7))\n    # あれば列を順次描く\n    for col in [\"train/box_loss\", \"train/cls_loss\", \"train/dfl_loss\", \"val/box_loss\", \"val/cls_loss\", \"val/dfl_loss\"]:\n        if col in df.columns:\n            plt.plot(df.index, df[col], label=col)\n    for col in [\"metrics/mAP50-95(B)\",\"metrics/mAP50(B)\",\"metrics/precision(B)\",\"metrics/recall(B)\"]:\n        if col in df.columns:\n            plt.plot(df.index, df[col], label=col)\n    plt.xlabel(\"epoch\")\n    plt.legend()\n    plt.grid(True)\n    plt.tight_layout()\n    plt.savefig(out_path)\n    plt.close()\n\n\n# ============ 3) 予測の可視化（GT=YOLO txt を読み込む） ============\n\ndef _yolo_txt_to_boxes(txt_path: Path, img_w: int, img_h: int):\n    \"\"\"\n    YOLO txt (class cx cy w h) 正規化 → ピクセルの xyxy に変換。\n    \"\"\"\n    boxes = []\n    if not txt_path.exists():\n        return boxes\n    with open(txt_path, \"r\") as f:\n        for line in f:\n            parts = line.strip().split()\n            if len(parts) < 5:\n                continue\n            cls = int(float(parts[0]))\n            cx = float(parts[1]) * img_w\n            cy = float(parts[2]) * img_h\n            w  = float(parts[3]) * img_w\n            h  = float(parts[4]) * img_h\n            x1 = cx - w / 2\n            y1 = cy - h / 2\n            x2 = cx + w / 2\n            y2 = cy + h / 2\n            boxes.append((cls, x1, y1, x2, y2))\n    return boxes\n\n\ndef predict_on_samples(\n    model,\n    yolo_dataset_dir: str,\n    split: str = \"val\",\n    num_samples: int = 8,\n    conf: float = 0.25,\n    iou: float = 0.7,\n    img_size: int = 320,\n    out_png: str = \"/kaggle/working/predictions.png\",\n):\n    \"\"\"\n    split（val/train）からランダムに画像を選び、GT(緑)と予測(赤)を重ね描き。\n    サブディレクトリも再帰的に探索。\n    \"\"\"\n    split_dir = Path(yolo_dataset_dir) / \"images\" / split\n    if not split_dir.exists():\n        alt = Path(yolo_dataset_dir) / \"images\" / \"train\"\n        print(f\"{split_dir} not found. Using {alt}\")\n        split_dir = alt\n\n    image_paths = [Path(p) for p in glob.glob(str(split_dir / \"**\" / \"*.png\"), recursive=True)]\n    if len(image_paths) == 0:\n        print(f\"No images found under {split_dir}\")\n        return\n\n    samples = random.sample(image_paths, k=min(num_samples, len(image_paths)))\n\n    # 2x2, 3x3 など自動レイアウト\n    cols = int(np.ceil(np.sqrt(len(samples))))\n    rows = int(np.ceil(len(samples) / cols))\n    fig, axes = plt.subplots(rows, cols, figsize=(4.5 * cols, 4.5 * rows))\n    axes = np.array(axes).reshape(-1)\n\n    for ax in axes[len(samples):]:\n        ax.axis(\"off\")\n\n    for i, img_path in enumerate(samples):\n        ax = axes[i]\n        img = Image.open(img_path).convert(\"RGB\")\n        W, H = img.size\n        ax.imshow(np.array(img))\n        ax.set_axis_off()\n\n        # --- GT: labels/<split>/<same_subdir>/<stem>.txt を読む ---\n        rel = img_path.relative_to(Path(yolo_dataset_dir) / \"images\" / split)\n        gt_txt = Path(yolo_dataset_dir) / \"labels\" / split / rel.with_suffix(\".txt\")\n        gt_boxes = _yolo_txt_to_boxes(gt_txt, W, H)\n        for cls, x1, y1, x2, y2 in gt_boxes:\n            rect = Rectangle((x1, y1), x2 - x1, y2 - y1, fill=False, linewidth=1.5, edgecolor=\"g\")\n            ax.add_patch(rect)\n\n        # --- Prediction ---\n        res = model.predict(\n            source=str(img_path),\n            conf=conf, iou=iou, max_det=1, imgsz=img_size, verbose=False\n        )[0]\n        if res.boxes is not None and len(res.boxes) > 0:\n            xyxy = res.boxes.xyxy.cpu().numpy()\n            confs = res.boxes.conf.cpu().numpy()\n            clses = res.boxes.cls.cpu().numpy().astype(int)\n            for (x1, y1, x2, y2), cf, c in zip(xyxy, confs, clses):\n                rect = Rectangle((x1, y1), x2 - x1, y2 - y1, fill=False, linewidth=1.5, edgecolor=\"r\")\n                ax.add_patch(rect)\n                ax.text(x1, max(y1 - 5, 0), f\"{c}:{cf:.2f}\", color=\"r\", fontsize=9)\n\n        ax.set_title(f\"{split}/{rel.as_posix()}\", fontsize=9)\n\n    plt.tight_layout()\n    out_png = str(out_png)\n    plt.savefig(out_png, dpi=160)\n    plt.show()\n    print(f\"Saved prediction grid: {out_png}\")\n\n\ndef _unrotate_boxes_ccw_xyxy(xyxy, angle, W, H):\n    \"\"\"\n    xyxy: (N,4) numpy [x1,y1,x2,y2] （回転後フレーム座標）\n    angle: 0/90/180/270 (degree, CCW). We invert this to original W×H.\n    戻り値: (N,4) numpy （元フレーム座標）\n    変換は点ごとに逆写像してから min/max を取る実装。\n    \"\"\"\n    if len(xyxy) == 0:\n        return xyxy\n\n    def inv_map_point(xr, yr, ang):\n        if ang == 0:\n            # x = xr, y = yr\n            return xr, yr\n        elif ang == 90:\n            # 原→回: (xr,yr)=(y, W-x) の逆写像: (x,y)=(W-yr, xr)\n            return W - yr, xr\n        elif ang == 180:\n            # 原→回: (xr,yr)=(W-x, H-y) の逆: (x,y)=(W-xr, H-yr)\n            return W - xr, H - yr\n        elif ang == 270:\n            # 原→回: (xr,yr)=(H-y, x) の逆: (x,y)=(yr, H-xr)\n            return yr, H - xr\n        else:\n            raise ValueError(\"angle must be 0/90/180/270\")\n\n    out = []\n    for (x1, y1, x2, y2) in xyxy:\n        # 4コーナーを逆回転\n        pts = [\n            inv_map_point(x1, y1, angle),\n            inv_map_point(x2, y1, angle),\n            inv_map_point(x1, y2, angle),\n            inv_map_point(x2, y2, angle),\n        ]\n        xs = [p[0] for p in pts]\n        ys = [p[1] for p in pts]\n        nx1, ny1, nx2, ny2 = min(xs), min(ys), max(xs), max(ys)\n        # 画像境界にクリップ\n        nx1 = max(0, min(W - 1, nx1))\n        ny1 = max(0, min(H - 1, ny1))\n        nx2 = max(0, min(W - 1, nx2))\n        ny2 = max(0, min(H - 1, ny2))\n        out.append([nx1, ny1, nx2, ny2])\n    return np.asarray(out, dtype=np.float32)\n\n\n# ─────────────────────────────────────────────────────────────\n# 回転 TTA 推論（単一画像）\n# ─────────────────────────────────────────────────────────────\ndef predict_with_rotation_tta(\n    model,\n    image_path,\n    angles=(0, 90, 180, 270),  # 使いたい角度（CCW）\n    conf=0.25,\n    iou=0.7,\n    img_size=640,\n    max_det=300,\n    agnostic_nms=False,        # Trueならクラス無視でNMS\n    force_top1=False,          # Trueなら最後に1件に絞る（スコア最大）\n):\n    \"\"\"\n    画像を角度ごとに回転→推論→boxを元向きに戻して統合→NMS。\n    戻り値: dict {\"boxes\": Nx4, \"scores\": N, \"classes\": N}\n    \"\"\"\n    img = Image.open(image_path).convert(\"RGB\")\n    W, H = img.size\n\n    all_boxes = []\n    all_scores = []\n    all_classes = []\n\n    for ang in angles:\n        # 90°単位のCCW回転（PILの定数で高品質&正確）\n        if ang % 360 == 0:\n            img_r = img\n        elif ang % 360 == 90:\n            img_r = img.transpose(Image.ROTATE_90)   # 90°CCW\n        elif ang % 360 == 180:\n            img_r = img.transpose(Image.ROTATE_180)\n        elif ang % 360 == 270:\n            img_r = img.transpose(Image.ROTATE_270)  # 270°CCW=90°CW\n        else:\n            raise ValueError(\"angles must be multiples of 90\")\n\n        # 推論（回転後フレーム）\n        res = model.predict(\n            source=np.array(img_r),\n            conf=conf, iou=iou, imgsz=img_size, max_det=max_det, verbose=False\n        )[0]\n\n        if res.boxes is None or len(res.boxes) == 0:\n            continue\n\n        xyxy = res.boxes.xyxy.cpu().numpy()\n        scores = res.boxes.conf.cpu().numpy()\n        clses = res.boxes.cls.cpu().numpy().astype(int) if res.boxes.cls is not None else np.zeros(len(xyxy), int)\n\n        # 予測boxを元向き（W×H）に逆回転\n        xyxy_orig = _unrotate_boxes_ccw_xyxy(xyxy, ang % 360, W, H)\n\n        all_boxes.append(xyxy_orig)\n        all_scores.append(scores)\n        all_classes.append(clses)\n\n    if len(all_boxes) == 0:\n        return {\"boxes\": np.zeros((0,4), dtype=np.float32),\n                \"scores\": np.zeros((0,), dtype=np.float32),\n                \"classes\": np.zeros((0,), dtype=np.int32)}\n\n    boxes = np.vstack(all_boxes)\n    scores = np.hstack(all_scores)\n    clses = np.hstack(all_classes)\n\n    # ── 結合NMS（クラス別 or クラス無視） ──\n    t_boxes = torch.as_tensor(boxes, dtype=torch.float32)\n    t_scores = torch.as_tensor(scores, dtype=torch.float32)\n    if agnostic_nms:\n        keep = nms(t_boxes, t_scores, iou)\n    else:\n        # クラス別NMS：各クラスごとにNMS→結合\n        keep_idx = []\n        for c in np.unique(clses):\n            idx = np.where(clses == c)[0]\n            kept = nms(t_boxes[idx], t_scores[idx], iou)\n            keep_idx.append(idx[kept.cpu().numpy()])\n        keep = torch.as_tensor(np.concatenate(keep_idx), dtype=torch.long)\n\n    boxes = boxes[keep.numpy()]\n    scores = scores[keep.numpy()]\n    clses = clses[keep.numpy()]\n\n    # 上位 max_det に制限\n    if len(scores) > max_det:\n        order = np.argsort(-scores)[:max_det]\n        boxes, scores, clses = boxes[order], scores[order], clses[order]\n\n    # 1件だけ欲しい場合\n    if force_top1 and len(scores) > 0:\n        i = int(np.argmax(scores))\n        boxes = boxes[i:i+1]\n        scores = scores[i:i+1]\n        clses = clses[i:i+1]\n\n    return {\"boxes\": boxes, \"scores\": scores, \"classes\": clses}\n\n\ndef predict_on_samples_with_tta(\n    model,\n    yolo_dataset_dir: str,\n    split: str = \"val\",\n    num_samples: int = 8,\n    conf: float = 0.25,\n    iou: float = 0.7,\n    img_size: int = 640,\n    out_png: str = \"/kaggle/working/predictions_tta.png\",\n    angles=(0,90,180,270),\n    only_one: bool = False,          # ← 1件だけに絞る\n    agnostic_nms: bool = True,\n):\n    import glob, random, matplotlib.pyplot as plt\n\n    split_dir = Path(yolo_dataset_dir) / \"images\" / split\n    if not split_dir.exists():\n        split_dir = Path(yolo_dataset_dir) / \"images\" / \"train\"\n\n    image_paths = [Path(p) for p in glob.glob(str(split_dir / \"**\" / \"*.png\"), recursive=True)]\n    if not image_paths:\n        print(\"no images\")\n        return\n\n    samples = random.sample(image_paths, k=min(num_samples, len(image_paths)))\n    cols = int(np.ceil(np.sqrt(len(samples))))\n    rows = int(np.ceil(len(samples) / cols))\n    fig, axes = plt.subplots(rows, cols, figsize=(4.5*cols, 4.5*rows))\n    axes = np.array(axes).reshape(-1)\n\n    for ax in axes[len(samples):]:\n        ax.axis(\"off\")\n\n    for i, img_path in enumerate(samples):\n        ax = axes[i]\n        img = Image.open(img_path).convert(\"RGB\")\n        W, H = img.size\n        ax.imshow(np.array(img))\n        ax.set_axis_off()\n\n        # GT を描画（省略したい場合はこのブロックを消す）\n        rel = img_path.relative_to(Path(yolo_dataset_dir) / \"images\" / split)\n        gt_txt = Path(yolo_dataset_dir) / \"labels\" / split / rel.with_suffix(\".txt\")\n        gt_boxes = _yolo_txt_to_boxes(gt_txt, W, H) if '_yolo_txt_to_boxes' in globals() else []\n        for cls, x1, y1, x2, y2 in gt_boxes:\n            ax.add_patch(Rectangle((x1, y1), x2-x1, y2-y1, fill=False, linewidth=1.5, edgecolor=\"g\"))\n\n        # 回転TTA予測\n        pred = predict_with_rotation_tta(\n            model, str(img_path),\n            angles=angles, conf=conf, iou=iou, img_size=img_size,\n            max_det=1 if only_one else 300,\n            agnostic_nms=agnostic_nms,\n            force_top1=only_one,\n        )\n\n        for (x1,y1,x2,y2), sc, c in zip(pred[\"boxes\"], pred[\"scores\"], pred[\"classes\"]):\n            ax.add_patch(Rectangle((x1, y1), x2-x1, y2-y1, fill=False, linewidth=1.5, edgecolor=\"r\"))\n            ax.text(x1, max(0, y1-5), f\"{int(c)}:{sc:.2f}\", color=\"r\", fontsize=9)\n\n        ax.set_title(f\"{split}/{rel.as_posix()}\", fontsize=9)\n\n    plt.tight_layout()\n    plt.savefig(out_png, dpi=160)\n    plt.show()\n    print(f\"Saved TTA preview: {out_png}\")\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T05:59:28.330136Z","iopub.execute_input":"2025-09-01T05:59:28.330414Z","iopub.status.idle":"2025-09-01T05:59:38.538634Z","shell.execute_reply.started":"2025-09-01T05:59:28.330391Z","shell.execute_reply":"2025-09-01T05:59:38.537972Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport shutil\nfrom pathlib import Path\nimport pandas as pd\nimport yaml\n\ndef safe_link_or_copy(src: str, dst: str):\n    \"\"\"容量節約のためハードリンク→失敗時コピー\"\"\"\n    os.makedirs(os.path.dirname(dst), exist_ok=True)\n    if os.path.exists(dst):\n        return\n    try:\n        os.link(src, dst)\n    except Exception:\n        shutil.copy2(src, dst)\n\ndef yolo_distribute_by_fold(\n    df: pd.DataFrame,\n    yolo_dataset_dir: str,\n    val_fold: int,\n    include_unlabeled: bool = True,      # ラベル無でも空.txtを作って全保存\n    preserve_subdirs: bool = True,       # modality/plane/axis の階層を温存\n    class_names: dict = None,            # 例: {0: \"aneurysm\"}\n):\n    \"\"\"\n    DataFrame の 'fold' で分割:\n      - train: fold != val_fold\n      - val:   fold == val_fold\n    画像は df['filepath']、ラベルは同階層の 'labels_yolo' に .txt がある前提で、\n    /images/ → /labels_yolo/ に置換し拡張子 .png→.txt として推定。\n    \"\"\"\n    req_cols = [\"filepath\", \"filename\", \"fold\"]\n    for c in req_cols:\n        if c not in df.columns:\n            raise ValueError(f\"DataFrame に必須列 '{c}' がありません。\")\n\n    # 出力先作成\n    for sub in [\"images/train\", \"images/val\", \"labels/train\", \"labels/val\"]:\n        os.makedirs(os.path.join(yolo_dataset_dir, sub), exist_ok=True)\n\n    stats = { \"train\": {\"img\":0, \"lbl\":0, \"skipped\":0},\n              \"val\":   {\"img\":0, \"lbl\":0, \"skipped\":0} }\n    manifests = []\n\n    for _, row in df.iterrows():\n        split = \"val\" if int(row[\"fold\"]) == int(val_fold) else \"train\"\n\n        img_src = str(row[\"filepath\"])\n        # ラベルパス推定: /images/ → /labels_yolo/、拡張子 .png → .txt\n        if \"/images/\" in img_src:\n            lbl_src = img_src.replace(\"/images/\", \"/labels_yolo/\")\n        else:\n            # 念のため最後の 'images' を置換\n            parts = img_src.split(\"/\")\n            try:\n                last_idx = len(parts) - 1 - parts[::-1].index(\"images\")\n                parts[last_idx] = \"labels_yolo\"\n                lbl_src = \"/\".join(parts)\n            except ValueError:\n                # 見つからない場合は同ディレクトリ扱い\n                lbl_src = str(Path(img_src).with_suffix(\".txt\"))\n        if lbl_src.endswith(\".png\"):\n            lbl_src = lbl_src[:-4] + \".txt\"\n        else:\n            lbl_src = str(Path(lbl_src).with_suffix(\".txt\"))\n\n        # 出力先の相対サブパス\n        if preserve_subdirs and all(k in df.columns for k in [\"modality\",\"plane\",\"axis\"]):\n            subdir = f\"{row['modality']}/{row['plane']}/{row['axis']}\"\n        else:\n            subdir = \"\"\n\n        # 出力フルパス\n        dst_img = os.path.join(yolo_dataset_dir, \"images\", split, subdir, row[\"filename\"])\n        dst_lbl = os.path.join(\n            yolo_dataset_dir, \"labels\", split, subdir, Path(row[\"filename\"]).with_suffix(\".txt\").name\n        )\n\n        # 存在チェック\n        img_ok = os.path.exists(img_src)\n        lbl_ok = os.path.exists(lbl_src)\n\n        if not img_ok:\n            stats[split][\"skipped\"] += 1\n            continue\n\n        # 画像の配置\n        safe_link_or_copy(img_src, dst_img)\n        stats[split][\"img\"] += 1\n\n        # ラベルの配置（無ければ include_unlabeled に応じて空ファイル）\n        if lbl_ok:\n            safe_link_or_copy(lbl_src, dst_lbl)\n            stats[split][\"lbl\"] += 1\n        else:\n            if include_unlabeled:\n                os.makedirs(os.path.dirname(dst_lbl), exist_ok=True)\n                open(dst_lbl, \"a\").close()\n                # 空ラベルは 'lbl' にカウントしない\n            else:\n                stats[split][\"skipped\"] += 1\n                # 画像だけ残すのは避けたい場合は、必要に応じて画像削除も検討\n\n        manifests.append({\n            \"split\": split,\n            \"filepath\": img_src,\n            \"labelpath\": lbl_src if lbl_ok else None,\n            \"dst_img\": dst_img,\n            \"dst_lbl\": dst_lbl,\n            \"modality\": row.get(\"modality\"),\n            \"plane\": row.get(\"plane\"),\n            \"axis\": row.get(\"axis\"),\n            \"SeriesInstanceUID\": row.get(\"SeriesInstanceUID\"),\n            \"fold\": row[\"fold\"],\n        })\n\n    # dataset.yaml\n    if class_names is None:\n        class_names = {0: \"aneurysm\"}  # 必要に応じて変更\n    dataset_yaml = {\n        \"path\": yolo_dataset_dir,\n        \"train\": \"images/train\",\n        \"val\": \"images/val\",\n        \"names\": class_names,\n    }\n    yaml_path = os.path.join(yolo_dataset_dir, \"dataset.yaml\")\n    with open(yaml_path, \"w\") as f:\n        yaml.dump(dataset_yaml, f, sort_keys=False, allow_unicode=True)\n\n    # manifest 保存\n    manifest_csv = os.path.join(yolo_dataset_dir, \"manifest.csv\")\n    pd.DataFrame(manifests).to_csv(manifest_csv, index=False)\n\n    # サマリ\n    print(f\"[Split summary]  base: {yolo_dataset_dir}  (val_fold={val_fold})\")\n    for split in [\"train\",\"val\"]:\n        print(f\"  {split:5s} imgs: {stats[split]['img']:5d} | lbls: {stats[split]['lbl']:5d} | skipped: {stats[split]['skipped']:5d}\")\n    print(f\"YAML:     {yaml_path}\")\n    print(f\"Manifest: {manifest_csv}\")\n\n    return yaml_path, manifest_csv\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T05:59:38.540395Z","iopub.execute_input":"2025-09-01T05:59:38.540807Z","iopub.status.idle":"2025-09-01T05:59:38.558023Z","shell.execute_reply.started":"2025-09-01T05:59:38.540784Z","shell.execute_reply":"2025-09-01T05:59:38.557228Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def _read_yolo_txt(txt_path: Path):\n    boxes = []\n    if not txt_path.exists():\n        return boxes\n    for line in txt_path.read_text().splitlines():\n        line = line.strip()\n        if not line:\n            continue\n        parts = line.split()\n        if len(parts) < 5:\n            continue\n        cls = int(float(parts[0]))\n        cx, cy, w, h = map(float, parts[1:5])\n        rest = parts[5:]\n        boxes.append((cls, cx, cy, w, h, rest))\n    return boxes\n\ndef _write_yolo_txt(txt_path: Path, boxes):\n    lines = []\n    for cls, cx, cy, w, h, rest in boxes:\n        tail = (\" \" + \" \".join(rest)) if rest else \"\"\n        lines.append(f\"{cls} {cx:.6f} {cy:.6f} {w:.6f} {h:.6f}{tail}\")\n    txt_path.parent.mkdir(parents=True, exist_ok=True)\n    txt_path.write_text(\"\\n\".join(lines) + (\"\\n\" if lines else \"\"))\n\ndef _rot90_ccw_yolo(cls_cxcywh_list, angle_ccw: int):\n    \"\"\"YOLO正規化(cx,cy,w,h)を 0/90/180/270° CCWで回転\"\"\"\n    out = []\n    for cls, cx, cy, w, h, rest in cls_cxcywh_list:\n        if angle_ccw == 0:\n            nx, ny, nw, nh = cx, cy, w, h\n        elif angle_ccw == 90:\n            nx, ny, nw, nh = cy, 1.0 - cx, h, w\n        elif angle_ccw == 180:\n            nx, ny, nw, nh = 1.0 - cx, 1.0 - cy, w, h\n        elif angle_ccw == 270:\n            nx, ny, nw, nh = 1.0 - cy, cx, h, w\n        else:\n            raise ValueError(\"angle must be 0/90/180/270\")\n        nx = min(max(nx, 0.0), 1.0); ny = min(max(ny, 0.0), 1.0)\n        nw = min(max(nw, 0.0), 1.0); nh = min(max(nh, 0.0), 1.0)\n        out.append((cls, nx, ny, nw, nh, rest))\n    return out\n\ndef augment_train_with_right_angle_rotations(yolo_dataset_dir: str,\n                                             angles=(90, 180, 270),\n                                             suffix_map={90:\"r90\",180:\"r180\",270:\"r270\"},\n                                             overwrite=False):\n    \"\"\"\n    images/train, labels/train の全サンプルを指定角度で回転コピーし、対応するYOLOラベルも回転。\n    \"\"\"\n    base = Path(yolo_dataset_dir)\n    img_root = base / \"images\" / \"train\"\n    lbl_root = base / \"labels\" / \"train\"\n    assert img_root.exists(), f\"{img_root} not found\"\n\n    img_paths = list(img_root.rglob(\"*.png\")) + list(img_root.rglob(\"*.jpg\")) + list(img_root.rglob(\"*.jpeg\"))\n    print(f\"[augment] train images: {len(img_paths)} | angles={angles}\")\n\n    for img_path in img_paths:\n        rel = img_path.relative_to(img_root)\n        txt_path = lbl_root / rel.with_suffix(\".txt\")\n        boxes = _read_yolo_txt(txt_path)\n        img = Image.open(img_path).convert(\"RGB\")\n\n        for ang in angles:\n            tag = suffix_map.get(ang, f\"r{ang}\")\n            new_name = img_path.stem + f\"_{tag}\" + img_path.suffix\n            out_img = img_root / rel.parent / new_name\n            out_txt = lbl_root / rel.parent / (Path(new_name).with_suffix(\".txt\").name)\n\n            if out_img.exists() and not overwrite:\n                continue\n\n            if ang == 0: img_r = img\n            elif ang == 90:  img_r = img.transpose(Image.ROTATE_90)\n            elif ang == 180: img_r = img.transpose(Image.ROTATE_180)\n            elif ang == 270: img_r = img.transpose(Image.ROTATE_270)\n            else: raise ValueError\n\n            out_img.parent.mkdir(parents=True, exist_ok=True)\n            img_r.save(out_img)\n\n            if boxes:\n                boxes_r = _rot90_ccw_yolo(boxes, ang)\n                _write_yolo_txt(out_txt, boxes_r)\n            else:\n                out_txt.parent.mkdir(parents=True, exist_ok=True)\n                if not out_txt.exists():\n                    out_txt.write_text(\"\")\n\n    print(\"[augment] done.\")\n\n\ndef run_5fold_training_with_rot90_cv(\n    df: pd.DataFrame,\n    base_out_dir: str,\n    pretrained_weights_path: str,\n    *,\n    folds: int = 5,\n    angles=(90,180,270),\n    include_unlabeled: bool = True,\n    class_names: dict | None = None,\n    epochs: int = 30,\n    batch_size: int = 16,\n    img_size: int = 640,\n    rect: bool = True,                 # アスペクト比維持\n    translate: float = 0.05,           # 軽い平行移動でクロップ近似\n    scale: float = 0.15,               # 軽いズームでクロップ近似\n):\n    if class_names is None:\n        class_names = {0: \"aneurysm\"}\n\n    per_fold_rows = []\n    for k in range(folds):\n        print(f\"\\n===== Fold {k}/{folds} =====\")\n        yolo_dataset_dir = str(Path(base_out_dir) / f\"fold{k}\")\n\n        # 1) fold で分割（val=k, train≠k）\n        _yaml, _manifest = yolo_distribute_by_fold(\n            df,\n            yolo_dataset_dir=yolo_dataset_dir,\n            val_fold=k,\n            include_unlabeled=include_unlabeled,\n            preserve_subdirs=True,\n            class_names=class_names,\n        )\n\n        # 2) train のみ 90/180/270 増強を追加生成\n        augment_train_with_right_angle_rotations(yolo_dataset_dir, angles=angles)\n\n        # 3) dataset.yaml 準備\n        yaml_path = prepare_dataset(yolo_dataset_dir, class_names=class_names)\n\n        # 4) 学習（rect=True、mosaic/mixupは医用想定でOFF）\n        project_dir = str(Path(base_out_dir) / \"runs\")\n        run_name = f\"fold{k}\"\n        model, results, run_dir = train_yolo_model(\n            yaml_path=yaml_path,\n            pretrained_weights_path=pretrained_weights_path,\n            project_dir=project_dir,\n            run_name=run_name,\n            epochs=epochs, batch_size=batch_size, img_size=img_size,\n            rect=rect,\n            mosaic=0.0, mixup=0.0, copy_paste=0.0,\n            degrees=5.0, translate=translate, scale=scale, shear=0.0, perspective=0.0,\n            fliplr=0.5, flipud=0.0,\n            patience=8, save_period=5, seed=42, device=None,\n        )\n\n        # 5) 検証メトリクス取得\n        try:\n            metrics = model.val(data=yaml_path, imgsz=img_size, device=None, split=\"val\")\n            m = _extract_val_metrics(metrics)\n        except Exception as e:\n            print(f\"[fold{k}] val failed: {e}\")\n            m = {\"mAP50-95\": np.nan, \"mAP50\": np.nan, \"precision\": np.nan, \"recall\": np.nan}\n\n        # 6) results.csv 最終行（val損失など）\n        last = _read_last_results_row(run_dir)\n\n        # 7) val 枚数 / GT数\n        val_images, val_boxes = _count_split_images_and_boxes(yolo_dataset_dir, split=\"val\")\n\n        per_fold_rows.append({\n            \"fold\": k, \"run_dir\": run_dir, \"dataset_dir\": yolo_dataset_dir,\n            \"val_images\": val_images, \"val_boxes\": val_boxes,\n            **m, **last,\n        })\n\n        # 8) サンプル可視化（任意）\n        try:\n            # predict_on_samples(\n            #     model, yolo_dataset_dir=yolo_dataset_dir, split=\"val\",\n            #     num_samples=8, conf=0.25, iou=0.7, img_size=img_size,\n            #     out_png=str(Path(run_dir) / \"predictions_grid.png\"),\n            # )\n            predict_on_samples_with_tta(\n                model, yolo_dataset_dir=yolo_dataset_dir, split=\"val\",\n                num_samples=8, conf=0.25, iou=0.7, img_size=img_size,\n                out_png=str(Path(run_dir) / \"predictions_grid.png\"),\n                angles=(0,90,180,270),\n                only_one=True\n            )\n        except Exception as e:\n            print(f\"[fold{k}] preview failed: {e}\")\n\n    # 9) CV集計\n    cv_dir = str(Path(base_out_dir) / \"cv_summary\")\n    df_raw, df_summary = _summarize_cv(per_fold_rows, out_dir=cv_dir)\n    return {\"per_fold\": df_raw, \"summary\": df_summary, \"runs_dir\": str(Path(base_out_dir) / \"runs\"), \"cv_dir\": cv_dir}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T05:59:38.559491Z","iopub.execute_input":"2025-09-01T05:59:38.559754Z","iopub.status.idle":"2025-09-01T05:59:38.592794Z","shell.execute_reply.started":"2025-09-01T05:59:38.559732Z","shell.execute_reply":"2025-09-01T05:59:38.591941Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def parse_path(p):\n    p = Path(p)\n    # .../MRA/AXIAL/axis2/images/<file> という末尾構造を前提\n    modality = p.parts[-5]   # MRA\n    plane    = p.parts[-4]   # AXIAL\n    axis     = p.parts[-3]   # axis2\n    filename = p.name\n    return pd.Series([modality, plane, axis, filename],\n                     index=[\"modality\", \"plane\", \"axis\", \"filename\"])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T05:59:38.593556Z","iopub.execute_input":"2025-09-01T05:59:38.593795Z","iopub.status.idle":"2025-09-01T05:59:38.632725Z","shell.execute_reply.started":"2025-09-01T05:59:38.593764Z","shell.execute_reply":"2025-09-01T05:59:38.631895Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_comp = pd.read_csv('/kaggle/input/rsna2025-extra/train_add_metadata_v3.csv')\ndf_comp","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T05:59:38.633669Z","iopub.execute_input":"2025-09-01T05:59:38.633992Z","iopub.status.idle":"2025-09-01T05:59:42.284377Z","shell.execute_reply.started":"2025-09-01T05:59:38.633969Z","shell.execute_reply":"2025-09-01T05:59:42.283343Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"paths = list(Path('/kaggle/input/rsna-bbox-annotation-fix/results/detetion_annotation').glob('*/*/*/images/*'))\ndf = pd.DataFrame({'filepath': paths})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T05:59:42.285365Z","iopub.execute_input":"2025-09-01T05:59:42.285671Z","iopub.status.idle":"2025-09-01T05:59:42.722252Z","shell.execute_reply.started":"2025-09-01T05:59:42.285646Z","shell.execute_reply":"2025-09-01T05:59:42.721374Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df[[\"modality\",\"plane\",\"axis\",\"filename\"]] = df['filepath'].apply(parse_path)\ndf['SeriesInstanceUID'] = df['filename'].apply(lambda x: x[:-4])\ndf","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T05:59:42.7246Z","iopub.execute_input":"2025-09-01T05:59:42.724841Z","iopub.status.idle":"2025-09-01T05:59:42.827694Z","shell.execute_reply.started":"2025-09-01T05:59:42.724821Z","shell.execute_reply":"2025-09-01T05:59:42.826951Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = df.merge(\n    df_comp[[\"SeriesInstanceUID\", \"fold\"]].drop_duplicates(\"SeriesInstanceUID\"),\n    on=\"SeriesInstanceUID\",\n    how=\"left\",\n    validate=\"m:1\",          # 1つのSeriesInstanceUIDにfoldが1件であることを保証\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T05:59:42.828677Z","iopub.execute_input":"2025-09-01T05:59:42.82893Z","iopub.status.idle":"2025-09-01T05:59:42.883973Z","shell.execute_reply.started":"2025-09-01T05:59:42.828908Z","shell.execute_reply":"2025-09-01T05:59:42.88328Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Axialを削除　ラベルエラーがあるので目視で確認済み\ndrop_patterns = [('AXIAL', 'axis2'),\n                ('SAGITTAL', 'axis2'),\n                ('CORONAL', 'axis0')\n               ]\nfor pattern in drop_patterns:\n    plane, axis = pattern\n    df = df[~((df['plane']==plane)&(df['axis']==axis))]\ndf = df.reset_index(drop=True)\ndf","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T05:59:42.884803Z","iopub.execute_input":"2025-09-01T05:59:42.885136Z","iopub.status.idle":"2025-09-01T05:59:42.900793Z","shell.execute_reply.started":"2025-09-01T05:59:42.885105Z","shell.execute_reply":"2025-09-01T05:59:42.899954Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# cv_result = run_5fold_training_with_cv(\n#     df,\n#     base_out_dir=\"/kaggle/working/yolo_cv\",\n#     pretrained_weights_path=\"yolov8n.pt\",\n#     epochs=30, batch_size=16, img_size=320,\n#     folds=5, include_unlabeled=True, rect=True,\n#     class_names={0: \"aneurysm\"},\n# )\ncv_result = run_5fold_training_with_rot90_cv(\n    df,\n    base_out_dir=\"/kaggle/working/yolo_cv_rot90\",\n    pretrained_weights_path=\"yolov8n.pt\",  # 例: 任意のv8/v11など\n    folds=5,\n    angles=(90,180,270),       # 90°回転TTAではなく“学習データの増強”\n    include_unlabeled=True,    # 空txtを作ってでも全件保存する場合\n    class_names={0: \"aneurysm\"},\n    epochs=30, batch_size=16, img_size=320,\n    rect=True,\n)\n\n# 集計の DataFrame\ndisplay(cv_result[\"per_fold\"])   # 各 fold の生メトリクス\ndisplay(cv_result[\"summary\"])    # CV の mean / std / 重み付き平均","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T06:00:30.898888Z","iopub.execute_input":"2025-09-01T06:00:30.899257Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}