File size: 3,294 Bytes
3cb9c50
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
d4a30e5
 
3cb9c50
 
 
 
d4a30e5
 
 
3cb9c50
 
 
 
 
 
 
 
 
 
 
 
 
 
 
d4a30e5
 
 
 
 
3cb9c50
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
"""Evaluate the trained model on the val (or test) split and dump per-class metrics.

Run after re-downloading the dataset and editing work/data.yaml so that the
`val:` / `test:` paths point at the local copy. From repo root:

    uv run python scripts/run_eval.py
    uv run python scripts/run_eval.py --split test
    uv run python scripts/run_eval.py --imgsz 640 --batch 16

Output:
- prints overall and per-class precision / recall / mAP@0.5 / mAP@0.5:0.95
- writes runs/detect/val/ with confusion matrices and curves (or runs/detect/val{N}/ if N>0)
- writes docs/eval_<split>.md ready to paste-replace the corresponding section of docs/model_card.md
"""

from __future__ import annotations

import argparse
from pathlib import Path

from ultralytics import YOLO

WEIGHTS = "runs/detect/yolov8n_v1_train/weights/best.pt"
DATA = "work/data.yaml"


def main() -> None:
    p = argparse.ArgumentParser(description=__doc__)
    p.add_argument("--weights", default=WEIGHTS)
    p.add_argument("--data", default=DATA)
    p.add_argument("--split", choices=["val", "test"], default="val")
    p.add_argument("--imgsz", type=int, default=640)
    p.add_argument("--batch", type=int, default=16)
    p.add_argument("--device", default=None, help="e.g. '0' for GPU, 'cpu', or omit to auto-pick")
    args = p.parse_args()

    model = YOLO(args.weights)
    metrics = model.val(
        data=args.data,
        split=args.split,
        imgsz=args.imgsz,
        batch=args.batch,
        device=args.device,
        plots=True,
    )

    names = model.names
    overall_p, overall_r = metrics.box.mp, metrics.box.mr
    overall_map50, overall_map = metrics.box.map50, metrics.box.map
    per_class_p = metrics.box.p
    per_class_r = metrics.box.r
    per_class_map50 = metrics.box.ap50
    per_class_map = metrics.box.ap
    # Ultralytics orders the per-class arrays by ap_class_index (only classes that
    # appeared in the eval), NOT by class id — so map array position -> class id.
    ap_class_index = list(metrics.box.ap_class_index)

    md_lines: list[str] = []
    md_lines.append(f"## Validation metrics ({args.split} split)\n")
    md_lines.append(f"_Image size {args.imgsz}, batch {args.batch}, generated by `scripts/run_eval.py`._\n")
    md_lines.append("\n### Overall\n")
    md_lines.append("| Metric | Value |")
    md_lines.append("|---|---|")
    md_lines.append(f"| mAP@0.5 | {overall_map50:.3f} |")
    md_lines.append(f"| mAP@0.5:0.95 | {overall_map:.3f} |")
    md_lines.append(f"| Precision | {overall_p:.3f} |")
    md_lines.append(f"| Recall | {overall_r:.3f} |\n")

    md_lines.append("### Per-class\n")
    md_lines.append("| Class | Precision | Recall | mAP@0.5 | mAP@0.5:0.95 |")
    md_lines.append("|---|---|---|---|---|")
    for arr_i, class_id in enumerate(ap_class_index):
        name = names[class_id]
        prec, rec = per_class_p[arr_i], per_class_r[arr_i]
        map50, mapfull = per_class_map50[arr_i], per_class_map[arr_i]
        md_lines.append(f"| {name} | {prec:.3f} | {rec:.3f} | {map50:.3f} | {mapfull:.3f} |")

    md_path = Path("docs") / f"eval_{args.split}.md"
    md_path.parent.mkdir(parents=True, exist_ok=True)
    md_path.write_text("\n".join(md_lines) + "\n")
    print(f"\nWrote {md_path}")


if __name__ == "__main__":
    main()