Download baim/bench_runtime.py from devildasdf/devils-agent: direct link, hf CLI and curl.
- Browser
- Download file 4.52 kB
-
https://huggingface.co/devildasdf/devils-agent/resolve/main/baim/bench_runtime.py
- Command line
-
hf download hf://devildasdf/devils-agent/baim/bench_runtime.py
-
curl -L -o bench_runtime.py https://huggingface.co/devildasdf/devils-agent/resolve/main/baim/bench_runtime.py
4.52 kB
| """Reproducible observation/execution microbenchmark; not policy task success.""" | |
| import argparse | |
| import json | |
| import platform | |
| import statistics | |
| import threading | |
| from pathlib import Path | |
| from time import perf_counter, process_time | |
| import psutil | |
| from playwright.sync_api import sync_playwright | |
| from .actions import Action, Kind, Decision | |
| from .authority import Authority | |
| from .browser import Browser | |
| from .runtime import Runtime | |
| def quantiles(values): | |
| ordered = sorted(values) | |
| return dict(median=statistics.median(values), p95=ordered[min(len(ordered)-1, int(.95*len(ordered)))]) | |
| def main(): | |
| parser = argparse.ArgumentParser() | |
| parser.add_argument('--output', default='reports/runtime-baseline.json') | |
| parser.add_argument('--repeats', type=int, default=20) | |
| parser.add_argument('--path-mode', choices=['traversal', 'naive'], default='traversal') | |
| args = parser.parse_args() | |
| process = psutil.Process() | |
| stop = threading.Event() | |
| peaks = {'python_rss_bytes': 0, 'process_tree_rss_bytes': 0} | |
| def sample(): | |
| while not stop.is_set(): | |
| own = process.memory_info().rss | |
| total = own | |
| for child in process.children(recursive=True): | |
| try: | |
| total += child.memory_info().rss | |
| except psutil.Error: | |
| pass | |
| peaks['python_rss_bytes'] = max(peaks['python_rss_bytes'], own) | |
| peaks['process_tree_rss_bytes'] = max(peaks['process_tree_rss_bytes'], total) | |
| stop.wait(.01) | |
| monitor = threading.Thread(target=sample, daemon=True) | |
| monitor.start() | |
| rows = [] | |
| try: | |
| with sync_playwright() as pw: | |
| with pw.chromium.launch(headless=True) as chromium: | |
| context = chromium.new_context(viewport={'width':1280, 'height':900}) | |
| browser = Browser(context, path_mode=args.path_mode) | |
| authority = Authority() | |
| for size in (40, 200, 2000): | |
| browser.page.set_content('<main>' + ''.join( | |
| f'<button onclick="this.dataset.activated=1">Record {i}</button>' for i in range(size)) + '</main>') | |
| browser.observe() # Warm up one observation. | |
| observations, cpu_times, actions, sizes = [], [], [], [] | |
| for trial in range(args.repeats): | |
| authority.replace(f'Benchmark fixture {trial}') | |
| runtime = Runtime(browser, authority, lambda *_: True) | |
| start, cpu = perf_counter(), process_time() | |
| state = browser.observe() | |
| observations.append((perf_counter()-start)*1000) | |
| cpu_times.append((process_time()-cpu)*1000) | |
| sizes.append(len(json.dumps(state.compact(), ensure_ascii=False).encode())) | |
| decision = Decision(authority.ticket(state), Action(Kind.CLICK, ('e0',))) | |
| outcome = runtime.execute(decision) | |
| if outcome.status != 'ok': | |
| raise RuntimeError(outcome.code) | |
| actions.append(outcome.wall_ms) | |
| rows.append(dict(elements=size, repeats=args.repeats, | |
| observation_wall_ms=quantiles(observations), | |
| observation_python_cpu_ms=quantiles(cpu_times), | |
| validated_click_wall_ms=quantiles(actions), | |
| compact_json_bytes=statistics.median(sizes))) | |
| context.close() | |
| finally: | |
| stop.set() | |
| monitor.join() | |
| report = dict(benchmark='runtime_microbenchmark_v1', path_mode=args.path_mode, platform=platform.platform(), | |
| cpu=platform.processor(), logical_cpus=psutil.cpu_count(), | |
| physical_cpus=psutil.cpu_count(logical=False), | |
| total_ram_bytes=psutil.virtual_memory().total, | |
| memory='10ms sampled peak RSS; sum includes shared pages, not unique memory', | |
| cpu_time='Python process only; excludes browser subprocess CPU', | |
| policy='test fixture actions, no learned policy', target_vps_validated=False, | |
| sampled_peaks=peaks, results=rows) | |
| path = Path(args.output) | |
| path.parent.mkdir(parents=True, exist_ok=True) | |
| path.write_text(json.dumps(report, indent=2), encoding='utf-8') | |
| print(json.dumps(report, indent=2)) | |
| if __name__ == '__main__': | |
| main() | |