| """Propose catalog quality updates from Artificial Analysis. |
| |
| Authoring-time helper β NEVER called at runtime (their terms forbid |
| client-side keys, the fleet would burn the rate limit, and a |
| recommendation must not change because a third-party endpoint |
| hiccuped). Run it when adding a model or refreshing the ordering; |
| review the printed diff and edit catalog.json yourself. The script |
| proposes, the commit decides. |
| |
| The catalog's `quality` stays OUR field: AA-informed where they cover a |
| model, editorially set where they don't (day-0 releases lag their evals; |
| some entries never appear). AA's Intelligence Index grades the |
| full-precision cloud model, not our Q4 build β fine for ordering, never |
| for display. |
| |
| Usage: |
| export AA_API_KEY=... # from https://artificialanalysis.ai (free tier) |
| python scripts/aa_quality_sync.py |
| |
| Attribution: scores by Artificial Analysis (https://artificialanalysis.ai). |
| """ |
|
|
| from __future__ import annotations |
|
|
| import json |
| import os |
| import sys |
| import urllib.request |
| from pathlib import Path |
|
|
| REPO_ROOT = Path(__file__).resolve().parent.parent |
| CATALOG_PATH = REPO_ROOT / "hermes_cli" / "local_runtime" / "catalog.json" |
| AA_URL = "https://artificialanalysis.ai/api/v2/data/llms/models" |
|
|
| |
| |
| |
| AA_SLUG_BY_ENTRY = { |
| "qwen3.8-27b": "qwen3-8-27b", |
| "qwen3.8-flash-next": "qwen3-8-flash-next", |
| "qwen3.6-35b-a3b": "qwen3-6-35b-a3b", |
| "deepseek-v4-flash": "deepseek-v4-flash", |
| } |
|
|
|
|
| def fetch_aa_models(api_key: str) -> dict[str, dict]: |
| req = urllib.request.Request(AA_URL, headers={"x-api-key": api_key}) |
| with urllib.request.urlopen(req, timeout=30) as r: |
| doc = json.load(r) |
| return {m["slug"]: m for m in doc.get("data", [])} |
|
|
|
|
| def main() -> int: |
| api_key = os.environ.get("AA_API_KEY", "").strip() |
| if not api_key: |
| print("AA_API_KEY not set β create a free key at " |
| "https://artificialanalysis.ai and export it.", file=sys.stderr) |
| return 2 |
|
|
| catalog = json.loads(CATALOG_PATH.read_text(encoding="utf-8")) |
| aa = fetch_aa_models(api_key) |
|
|
| print(f"{'entry':24s} {'catalog q':>9s} {'AA index':>9s} note") |
| print("-" * 70) |
| for model in catalog["models"]: |
| entry_id = model["id"] |
| current = model.get("quality", 0) |
| slug = AA_SLUG_BY_ENTRY.get(entry_id) |
| if not slug: |
| print(f"{entry_id:24s} {current:>9d} {'β':>9s} editorial only (no AA mapping)") |
| continue |
| hit = aa.get(slug) |
| if hit is None: |
| print(f"{entry_id:24s} {current:>9d} {'β':>9s} not in AA data (slug {slug!r})") |
| continue |
| index = (hit.get("evaluations") or {}).get( |
| "artificial_analysis_intelligence_index") |
| if index is None: |
| print(f"{entry_id:24s} {current:>9d} {'β':>9s} AA row lacks the index") |
| continue |
| proposed = round(float(index)) |
| marker = "" if proposed == current else " <-- proposes change" |
| print(f"{entry_id:24s} {current:>9d} {proposed:>9d}{marker}") |
|
|
| print("\nReview against the decision table before editing: a quality " |
| "change that flips cells in tests/hermes_cli/" |
| "test_local_recommendation.py is the actual decision being made.") |
| print("Attribution: scores by Artificial Analysis " |
| "(https://artificialanalysis.ai).") |
| return 0 |
|
|
|
|
| if __name__ == "__main__": |
| sys.exit(main()) |
|
|