Download source/scripts/publishing/prepare_hf_v21.py from Voltline/vimeml-tiny-ja-v2.1: direct link, hf CLI and curl.
- Browser
- Download file 9.49 kB
-
https://huggingface.co/Voltline/vimeml-tiny-ja-v2.1/resolve/main/source/scripts/publishing/prepare_hf_v21.py
- Command line
-
hf download hf://Voltline/vimeml-tiny-ja-v2.1/source/scripts/publishing/prepare_hf_v21.py
-
curl -L -o prepare_hf_v21.py https://huggingface.co/Voltline/vimeml-tiny-ja-v2.1/resolve/main/source/scripts/publishing/prepare_hf_v21.py
9.49 kB
| """Package the selected V2.1 FP32 bundle, INT8 Core ML model and public metrics.""" | |
| import argparse | |
| import hashlib | |
| import json | |
| import re | |
| import shutil | |
| import subprocess | |
| from datetime import datetime, timezone | |
| from pathlib import Path | |
| ROOT = Path(__file__).resolve().parents[2] | |
| BUNDLE = ROOT / "artifacts/deployment/tiny-ja-v2.1-extend5-bundle-v1" | |
| COREML = ROOT / "artifacts/deployment/tiny-ja-v2.1-extend5-int8-b32-v1" | |
| TEMPLATES = ROOT / "templates/huggingface-v21" | |
| def read(path): | |
| return json.loads(Path(path).read_text(encoding="utf-8")) | |
| def write(path, value): | |
| path.parent.mkdir(parents=True, exist_ok=True) | |
| path.write_text(json.dumps(value, ensure_ascii=False, indent=2, allow_nan=False) + "\n", | |
| encoding="utf-8", newline="\n") | |
| def sha(path): | |
| with Path(path).open("rb") as stream: | |
| return hashlib.file_digest(stream, "sha256").hexdigest() | |
| def git(*args): | |
| return subprocess.check_output(["git", *args], cwd=ROOT).decode("utf-8").strip() | |
| def evaluation_summary(bundle, coreml): | |
| reports, ranking = {}, {} | |
| def report(relative): | |
| path = ROOT / relative | |
| reports[relative] = sha(path) | |
| return read(path) | |
| for role, expected in {"ajimee": (200, 144), "development": (137, 122), | |
| "expanded-development": (2000, 1481)}.items(): | |
| data = report(f"outputs/deployment/v21-int8-{role}-v1/metrics.json") | |
| if (data["model"]["checkpoint_sha256"] != bundle["checkpoint_sha256"] | |
| or data["model"]["tokenizer_sha256"] != bundle["files"]["tokenizer.model"]["sha256"] | |
| or data["model"]["coreml_manifest_sha256"] != coreml): | |
| raise ValueError("Saved quality report belongs to another model.") | |
| metrics = data["metrics"]["all"]["lm_context_sum"] | |
| if (metrics["cases"], metrics["top1_correct"]) != expected: | |
| raise ValueError("Metrics differ from the V2.1 model card.") | |
| ranking[role] = { | |
| "original_order": data["metrics"]["all"]["azookey"], | |
| "fp32_lm": data["baseline_metrics"]["all"]["lm_context_sum"], | |
| "int8_lm": metrics, | |
| "fp32_comparison": {k: v for k, v in data["comparison"].items() if k != "changes"}, | |
| "labels_formal_gold": data.get("labels_formal_gold"), | |
| } | |
| alignment = {} | |
| for kind in ("fp32", "int8"): | |
| data = report(f"outputs/deployment/v21-{kind}-alignment-v1/alignment.json") | |
| alignment[kind] = {"strict_passed": data["passed"], "tolerances": data["tolerances"], | |
| "logits": data["logits"], "invariants": data["invariants"]} | |
| acceptance = report("outputs/deployment/v21-closeout-v1/acceptance.json") | |
| device = acceptance["physical_keyboard_extension"].copy() | |
| device.pop("manual_confirmation", None) | |
| return {"format": "vimeml_public_evaluation_v2", "model_version": "2.1-extend5-step40000-int8-b32-v1", | |
| "checkpoint_sha256": bundle["checkpoint_sha256"], "coreml_manifest_sha256": coreml, | |
| "training": {"full_validation_bpc_fp32": 3.0802686334, "test_used_for_selection": False, | |
| "v20_epochs": 4, "restart1_epochs": 2, "extend5_stopped_step": 161095, | |
| "selected_extend5_step": 40000}, | |
| "ranking": ranking, "alignment": alignment, | |
| "fixed_generation": acceptance["same_pool_quality"]["fixed_generation"], | |
| "quantization_diagnostics": report("outputs/deployment/v21-quantization-analysis-v1/analysis.json"), | |
| "physical_keyboard_extension": device, | |
| "limitations": acceptance["failed_or_unresolved"], | |
| "experiment_decision": "V2 series complete; current INT8 task-quality loss and observed UI tail accepted.", | |
| "source_reports_sha256": reports, | |
| "scope": "Saved fixed-pool metrics and bounded device workload; draft expanded labels, no blind scoring. " | |
| "No benchmark texts, per-case scores, user inputs or raw device traces distributed."} | |
| def prepare(output, repo_id): | |
| if not re.fullmatch(r"[A-Za-z0-9_.-]+/[A-Za-z0-9_.-]+", repo_id): | |
| raise ValueError("Use OWNER/REPO.") | |
| if output.exists(): | |
| raise ValueError("Output exists; use a new release directory.") | |
| if git("status", "--porcelain"): | |
| raise ValueError("Commit the source before creating the release snapshot.") | |
| bundle, model = read(BUNDLE / "manifest.json"), read(COREML / "manifest.json") | |
| bundle_sha, coreml_sha = sha(BUNDLE / "manifest.json"), sha(COREML / "manifest.json") | |
| if (bundle["format"] != "vimeml_inference_bundle_v2" or bundle["checkpoint_step"] != 40000 | |
| or bundle["parameter_count"] != 12537920 or not bundle["tied_lm_head"] | |
| or model["kind"] != "linear8_fp32_compute" or model["minimum_ios"] != 18 | |
| or model["package_bytes"] != 14330856 | |
| or model["bundle"]["bundle_manifest_sha256"] != bundle_sha): | |
| raise ValueError("Expected selected V2.1 step40000 and its INT8 package.") | |
| summary = evaluation_summary(bundle, coreml_sha) | |
| commit, known = git("rev-parse", "HEAD"), {} | |
| output.mkdir(parents=True, exist_ok=False) | |
| def copy(source, name, digest=None, size=None): | |
| if source.is_symlink() or not source.is_file() or (size is not None and source.stat().st_size != size): | |
| raise ValueError(f"Invalid source file: {source}") | |
| target = output / name | |
| target.parent.mkdir(parents=True, exist_ok=True) | |
| shutil.copyfile(source, target) | |
| if target.stat().st_size != source.stat().st_size: | |
| raise ValueError("Copy size mismatch.") | |
| if digest: | |
| known[name] = digest | |
| for name, entry in bundle["files"].items(): | |
| copy(BUNDLE / name, f"inference/{name}", entry["sha256"], entry["bytes"]) | |
| copy(BUNDLE / "manifest.json", "inference/manifest.json", bundle_sha) | |
| for name, entry in model["package_files"].items(): | |
| copy(COREML / "model.mlpackage" / name, f"coreml/ios18-int8-block32/model.mlpackage/{name}", | |
| entry.get("sha256"), entry["bytes"]) | |
| copy(COREML / "manifest.json", "coreml/ios18-int8-block32/manifest.json", coreml_sha) | |
| source_files = subprocess.check_output( | |
| ["git", "ls-files", "-z", "--", "src", "scripts", "configs", "templates/huggingface-v21", | |
| "requirements.txt", "requirements-autodl.txt", "requirements-evaluation.txt", "LICENSE"], cwd=ROOT | |
| ).decode("utf-8").split("\0") | |
| for name in filter(None, source_files): | |
| copy(ROOT / name, f"source/{name}") | |
| (output / "source/README.md").write_text( | |
| f"VimeML source/configuration from Git commit {commit}. GPL-2.0; see LICENSE.\n" | |
| "V2 inference uses vimeml.deployment.v2.BundleLM and the original RMSNorm/SwiGLU architecture.\n" | |
| "Corpus texts, optimizer state, benchmark pools, device traces and the Vime client are excluded.\n", | |
| encoding="utf-8", newline="\n") | |
| copy(ROOT / "LICENSE", "LICENSE") | |
| for name in ("infer.py", "verify_release.py"): | |
| copy(TEMPLATES / name, name) | |
| card = (TEMPLATES / "README.md").read_text(encoding="utf-8") | |
| for key, value in {"SOURCE_COMMIT": commit, "BUNDLE_SHA": bundle_sha, "COREML_SHA": coreml_sha, | |
| "TOKENIZER_SHA": bundle["files"]["tokenizer.model"]["sha256"], | |
| "CHECKPOINT_SHA": bundle["checkpoint_sha256"]}.items(): | |
| card = card.replace(f"@@{key}@@", value) | |
| if "@@" in card: | |
| raise ValueError("Unresolved model card field.") | |
| (output / "README.md").write_text(card, encoding="utf-8", newline="\n") | |
| write(output / "evaluation/summary.json", summary) | |
| # Reuse frozen bundle hashes; hash each new release file at most once here. | |
| files = {p.relative_to(output).as_posix(): {"bytes": p.stat().st_size, | |
| "sha256": known.get(p.relative_to(output).as_posix()) or sha(p)} | |
| for p in sorted(output.rglob("*")) if p.is_file()} | |
| release = {"format": "vimeml_hf_release_v1", "repo_id": repo_id, "license": "gpl-2.0", | |
| "model_version": "2.1", "architecture": "tiny_gpt_v2", | |
| "source_repository": "https://github.com/Voltline/VimeML", "source_commit": commit, | |
| "created_utc": datetime.now(timezone.utc).isoformat(), | |
| "bundle_manifest_sha256": bundle_sha, "coreml_manifest_sha256": coreml_sha, | |
| "files": files, "policy": "FP32 inference and weight-only INT8; original manifests preserved. " | |
| "Frozen bundle digests reused, new files inventoried once; no training state or credentials."} | |
| write(output / "RELEASE.json", release) | |
| sums = {**{name: entry["sha256"] for name, entry in files.items()}, | |
| "RELEASE.json": sha(output / "RELEASE.json")} | |
| (output / "SHA256SUMS.txt").write_text( | |
| "".join(f"{digest} {name}\n" for name, digest in sorted(sums.items())), encoding="utf-8", newline="\n") | |
| print(json.dumps({"release": str(output), "files": len(files) + 2, | |
| "bytes": sum(p.stat().st_size for p in output.rglob("*") if p.is_file()), | |
| "source_commit": commit}, indent=2)) | |
| if __name__ == "__main__": | |
| parser = argparse.ArgumentParser(description=__doc__) | |
| parser.add_argument("--output", type=Path, required=True) | |
| parser.add_argument("--repo-id", default="Voltline/vimeml-tiny-ja-v2.1") | |
| args = parser.parse_args() | |
| prepare(args.output.resolve(), args.repo_id) | |