"""Strict, offline validator for this local six-claim package. This deliberately is *not* described as a campaign/Hub validator: no remote publication is part of this run. It validates only the files present in this directory and writes deterministic machine-readable results. """ from __future__ import annotations import argparse import hashlib import json import subprocess from pathlib import Path ROOT = Path(__file__).resolve().parent OUT = ROOT / "outputs" EXPECTED_ARCHIVE = "e8d22bfd259aaa60385841d8643109ecb66f7eb1081dd76429f5215f05a032e8" EXPECTED_CODE = "6be8f8fbc2169290af6f4ba5e4bd53a5c6485f7b" EXPECTED_CLAIMS = [ "Theorem 3.3 proves that any k-layer state-space model solving the function-composition tasks under injectivity conditions must have total log state-space size scaling as Ω(m·log|V| − q·log|Y|), linear in the hidden dimension m (Theorem 3.3).", "Theorem 3.7 proves that sliding-window Transformers solving the same tasks under a local-sensitivity condition require total window size scaling with the context-dependency range R (Theorem 3.7).", "Theorem 4.3 constructs a two-layer hybrid (Mamba + attention) model that solves the selective copying task using embedding dimension O(max(log|V|, log L)) and working memory Õ(N), versus Ω(L) required by pure Transformers (Theorem 4.3).", "Theorem 4.6 constructs a three-layer hybrid model that achieves 99% accuracy on the associative recall task using embedding dimension O(max(log|V|, log L)) and window size Õ(|V|) (Theorem 4.6).", "On the selective copying task, the learned hybrid model reaches perfect accuracy with roughly 2,000 parameters while pure Transformer/SSM models need roughly 12,000 parameters to match it, a 6x parameter gap (Figure 4).", "On multi-key associative recall, the hybrid model reaches 60% accuracy using 6x fewer parameters than pure Transformers, which plateau near 40% accuracy on single-key associative recall (Figures 5-6).", ] def sha(path: Path) -> str: return hashlib.sha256(path.read_bytes()).hexdigest() def read(name: str) -> dict: return json.loads((OUT / name).read_text()) def main() -> None: parser = argparse.ArgumentParser() parser.add_argument("--require-replay", action="store_true") args = parser.parse_args() c1, c2, c3, c4 = (read(f"claim{i}.json") for i in range(1, 5)) c5, c6 = read("claim5.json"), read("claim6.json") n3, n4 = read("claim3_native.json"), read("claim4_native.json") claims = json.loads((ROOT / "CLAIMS.json").read_text())["claims"] texts = [x["text"] for x in claims] provenance = json.loads((ROOT / "SOURCE_PROVENANCE.json").read_text()) logbook = json.loads((ROOT / "logbook.json").read_text()) checks: list[dict] = [] def check(name: str, ok: bool, detail: str) -> None: checks.append({"name": name, "passed": bool(ok), "detail": detail}) check("frozen_six_claims", texts == EXPECTED_CLAIMS, "CLAIMS.json exactly equals the six registered claim strings.") route_files = [ROOT / "pages/index.md", ROOT / "pages/executive-summary/page.md"] + [ ROOT / "pages" / x / "page.md" for x in [ "claim-1-theorem-3-3-literal-bound", "claim-2-theorem-3-7-window-bound", "claim-3-theorem-4-3-selective-copy", "claim-4-theorem-4-6-associative-recall", "claim-5-figure-4-selective-copy-learning", "claim-6-figures-5-6-associative-recall-learning", ] ] check("complete_eight_routes", len(logbook["root"]["children"]) == 7 and all(p.is_file() for p in route_files), "Index, executive summary, and six claim routes exist.") check("exact_authored_archive", sha(ROOT / "source/2603.08859v1.tar.gz") == EXPECTED_ARCHIVE and provenance["archive_sha256"] == EXPECTED_ARCHIVE, "arXiv e-print archive matches the retrieval SHA-256.") source_checkout = ROOT / "source/official-code" embedded_git = (source_checkout / ".git").exists() head = subprocess.check_output(["git", "-C", str(source_checkout), "rev-parse", "HEAD"], text=True).strip() if embedded_git else None # official-code is a clean subdirectory of this package repository; scope # the status query to that path so page edits do not masquerade as edits # to the pinned official checkout. status = subprocess.check_output(["git", "-C", str(ROOT), "status", "--porcelain", "--", "source/official-code"], text=True).strip() code_identity_ok = (head == EXPECTED_CODE) if embedded_git else True code_detail = ("Linked source repository is detached at the recorded clean commit." if embedded_git else "The linked source is a clean tracked snapshot without embedded Git metadata; external repository commit 6be8f8fbc2169290af6f4ba5e4bd53a5c6485f7b remains recorded in SOURCE_PROVENANCE.json.") check("pinned_clean_code_checkout", code_identity_ok and not status, code_detail) check("claim1_literal_falsification", c1["printed_bound_audit"]["max_literal_bound_over_admissible_grid"] <= 0 and c1["one_state_guessing_accuracy"]["2"] == 0.5 and all(row["A_star"]["1"] == 0.5 for row in c1["exhaustive_accuracy_curves"][:2]), "Under injectivity the printed RHS is non-positive; one-state 1/2 witnesses are retained.") check("claim2_receptive_field_witnesses", c2["receptive_field_sweep"]["violations_outside_receptive_field"] == 0 and c2["max_abs_delta_when_sumW_below_R"] == 0.0 and c2["control_full_window_separates_frac"] == 1.0, "Real causal-attention stacks preserve the designed outside-window witness.") check("claim3_construction_and_controls", c3["min_accuracy_over_all_configs"] == 1.0 and c3["total_inputs_tested"] >= 1_500_000 and all(x["full_construction_accuracy"] == 1.0 and x["control_window_minus_1_accuracy"] < 1.0 and x["control_no_ssm_query_accuracy"] < 1.0 for x in c3["negative_controls"]), "Independent finite construction succeeds while destructive controls fail.") check("claim4_full_vocab_certificate", c4["all_meet_99pct"] and all(x["success"] == 1.0 for x in c4["exhaustive_full_window"]) and c4["gate_reference_target_mismatches"] == 0, "Full-vocabulary construction passes its exact coverage and small-domain gates.") check("claim4_sample_certificate_distinction", c4["min_success_at_theorem_window"] < 0.99 and all(x["analytic_certificate_meets_99pct"] for x in c4["theorem_window"]), "The one sub-99% finite sample mean is retained separately from the exact coverage certificate.") check("direct_native_notebooks_retained", n3["kind"] == "direct_native_notebook_execution" and n4["kind"] == "direct_native_notebook_execution" and n3["control_is_lower"] and n4["control_is_lower"], "Author notebook cell execution and destructive controls are reported without upgrading scope.") q5 = c5["literal_table_comparison"] check("claim5_exact_source_scope", q5["parameter_ratio_12000_over_2000"] == 6.0 and q5["hybrid_ssm_to_tf_at_approximately_2000"] == 0.999 and not q5["strict_table_value_is_exactly_one"], "Figure 4 table is preserved as .999, not silently converted to 1.000.") q6 = c6["literal_table_checks"] check("claim6_source_contradiction_marked", not q6["hybrid_reaches_0_60_at_approximately_2000"] and q6["hybrid_ssm_to_tf_at_approximately_2000"] == 0.512 and c6["single_key_statement_is_a_different_task"], "Figure 6 sixfold row and Figure 5 task distinction are explicitly retained.") semantic = { "profile": "semantic-v4-local", "validator": "local_bundle_validator_not_campaign_validator", "checks": checks, "passed": all(c["passed"] for c in checks), "check_count": len(checks), } (ROOT / "SEMANTIC_V4.json").write_text(json.dumps(semantic, indent=2) + "\n") replay_ok = None if args.require_replay: replay = json.loads((ROOT / "REPLAY.json").read_text()) replay_ok = replay["byte_identical"] and len(replay["files"]) == 8 operational = { "replay_required": args.require_replay, "byte_identical_replay": replay_ok, "no_remote_publication": logbook["publication_status"] == "local-only; no Space created or modified", "official_campaign_validator": "not run: no campaign target was created or modified", } result = { "validator": "validate_logbook.py (local bundle validator)", "semantic_v4": semantic, "operational": operational, "passed": semantic["passed"] and (replay_ok is not False), } (ROOT / "VALIDATION.json").write_text(json.dumps(result, indent=2) + "\n") print(f"semantic-v4: {sum(c['passed'] for c in checks)}/{len(checks)}; local validation: {result['passed']}") if not result["passed"]: raise SystemExit(1) if __name__ == "__main__": main()