| """Strict, offline validator for this local six-claim package. |
| |
| This deliberately is *not* described as a campaign/Hub validator: no remote |
| publication is part of this run. It validates only the files present in this |
| directory and writes deterministic machine-readable results. |
| """ |
|
|
| from __future__ import annotations |
|
|
| import argparse |
| import hashlib |
| import json |
| import subprocess |
| from pathlib import Path |
|
|
|
|
| ROOT = Path(__file__).resolve().parent |
| OUT = ROOT / "outputs" |
| EXPECTED_ARCHIVE = "e8d22bfd259aaa60385841d8643109ecb66f7eb1081dd76429f5215f05a032e8" |
| EXPECTED_CODE = "6be8f8fbc2169290af6f4ba5e4bd53a5c6485f7b" |
| EXPECTED_CLAIMS = [ |
| "Theorem 3.3 proves that any k-layer state-space model solving the function-composition tasks under injectivity conditions must have total log state-space size scaling as Ω(m·log|V| − q·log|Y|), linear in the hidden dimension m (Theorem 3.3).", |
| "Theorem 3.7 proves that sliding-window Transformers solving the same tasks under a local-sensitivity condition require total window size scaling with the context-dependency range R (Theorem 3.7).", |
| "Theorem 4.3 constructs a two-layer hybrid (Mamba + attention) model that solves the selective copying task using embedding dimension O(max(log|V|, log L)) and working memory Õ(N), versus Ω(L) required by pure Transformers (Theorem 4.3).", |
| "Theorem 4.6 constructs a three-layer hybrid model that achieves 99% accuracy on the associative recall task using embedding dimension O(max(log|V|, log L)) and window size Õ(|V|) (Theorem 4.6).", |
| "On the selective copying task, the learned hybrid model reaches perfect accuracy with roughly 2,000 parameters while pure Transformer/SSM models need roughly 12,000 parameters to match it, a 6x parameter gap (Figure 4).", |
| "On multi-key associative recall, the hybrid model reaches 60% accuracy using 6x fewer parameters than pure Transformers, which plateau near 40% accuracy on single-key associative recall (Figures 5-6).", |
| ] |
|
|
|
|
| def sha(path: Path) -> str: |
| return hashlib.sha256(path.read_bytes()).hexdigest() |
|
|
|
|
| def read(name: str) -> dict: |
| return json.loads((OUT / name).read_text()) |
|
|
|
|
| def main() -> None: |
| parser = argparse.ArgumentParser() |
| parser.add_argument("--require-replay", action="store_true") |
| args = parser.parse_args() |
| c1, c2, c3, c4 = (read(f"claim{i}.json") for i in range(1, 5)) |
| c5, c6 = read("claim5.json"), read("claim6.json") |
| n3, n4 = read("claim3_native.json"), read("claim4_native.json") |
| claims = json.loads((ROOT / "CLAIMS.json").read_text())["claims"] |
| texts = [x["text"] for x in claims] |
| provenance = json.loads((ROOT / "SOURCE_PROVENANCE.json").read_text()) |
| logbook = json.loads((ROOT / "logbook.json").read_text()) |
| checks: list[dict] = [] |
|
|
| def check(name: str, ok: bool, detail: str) -> None: |
| checks.append({"name": name, "passed": bool(ok), "detail": detail}) |
|
|
| check("frozen_six_claims", texts == EXPECTED_CLAIMS, |
| "CLAIMS.json exactly equals the six registered claim strings.") |
| route_files = [ROOT / "pages/index.md", ROOT / "pages/executive-summary/page.md"] + [ |
| ROOT / "pages" / x / "page.md" for x in [ |
| "claim-1-theorem-3-3-literal-bound", "claim-2-theorem-3-7-window-bound", |
| "claim-3-theorem-4-3-selective-copy", "claim-4-theorem-4-6-associative-recall", |
| "claim-5-figure-4-selective-copy-learning", "claim-6-figures-5-6-associative-recall-learning", |
| ] |
| ] |
| check("complete_eight_routes", len(logbook["root"]["children"]) == 7 and all(p.is_file() for p in route_files), |
| "Index, executive summary, and six claim routes exist.") |
| check("exact_authored_archive", sha(ROOT / "source/2603.08859v1.tar.gz") == EXPECTED_ARCHIVE |
| and provenance["archive_sha256"] == EXPECTED_ARCHIVE, |
| "arXiv e-print archive matches the retrieval SHA-256.") |
| source_checkout = ROOT / "source/official-code" |
| embedded_git = (source_checkout / ".git").exists() |
| head = subprocess.check_output(["git", "-C", str(source_checkout), "rev-parse", "HEAD"], text=True).strip() if embedded_git else None |
| |
| |
| |
| status = subprocess.check_output(["git", "-C", str(ROOT), "status", "--porcelain", "--", "source/official-code"], text=True).strip() |
| code_identity_ok = (head == EXPECTED_CODE) if embedded_git else True |
| code_detail = ("Linked source repository is detached at the recorded clean commit." |
| if embedded_git else |
| "The linked source is a clean tracked snapshot without embedded Git metadata; external repository commit 6be8f8fbc2169290af6f4ba5e4bd53a5c6485f7b remains recorded in SOURCE_PROVENANCE.json.") |
| check("pinned_clean_code_checkout", code_identity_ok and not status, code_detail) |
| check("claim1_literal_falsification", c1["printed_bound_audit"]["max_literal_bound_over_admissible_grid"] <= 0 |
| and c1["one_state_guessing_accuracy"]["2"] == 0.5 |
| and all(row["A_star"]["1"] == 0.5 for row in c1["exhaustive_accuracy_curves"][:2]), |
| "Under injectivity the printed RHS is non-positive; one-state 1/2 witnesses are retained.") |
| check("claim2_receptive_field_witnesses", c2["receptive_field_sweep"]["violations_outside_receptive_field"] == 0 |
| and c2["max_abs_delta_when_sumW_below_R"] == 0.0 |
| and c2["control_full_window_separates_frac"] == 1.0, |
| "Real causal-attention stacks preserve the designed outside-window witness.") |
| check("claim3_construction_and_controls", c3["min_accuracy_over_all_configs"] == 1.0 |
| and c3["total_inputs_tested"] >= 1_500_000 |
| and all(x["full_construction_accuracy"] == 1.0 |
| and x["control_window_minus_1_accuracy"] < 1.0 |
| and x["control_no_ssm_query_accuracy"] < 1.0 |
| for x in c3["negative_controls"]), |
| "Independent finite construction succeeds while destructive controls fail.") |
| check("claim4_full_vocab_certificate", c4["all_meet_99pct"] |
| and all(x["success"] == 1.0 for x in c4["exhaustive_full_window"]) |
| and c4["gate_reference_target_mismatches"] == 0, |
| "Full-vocabulary construction passes its exact coverage and small-domain gates.") |
| check("claim4_sample_certificate_distinction", c4["min_success_at_theorem_window"] < 0.99 |
| and all(x["analytic_certificate_meets_99pct"] for x in c4["theorem_window"]), |
| "The one sub-99% finite sample mean is retained separately from the exact coverage certificate.") |
| check("direct_native_notebooks_retained", n3["kind"] == "direct_native_notebook_execution" |
| and n4["kind"] == "direct_native_notebook_execution" |
| and n3["control_is_lower"] and n4["control_is_lower"], |
| "Author notebook cell execution and destructive controls are reported without upgrading scope.") |
| q5 = c5["literal_table_comparison"] |
| check("claim5_exact_source_scope", q5["parameter_ratio_12000_over_2000"] == 6.0 |
| and q5["hybrid_ssm_to_tf_at_approximately_2000"] == 0.999 |
| and not q5["strict_table_value_is_exactly_one"], |
| "Figure 4 table is preserved as .999, not silently converted to 1.000.") |
| q6 = c6["literal_table_checks"] |
| check("claim6_source_contradiction_marked", not q6["hybrid_reaches_0_60_at_approximately_2000"] |
| and q6["hybrid_ssm_to_tf_at_approximately_2000"] == 0.512 |
| and c6["single_key_statement_is_a_different_task"], |
| "Figure 6 sixfold row and Figure 5 task distinction are explicitly retained.") |
|
|
| semantic = { |
| "profile": "semantic-v4-local", |
| "validator": "local_bundle_validator_not_campaign_validator", |
| "checks": checks, |
| "passed": all(c["passed"] for c in checks), |
| "check_count": len(checks), |
| } |
| (ROOT / "SEMANTIC_V4.json").write_text(json.dumps(semantic, indent=2) + "\n") |
|
|
| replay_ok = None |
| if args.require_replay: |
| replay = json.loads((ROOT / "REPLAY.json").read_text()) |
| replay_ok = replay["byte_identical"] and len(replay["files"]) == 8 |
| operational = { |
| "replay_required": args.require_replay, |
| "byte_identical_replay": replay_ok, |
| "no_remote_publication": logbook["publication_status"] == "local-only; no Space created or modified", |
| "official_campaign_validator": "not run: no campaign target was created or modified", |
| } |
| result = { |
| "validator": "validate_logbook.py (local bundle validator)", |
| "semantic_v4": semantic, |
| "operational": operational, |
| "passed": semantic["passed"] and (replay_ok is not False), |
| } |
| (ROOT / "VALIDATION.json").write_text(json.dumps(result, indent=2) + "\n") |
| print(f"semantic-v4: {sum(c['passed'] for c in checks)}/{len(checks)}; local validation: {result['passed']}") |
| if not result["passed"]: |
| raise SystemExit(1) |
|
|
|
|
| if __name__ == "__main__": |
| main() |
|
|