ProCreations's picture
Complete expressivity repair audit
bbfd686
Raw
History Blame Contribute Delete
9.08 kB
"""Strict, offline validator for this local six-claim package.
This deliberately is *not* described as a campaign/Hub validator: no remote
publication is part of this run. It validates only the files present in this
directory and writes deterministic machine-readable results.
"""
from __future__ import annotations
import argparse
import hashlib
import json
import subprocess
from pathlib import Path
ROOT = Path(__file__).resolve().parent
OUT = ROOT / "outputs"
EXPECTED_ARCHIVE = "e8d22bfd259aaa60385841d8643109ecb66f7eb1081dd76429f5215f05a032e8"
EXPECTED_CODE = "6be8f8fbc2169290af6f4ba5e4bd53a5c6485f7b"
EXPECTED_CLAIMS = [
"Theorem 3.3 proves that any k-layer state-space model solving the function-composition tasks under injectivity conditions must have total log state-space size scaling as Ω(m·log|V| − q·log|Y|), linear in the hidden dimension m (Theorem 3.3).",
"Theorem 3.7 proves that sliding-window Transformers solving the same tasks under a local-sensitivity condition require total window size scaling with the context-dependency range R (Theorem 3.7).",
"Theorem 4.3 constructs a two-layer hybrid (Mamba + attention) model that solves the selective copying task using embedding dimension O(max(log|V|, log L)) and working memory Õ(N), versus Ω(L) required by pure Transformers (Theorem 4.3).",
"Theorem 4.6 constructs a three-layer hybrid model that achieves 99% accuracy on the associative recall task using embedding dimension O(max(log|V|, log L)) and window size Õ(|V|) (Theorem 4.6).",
"On the selective copying task, the learned hybrid model reaches perfect accuracy with roughly 2,000 parameters while pure Transformer/SSM models need roughly 12,000 parameters to match it, a 6x parameter gap (Figure 4).",
"On multi-key associative recall, the hybrid model reaches 60% accuracy using 6x fewer parameters than pure Transformers, which plateau near 40% accuracy on single-key associative recall (Figures 5-6).",
]
def sha(path: Path) -> str:
return hashlib.sha256(path.read_bytes()).hexdigest()
def read(name: str) -> dict:
return json.loads((OUT / name).read_text())
def main() -> None:
parser = argparse.ArgumentParser()
parser.add_argument("--require-replay", action="store_true")
args = parser.parse_args()
c1, c2, c3, c4 = (read(f"claim{i}.json") for i in range(1, 5))
c5, c6 = read("claim5.json"), read("claim6.json")
n3, n4 = read("claim3_native.json"), read("claim4_native.json")
claims = json.loads((ROOT / "CLAIMS.json").read_text())["claims"]
texts = [x["text"] for x in claims]
provenance = json.loads((ROOT / "SOURCE_PROVENANCE.json").read_text())
logbook = json.loads((ROOT / "logbook.json").read_text())
checks: list[dict] = []
def check(name: str, ok: bool, detail: str) -> None:
checks.append({"name": name, "passed": bool(ok), "detail": detail})
check("frozen_six_claims", texts == EXPECTED_CLAIMS,
"CLAIMS.json exactly equals the six registered claim strings.")
route_files = [ROOT / "pages/index.md", ROOT / "pages/executive-summary/page.md"] + [
ROOT / "pages" / x / "page.md" for x in [
"claim-1-theorem-3-3-literal-bound", "claim-2-theorem-3-7-window-bound",
"claim-3-theorem-4-3-selective-copy", "claim-4-theorem-4-6-associative-recall",
"claim-5-figure-4-selective-copy-learning", "claim-6-figures-5-6-associative-recall-learning",
]
]
check("complete_eight_routes", len(logbook["root"]["children"]) == 7 and all(p.is_file() for p in route_files),
"Index, executive summary, and six claim routes exist.")
check("exact_authored_archive", sha(ROOT / "source/2603.08859v1.tar.gz") == EXPECTED_ARCHIVE
and provenance["archive_sha256"] == EXPECTED_ARCHIVE,
"arXiv e-print archive matches the retrieval SHA-256.")
source_checkout = ROOT / "source/official-code"
embedded_git = (source_checkout / ".git").exists()
head = subprocess.check_output(["git", "-C", str(source_checkout), "rev-parse", "HEAD"], text=True).strip() if embedded_git else None
# official-code is a clean subdirectory of this package repository; scope
# the status query to that path so page edits do not masquerade as edits
# to the pinned official checkout.
status = subprocess.check_output(["git", "-C", str(ROOT), "status", "--porcelain", "--", "source/official-code"], text=True).strip()
code_identity_ok = (head == EXPECTED_CODE) if embedded_git else True
code_detail = ("Linked source repository is detached at the recorded clean commit."
if embedded_git else
"The linked source is a clean tracked snapshot without embedded Git metadata; external repository commit 6be8f8fbc2169290af6f4ba5e4bd53a5c6485f7b remains recorded in SOURCE_PROVENANCE.json.")
check("pinned_clean_code_checkout", code_identity_ok and not status, code_detail)
check("claim1_literal_falsification", c1["printed_bound_audit"]["max_literal_bound_over_admissible_grid"] <= 0
and c1["one_state_guessing_accuracy"]["2"] == 0.5
and all(row["A_star"]["1"] == 0.5 for row in c1["exhaustive_accuracy_curves"][:2]),
"Under injectivity the printed RHS is non-positive; one-state 1/2 witnesses are retained.")
check("claim2_receptive_field_witnesses", c2["receptive_field_sweep"]["violations_outside_receptive_field"] == 0
and c2["max_abs_delta_when_sumW_below_R"] == 0.0
and c2["control_full_window_separates_frac"] == 1.0,
"Real causal-attention stacks preserve the designed outside-window witness.")
check("claim3_construction_and_controls", c3["min_accuracy_over_all_configs"] == 1.0
and c3["total_inputs_tested"] >= 1_500_000
and all(x["full_construction_accuracy"] == 1.0
and x["control_window_minus_1_accuracy"] < 1.0
and x["control_no_ssm_query_accuracy"] < 1.0
for x in c3["negative_controls"]),
"Independent finite construction succeeds while destructive controls fail.")
check("claim4_full_vocab_certificate", c4["all_meet_99pct"]
and all(x["success"] == 1.0 for x in c4["exhaustive_full_window"])
and c4["gate_reference_target_mismatches"] == 0,
"Full-vocabulary construction passes its exact coverage and small-domain gates.")
check("claim4_sample_certificate_distinction", c4["min_success_at_theorem_window"] < 0.99
and all(x["analytic_certificate_meets_99pct"] for x in c4["theorem_window"]),
"The one sub-99% finite sample mean is retained separately from the exact coverage certificate.")
check("direct_native_notebooks_retained", n3["kind"] == "direct_native_notebook_execution"
and n4["kind"] == "direct_native_notebook_execution"
and n3["control_is_lower"] and n4["control_is_lower"],
"Author notebook cell execution and destructive controls are reported without upgrading scope.")
q5 = c5["literal_table_comparison"]
check("claim5_exact_source_scope", q5["parameter_ratio_12000_over_2000"] == 6.0
and q5["hybrid_ssm_to_tf_at_approximately_2000"] == 0.999
and not q5["strict_table_value_is_exactly_one"],
"Figure 4 table is preserved as .999, not silently converted to 1.000.")
q6 = c6["literal_table_checks"]
check("claim6_source_contradiction_marked", not q6["hybrid_reaches_0_60_at_approximately_2000"]
and q6["hybrid_ssm_to_tf_at_approximately_2000"] == 0.512
and c6["single_key_statement_is_a_different_task"],
"Figure 6 sixfold row and Figure 5 task distinction are explicitly retained.")
semantic = {
"profile": "semantic-v4-local",
"validator": "local_bundle_validator_not_campaign_validator",
"checks": checks,
"passed": all(c["passed"] for c in checks),
"check_count": len(checks),
}
(ROOT / "SEMANTIC_V4.json").write_text(json.dumps(semantic, indent=2) + "\n")
replay_ok = None
if args.require_replay:
replay = json.loads((ROOT / "REPLAY.json").read_text())
replay_ok = replay["byte_identical"] and len(replay["files"]) == 8
operational = {
"replay_required": args.require_replay,
"byte_identical_replay": replay_ok,
"no_remote_publication": logbook["publication_status"] == "local-only; no Space created or modified",
"official_campaign_validator": "not run: no campaign target was created or modified",
}
result = {
"validator": "validate_logbook.py (local bundle validator)",
"semantic_v4": semantic,
"operational": operational,
"passed": semantic["passed"] and (replay_ok is not False),
}
(ROOT / "VALIDATION.json").write_text(json.dumps(result, indent=2) + "\n")
print(f"semantic-v4: {sum(c['passed'] for c in checks)}/{len(checks)}; local validation: {result['passed']}")
if not result["passed"]:
raise SystemExit(1)
if __name__ == "__main__":
main()