File size: 12,511 Bytes
2042bc3
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
#!/usr/bin/env python3
"""Fail-closed validator for the six-claim FFOLayer reproduction."""

from __future__ import annotations

import hashlib
import json
import os
import subprocess
import sys
import tempfile
from pathlib import Path


ROOT = Path(__file__).resolve().parent
COMMIT = "28905f3e1750fca5b8918954d5d2ea5bed0cbacc"
TREE = "f236d623acd0a089adebafd61c7c239434c9e6b2"
SOURCE_FILES = 93
SOURCE_MANIFEST = "8ca8beef7468dacb0e6a91d4a38e28dafd5256f6b44f680dc65ad27a3678f9c6"
PDF_SHA = "41245b95365c2ffede396f2cb48071f3bf5b749039124af5c804830994e0a14c"
TAR_SHA = "043f3bd94fa18e7cf62f311a3090f1213c81d5c5d65059490d514d1d1a77db13"
LPGD_ERROR = "Unsupported mode lpgd; the supported modes are 'dense', 'lsqr' and 'lsmr'"
CLAIMS = [
    "FFOLayer computes an ε-approximate hypergradient using an active-set Lagrangian oracle that requires no Hessian evaluations, achieving Õ(1) first-order oracle calls per hypergradient estimate (Section 4.2, Algorithm 1).",
    "Theorem 4.1 proves that the 'ghost bilevel optimization' reformulation, which treats active constraints as equalities, preserves the accuracy of the hypergradient computed at the original constrained-optimization solution (Section 4.1, Theorem 4.1).",
    "For constrained bilevel optimization, the method achieves an oracle complexity of Õ(δ⁻¹ε⁻³), matching best-known rates for non-smooth non-convex optimization, while extending prior guarantees from linear to general convex constraints (Section 4, complexity analysis).",
    "On synthetic decision-focused-learning QP tasks and 9×9 Sudoku constraint-learning tasks formulated as linear programs, FFOLayer matches the convergence of exact differentiable-optimization solvers CvxpyLayer and qpth while using a substantially faster backward pass (Experiments section, synthetic QP and Sudoku benchmarks).",
    "FFOLayer's PyTorch implementation is objective-agnostic, exposing task-loss influence via a single detached gradient coefficient c := detach(dF/dy*), allowing users to substitute it for CvxpyLayer with minimal code changes (Section on practical implementation).",
    "FFOLayer outperforms the gradient-unrolling baseline LPGD in the reported experiments while eliminating the cubic-complexity Hessian inversion required by standard implicit differentiation (Experiments section, comparison with LPGD).",
]


def need(condition: bool, message: str) -> None:
    if not condition:
        raise RuntimeError(message)


def sha256(path: Path) -> str:
    digest = hashlib.sha256()
    with path.open("rb") as handle:
        for chunk in iter(lambda: handle.read(8 * 1024 * 1024), b""):
            digest.update(chunk)
    return digest.hexdigest()


def load(relative: str) -> object:
    return json.loads((ROOT / relative).read_text(encoding="utf-8"))


def source_manifest() -> tuple[int, str]:
    source = ROOT / "source_current"
    files = sorted(
        path for path in source.rglob("*")
        if path.is_file() and "__pycache__" not in path.parts and path.suffix != ".pyc"
    )
    lines = [f"{sha256(path)}  {path.relative_to(source).as_posix()}" for path in files]
    digest = hashlib.sha256(("\n".join(lines) + "\n").encode()).hexdigest()
    return len(files), digest


def compare_generated(directory: Path) -> None:
    names = {
        "claim1_rate_repetitions.csv",
        "claim1_rate_summary.json",
        "claim2_native_synthetic_benchmark.csv",
        "claim2_native_synthetic_benchmark.json",
    }
    actual = {path.name for path in directory.iterdir() if path.is_file()}
    need(actual == names, f"replay output set changed: {sorted(actual)}")
    for name in sorted(names):
        need(
            (directory / name).read_bytes() == (ROOT / "outputs" / name).read_bytes(),
            f"replay mismatch: {name}",
        )


def validate_manifest() -> int:
    manifest = ROOT / "BUNDLE_SHA256SUMS.txt"
    need(manifest.is_file(), "missing recursive manifest")
    lines = manifest.read_text(encoding="utf-8").splitlines()
    expected_paths = sorted(
        path.relative_to(ROOT).as_posix()
        for path in ROOT.rglob("*")
        if path.is_file()
        and path != manifest
        and "__pycache__" not in path.parts
        and path.suffix != ".pyc"
    )
    paths = []
    for line in lines:
        digest, relative = line.split("  ", 1)
        paths.append(relative)
        need(sha256(ROOT / relative) == digest, f"manifest mismatch: {relative}")
    need(paths == expected_paths, "manifest path set mismatch")
    return len(lines)


def main() -> None:
    claims = load("official_claims.json")
    logbook = load("logbook.json")
    matrix = load("EVIDENCE_MATRIX.json")
    pin = load("SOURCE_PIN.json")
    need(claims == CLAIMS, "official claims drift")
    need(logbook["claims"] == CLAIMS, "logbook claims drift")
    need(matrix["paper_id"] == "jJur8Fq7IK", "matrix paper mismatch")
    need([row["literal_claim"] for row in matrix["claims"]] == CLAIMS, "matrix claims drift")
    gate = matrix["release_quality_gate"]
    need(gate["semantic_quality_gate_version"] == 4, "semantic gate version changed")
    need(gate["registered_claims"] == 6, "claim count changed")
    need(gate["supported_by_independent_evidence"] == 6, "support count changed")
    need(gate["literal_falsifications"] == 2, "literal falsification census changed")
    need(gate["direct_rate_claims"] == 2, "direct rate census changed")
    need(gate["expected_verified_points"] == 12, "expected points changed")
    need(pin["commit"] == COMMIT and pin["git_tree"] == TREE, "source pin drift")
    count, digest = source_manifest()
    need(count == SOURCE_FILES and digest == SOURCE_MANIFEST, "official source snapshot drift")
    need(sha256(ROOT / "source_paper/arxiv-2512.02494.pdf") == PDF_SHA, "PDF hash drift")
    need(sha256(ROOT / "source_paper/arxiv-2512.02494.tar") == TAR_SHA, "source tar hash drift")

    rate = load("outputs/claim1_rate_summary.json")
    need(rate["rows"] == 14 and rate["repetitions_per_scale"] == 2, "claim-1 rate census changed")
    need(rate["scales"] == [10, 20, 50, 100, 200, 500, 1000], "claim-1 scales changed")
    need(rate["all_errors_le_2epsilon"] is True, "claim-1 accuracy failed")
    need(rate["oracle_evaluations_log_fit_r_squared"] >= 0.999, "claim-1 log rate fit failed")
    need(rate["destructive_control"]["control_triggered"] is True, "claim-1 control failed")

    theory = load("outputs/theory_and_active_set.json")
    ghost = theory["claim_2_ghost_active_set"]
    need(ghost["max_regular_point_absolute_error"] <= 4e-11, "claim-2 ghost identity failed")
    boundary = ghost["destructive_boundary_control"]
    need(boundary["differentiability_assumption_violated"] is True, "claim-2 control failed")
    need(abs(boundary["left_derivative"] - boundary["right_derivative"]) >= 0.24, "claim-2 boundary did not separate")

    complexity = load("outputs/claim3_general_convex_scaling.json")
    need(complexity["rows"] == 8 and complexity["repetitions_per_scale"] == 2, "claim-3 rate census changed")
    need(complexity["scales"] == [10, 20, 50, 100], "claim-3 scales changed")
    need(complexity["all_finite"] is True, "claim-3 non-finite gradient")
    need(complexity["all_soc_constraints_active_at_solver_tolerance"] is True, "claim-3 SOC activity failed")
    need(complexity["measured_exponent_no_worse_than_claimed_upper_bound"] is True, "claim-3 measured rate failed")
    need(complexity["solver_iteration_log_log_r_squared"] >= 0.98, "claim-3 rate fit failed")
    need(complexity["max_relative_hypergradient_error"] <= 0.04, "claim-3 gradient error failed")

    static = load("outputs/implementation_static_audit.json")
    need(static["official_repo_commit"] == COMMIT, "static audit pin drift")
    need(static["hessian_tokens_total"] == 0, "Hessian token appeared")
    need(static["explicit_inverse_calls_total"] == 0, "explicit inverse appeared")
    for name in ("box_qp", "nonnegative_budget_qp", "soc_qp"):
        report = load(f"outputs/objective_agnostic_{name}.json")
        need(len(report["comparisons"]) == 3, f"{name} objective census changed")
        need(report["all_finite"] is True, f"{name} non-finite gradient")
        need(report["max_relative_l2_error"] <= 2e-4, f"{name} gradient error failed")
        need(report["min_cosine_similarity"] >= 0.999999, f"{name} gradient cosine failed")
    objective_control = load("outputs/objective_agnostic_failure_control.json")
    need(objective_control["returncode"] == 0, "claim-5 active-boundary control failed")

    benchmark = load("outputs/claim2_native_synthetic_benchmark.json")
    need(benchmark["official_repository_commit"] == COMMIT, "benchmark pin drift")
    need(benchmark["total_computation_speedup_ffolayer_over_qpth"] >= 5.0, "native total speed failed")
    need(benchmark["absolute_test_df_loss_gap"] <= 5e-4, "native endpoint similarity failed")
    need(benchmark["destructive_boundary_control"]["control_triggered"] is True, "claim-4 falsification failed")
    need(benchmark["measurements"][0]["backward_seconds"] > benchmark["measurements"][1]["backward_seconds"], "claim-4 backward ordering changed")
    cvx = load("outputs/claim2_cvxpylayer_native_batch.json")
    need(cvx["batch_shape"] == [8, 800] and cvx["finite_gradient"] is True, "CvxpyLayer batch failed")
    sudoku = load("outputs/claim2_ffolayer_sudoku_epoch.json")
    need(sudoku["training_samples"] == 9000 and sudoku["test_samples"] == 1000, "Sudoku scale changed")
    need(sudoku["train_batches"] == 1125 and sudoku["test_loss"] < 0.1, "Sudoku execution failed")

    lpgd = load("outputs/claim6_lpgd_release_failure.json")
    full = lpgd["full_native_9x9_attempt"]
    micro = lpgd["released_micro_control"]
    need(lpgd["verdict"] == "falsified_as_literally_registered", "claim-6 verdict changed")
    need(full["train_batches_requested"] == 1125 and full["completed_train_records"] == 0, "claim-6 native scale changed")
    need(full["exception"] == LPGD_ERROR, "claim-6 full failure changed")
    need(micro["registered_lpgd_mode_path"]["exception"] == LPGD_ERROR, "claim-6 micro failure changed")
    need(micro["valid_diffcp_mode_path"]["status"] == "pass", "claim-6 valid-mode control failed")
    native_log = ROOT / "sudoku_results_8/lpgd/central_failures.log"
    epoch_csv = ROOT / "sudoku_results_8/lpgd/lpgd_n3_lr0.1_seed3_20260727_022547.csv"
    step_csv = ROOT / "sudoku_results_8/lpgd_steps/lpgd_n3_lr0.1_seed3_20260727_022547.csv"
    need(sha256(native_log) == full["failure_log_sha256"], "claim-6 native log drift")
    need(sha256(epoch_csv) == full["epoch_csv_sha256"], "claim-6 epoch record drift")
    need(sha256(step_csv) == full["step_csv_sha256"], "claim-6 step record drift")
    utils_text = (ROOT / "source_current/baselines/cvxpylayers_local/utils.py").read_text(encoding="utf-8")
    need("# import diffcp_lpgd" in utils_text and "mode='lpgd'" in utils_text, "claim-6 source mechanism drift")

    for row in matrix["claims"]:
        for key in ("oracle_artifacts", "control_artifacts", "independent_evidence", "executed_outputs"):
            for relative in row[key]:
                path = ROOT / relative
                need(path.is_file() and path.stat().st_size > 0, f"missing evidence: {relative}")

    environment = {
        **os.environ,
        "PYTHONDONTWRITEBYTECODE": "1",
        "PYTHONHASHSEED": "0",
        "PYTHONWARNINGS": "error",
    }
    replay_hashes = []
    for repetition in range(2):
        with tempfile.TemporaryDirectory(prefix=f"ffolayer-replay-{repetition}-") as temporary:
            output = Path(temporary) / "outputs"
            subprocess.run(
                [sys.executable, str(ROOT / "native_claim_audit.py"), "--output-dir", str(output)],
                cwd=ROOT,
                env=environment,
                check=True,
                stdout=subprocess.DEVNULL,
                timeout=120,
            )
            compare_generated(output)
            replay_hashes.append(sha256(output / "claim1_rate_summary.json"))
    need(len(set(replay_hashes)) == 1, "paired replay hashes differ")

    entries = validate_manifest()
    print(json.dumps({
        "status": "PASS",
        "claims": "6/6",
        "expected_points": "12/12",
        "literal_falsifications": 2,
        "native_program_objective_pairs": 9,
        "rate_trials": 22,
        "paired_warning_strict_replays": 2,
        "manifest_entries": entries,
    }, sort_keys=True))


if __name__ == "__main__":
    main()