agent-harness / scripts /audit_public_release.py
cuber12's picture
Add Zenodo DOI to paper and publication metadata
bf40baf verified
Raw
History Blame Contribute Delete
5.34 kB
"""Fail-closed checks for a public Git/Zenodo release candidate."""
from __future__ import annotations
import hashlib
import json
from pathlib import Path
import re
import subprocess
ROOT = Path(__file__).resolve().parents[1]
CANONICAL_PDF = "output/pdf/dissecting_repository_scale_code_agent_harnesses.pdf"
CANONICAL_DOI = "10.5281/zenodo.21781711"
REQUIRED = {
".gitignore",
".zenodo.json",
"CITATION.cff",
"LICENSE",
"LICENSE-DATA",
"README.md",
"README_ZENODO.md",
"REPRODUCING.md",
"THIRD_PARTY_NOTICES.md",
"paper/main.tex",
"paper/references.bib",
CANONICAL_PDF,
"output/pdf/SHA256SUMS",
}
SECRET_PATTERNS = {
"private key": re.compile(r"-----BEGIN (?:RSA |EC |OPENSSH )?PRIVATE KEY-----"),
"OpenAI-style key": re.compile(r"\bsk-[A-Za-z0-9_-]{20,}\b"),
"GitHub token": re.compile(r"\b(?:gh[pousr]_[A-Za-z0-9]{20,}|github_pat_[A-Za-z0-9_]{20,})\b"),
"AWS access key": re.compile(r"\bAKIA[0-9A-Z]{16}\b"),
}
PORTABLE_SUFFIXES = {".py", ".toml", ".md", ".tex", ".bib", ".txt", ".lock"}
PROVENANCE_PREFIXES = (
"results/",
"tasks/patches/",
"tasks/selection/",
"tasks/validation/",
)
def sha256(path: Path) -> str:
return hashlib.sha256(path.read_bytes()).hexdigest()
def candidate_files() -> list[str]:
result = subprocess.run(
["git", "ls-files", "--cached", "--others", "--exclude-standard"],
cwd=ROOT,
check=True,
capture_output=True,
text=True,
)
return sorted(set(result.stdout.splitlines()))
def audit() -> dict[str, object]:
errors: list[str] = []
files = candidate_files()
missing = sorted(path for path in REQUIRED if not (ROOT / path).is_file())
if missing:
errors.append(f"missing required release files: {missing}")
pdfs = sorted(
path for path in files
if path.startswith("output/pdf/")
and path.endswith(".pdf")
and (ROOT / path).is_file()
)
if pdfs != [CANONICAL_PDF]:
errors.append(f"expected exactly one canonical PDF, found: {pdfs}")
oversized = []
for relative in files:
path = ROOT / relative
if path.is_file() and path.stat().st_size > 10 * 1024 * 1024:
oversized.append((relative, path.stat().st_size))
if oversized:
errors.append(f"files larger than 10 MiB require explicit release handling: {oversized}")
forbidden_outputs = [
path for path in files
if path.startswith(("results/raw/", "indexes/", "data/repos/"))
and not path.endswith(".gitkeep")
]
if forbidden_outputs:
errors.append(f"heavy local artifacts are tracked: {forbidden_outputs}")
secret_hits: list[str] = []
portability_hits: list[str] = []
for relative in files:
path = ROOT / relative
if not path.is_file() or path.stat().st_size > 10 * 1024 * 1024:
continue
try:
text = path.read_text(encoding="utf-8")
except UnicodeDecodeError:
continue
for label, pattern in SECRET_PATTERNS.items():
if pattern.search(text):
secret_hits.append(f"{relative}: {label}")
if (
path.suffix in PORTABLE_SUFFIXES
and not relative.startswith(PROVENANCE_PREFIXES)
and re.search(r"/(?:Users|home)/[^/\s]+/", text)
):
portability_hits.append(relative)
if secret_hits:
errors.append(f"possible credentials detected: {secret_hits}")
if portability_hits:
errors.append(f"machine-specific paths in executable/public text: {portability_hits}")
zenodo_path = ROOT / ".zenodo.json"
if zenodo_path.is_file():
metadata = json.loads(zenodo_path.read_text(encoding="utf-8"))
for key in ("title", "upload_type", "publication_type", "creators", "license"):
if not metadata.get(key):
errors.append(f".zenodo.json missing {key}")
if metadata.get("license") != "cc-by-4.0":
errors.append("Zenodo content license must be cc-by-4.0")
doi_files = (
"README.md",
"README_ZENODO.md",
"CITATION.cff",
".zenodo.json",
"paper/main.tex",
)
missing_doi = [
relative for relative in doi_files
if not (ROOT / relative).is_file()
or CANONICAL_DOI not in (ROOT / relative).read_text(encoding="utf-8")
]
if missing_doi:
errors.append(f"canonical DOI missing from release metadata: {missing_doi}")
checksum_path = ROOT / "output/pdf/SHA256SUMS"
pdf_path = ROOT / CANONICAL_PDF
if checksum_path.is_file() and pdf_path.is_file():
expected_line = f"{sha256(pdf_path)} {pdf_path.name}"
lines = [line.strip() for line in checksum_path.read_text().splitlines() if line.strip()]
if lines != [expected_line]:
errors.append("output/pdf/SHA256SUMS is stale or contains a noncanonical PDF")
return {
"status": "pass" if not errors else "fail",
"files_checked": len(files),
"canonical_pdf": CANONICAL_PDF,
"errors": errors,
}
def main() -> None:
result = audit()
print(json.dumps(result, indent=2, sort_keys=True))
if result["status"] != "pass":
raise SystemExit(1)
if __name__ == "__main__":
main()