wisp-coder-110m / evidence /source /training_data_contract.py
philipjohnbasile's picture
Publish audited Wisp Coder 110M release
818282c verified
Raw
History Blame Contribute Delete
22.4 kB
"""Fail-closed checks for Wisp's training-data disclosure receipt."""
import hashlib
import json
import os
EXPECTED_SOURCE_REVISIONS = {
"bigcode/starcoderdata": "9fc30b578cedaec69e47302df72cf00feed7c8c4",
"HuggingFaceFW/fineweb-edu": (
"87f09149ef4734204d70ed1d046ddc9ca3f2b8f9"
),
}
EXPECTED_SOURCES = {
"bigcode/starcoderdata": {
"configured_weight": 0.92,
"post_build_cache_ref_revision": EXPECTED_SOURCE_REVISIONS[
"bigcode/starcoderdata"
],
"current_main_revision_at_audit": EXPECTED_SOURCE_REVISIONS[
"bigcode/starcoderdata"
],
"dataset_card": {
"url": (
"https://huggingface.co/datasets/bigcode/starcoderdata/blob/"
"9fc30b578cedaec69e47302df72cf00feed7c8c4/README.md"
),
"sha256": (
"7a3e42cc82fb48b6b81f2ef06eab94af33e605eff743c6a4b8a3b1852ced7c0a"
),
"license_label": "other",
"terms": (
"Original repository licenses apply, including attribution "
"clauses when relevant. Users must follow the source dataset "
"update and removal terms."
),
},
"observed_stream_row_schema": {
"subset": "python",
"fields": [
"content",
"id",
"max_stars_count",
"max_stars_repo_name",
"max_stars_repo_path",
],
"canonical_fields_sha256": (
"ddfa03121c2f5e5766eada883df62dcaef2a04540a49d849692831ae81fbddd4"
),
},
},
"HuggingFaceFW/fineweb-edu": {
"configured_weight": 0.08,
"post_build_cache_ref_revision": EXPECTED_SOURCE_REVISIONS[
"HuggingFaceFW/fineweb-edu"
],
"current_main_revision_at_audit": EXPECTED_SOURCE_REVISIONS[
"HuggingFaceFW/fineweb-edu"
],
"dataset_card": {
"url": (
"https://huggingface.co/datasets/HuggingFaceFW/fineweb-edu/"
"blob/87f09149ef4734204d70ed1d046ddc9ca3f2b8f9/README.md"
),
"sha256": (
"a0cc8998a20499432b28b6575f3046b714938eb8e11b8d59a1d25ddf3716061e"
),
"license_label": "odc-by",
"terms": (
"The dataset is distributed under ODC-By 1.0 and remains "
"subject to Common Crawl terms."
),
},
"observed_stream_row_schema": {
"subset": "sample-10BT",
"fields": [
"dump",
"file_path",
"id",
"int_score",
"language",
"language_score",
"score",
"text",
"token_count",
"url",
],
"canonical_fields_sha256": (
"a7b0323d3e758514f936736e75a919bda456e98164299c1c1ce5970f65678f91"
),
},
},
}
EXPECTED_RUN1_GATE = {
"structural_filters_applied": True,
"extension_parser_activated_for_hub_rows": False,
"configured_path_field": None,
"path_field_used": "path",
"starcoderdata_path_field": "max_stars_repo_path",
"reason": (
"The run 1 iterator requested the absent path column, so Hub rows "
"reached the structural filters without a file extension. The Python "
"ast.parse and JSON json.loads branches therefore did not activate."
),
"corpus_script_at_build_sha256": (
"7180e0d69a543fa2ddcf76ef6fa035a14bab2f7e0dfc8a31413b416ad891886e"
),
}
EXPECTED_CORRECTION_BEHAVIOR = (
"Known Hub schemas select their real path column and fail closed if it is "
"missing or empty. A leading StarCoderData reponame metadata line is "
"removed only for syntax parsing, while the original text remains the "
"training payload."
)
EXPECTED_TOKENIZER_SAMPLING = {
"documents_requested": 400000,
"strategy": "round_robin_by_configured_source_entry",
"configured_token_weights_applied": False,
"source_entries": 11,
"exact_row_manifest_preserved": False,
"build_log": {
"path": "evidence/tokenizer_build.log",
"sha256": (
"7c28f91dc527e0cc37d23c520ba183aef845f948b64ea822f3b3ca17de264467"
),
},
"tokenizer": {
"path": "tokenizer/code32k.json",
"sha256": (
"401a28c1f079050c48f6438830ca772d161d897e3cf2f30588d9ddc587dc6081"
),
},
"statement": (
"The tokenizer sample included FineWeb-Edu and sampled source entries "
"evenly by document, not according to the later training-token weights."
),
}
EXPECTED_FINAL_BUILD_EVIDENCE = {
"log": {
"path": "evidence/run1_corpus_build.log",
"sha256": (
"ef5de5c46aac1ff601158b43a3cde481ae6090ea04eefc772c531ff2ac78295e"
),
},
"final_index": {
"path": "data/shards/index.json",
"sha256": (
"862b1a9b7cc6c3c0d31299e21b352e2b736de767a99bf7fa38213d6c60fc0db0"
),
},
"realized_train_tokens": {
"bigcode/starcoderdata:python": 1200566506,
"bigcode/starcoderdata:javascript": 650722190,
"bigcode/starcoderdata:typescript": 600701168,
"bigcode/starcoderdata:go": 451081764,
"bigcode/starcoderdata:rust": 451511705,
"bigcode/starcoderdata:java": 400525337,
"bigcode/starcoderdata:c": 250363883,
"bigcode/starcoderdata:shell": 150445965,
"bigcode/starcoderdata:sql": 100765073,
"bigcode/starcoderdata:markdown": 351026251,
"HuggingFaceFW/fineweb-edu:sample-10BT": 400290897,
},
"total_train_tokens": 5008000739,
"realized_train_percent": {
"implementation_code": 84.997663,
"starcoderdata_including_markdown": 92.006972,
"markdown": 7.009309,
"fineweb_edu": 7.993028,
},
"scope": (
"Exact aggregate train-token totals only. Row identities, rejection "
"counts, per-source validation overshoot, and row-level obligations "
"remain unavailable."
),
}
EXPECTED_RUN1_FIM_APPLICATION = {
"selection_unit": "tokenized_chunk",
"maximum_chunk_tokens": 1024,
"configured_transform_probability_per_chunk": 0.7,
"selected_orderings": {
"psm_probability": 0.5,
"spm_probability": 0.5,
},
"training_window_tokens": 2051,
"configured_rate_is_per_source_document": False,
"configured_rate_is_per_training_window": False,
"build_source": {
"git_commit": (
"a534de4d542167bdcea8adfda8fbf25d6cd0db44"
),
"path": "scripts/prepare_data.py",
"git_blob_sha1": "18b7e158ecec3467be28e1b18a5bab72c0ee1c77",
"sha256": (
"6ebbd49a92de87582c429e2c0a5e2fd22792b7db1cbf44e37651b4eceaef7ff6"
),
},
"build_log": {
"path": "evidence/run1_corpus_build.log",
"sha256": (
"ef5de5c46aac1ff601158b43a3cde481ae6090ea04eefc772c531ff2ac78295e"
),
},
"statement": (
"Run 1 split each tokenized source document into chunks of at most "
"1024 tokens and selected FIM independently for each chunk. The "
"configured 0.7 is not a per-document or per-window rate."
),
}
EXPECTED_RUN1_INTEGRITY = {
"source_files": 52,
"source_bytes": 10056013702,
"source_index_sha256": (
"862b1a9b7cc6c3c0d31299e21b352e2b736de767a99bf7fa38213d6c60fc0db0"
),
"attestation_kind": "post_build_current_bytes_and_visible_grammar",
"boundary_recovery": {
"mode": "deterministic_visible_grammar_normalization",
"detectable_reassembly_groups": 27,
"restored_internal_eos_tokens": 33,
"exact_original_units_proven": False,
},
"normalized_splits": {
"train": {
"source_tokens": 5008000739,
"derived_tokens": 4992043184,
"units": 7629643,
"fim_units": 5319185,
},
"val": {
"source_tokens": 20006112,
"derived_tokens": 19949502,
"units": 27087,
"fim_units": 18870,
},
},
"scheduled_run2_training_positions": 4999872512,
}
EXPECTED_RUN2_BUILD_CONTRACT = {
"schema_version": 2,
"strategy": "run1_deterministic_no_fim_normalization_v1",
"source_index": {
"path": "data/shards/index.json",
"sha256": EXPECTED_RUN1_INTEGRITY["source_index_sha256"],
"schema_version": 1,
},
"source_integrity_receipt": (
"config/run1_shard_integrity_receipt.json"
),
"source_integrity_receipt_sha256": (
"5831ecd4a471fbe07e19b212bc3de44bed0b0b6b456083e888f66802937bf471"
),
"require_fresh_output_dir": True,
}
EXPECTED_LIMITATIONS = {
"source_row_metadata_preserved_in_shards": False,
"per_row_license_mapping_preserved": False,
"per_row_attribution_index_available": False,
"realized_per_source_train_tokens_recovered_from_final_log": True,
"per_source_validation_counts_recorded": False,
"locally_verified_permissive_only": False,
"source_revisions_captured_during_build": False,
"exact_original_unit_boundaries_proven": False,
"statement": (
"Run 1 preserves configured source weights, exact aggregate "
"train-token totals from the recovered final build log, post-build "
"revision evidence, and a post-build hash of every current shard. It "
"does not preserve ordered raw rows, repository paths, rejection "
"counts, per-source validation overshoot, per-row licenses, attribution "
"mapping, or exact original unit boundaries needed for a local "
"permissive-only and example-exact audit."
),
}
PROHIBITED_PUBLICATION_TEXT = (
"92 percent permissively licensed",
"permissive-only by construction upstream",
"70 percent of its pretraining documents",
"FIM is applied per document at prepare time",
"to 70 percent of documents",
(
"Every document passed a quality gate before tokenization: Python had "
"to survive"
),
)
EXPECTED_DISCLOSURES = [
(
"Do not claim that every Python or JSON training document passed an "
"extension parser."
),
(
"Do not claim that the run 1 shards were locally verified as "
"permissive-only."
),
(
"State that original StarCoderData repository terms and relevant "
"attribution clauses still apply."
),
(
"State that FineWeb-Edu is ODC-By 1.0 and remains subject to Common "
"Crawl terms."
),
(
"State that Apache 2.0 covers the Wisp artifact and does not override "
"source-data or generated-code terms."
),
(
"State that the tokenizer sampled source entries round-robin by "
"document rather than using the configured training-token weights."
),
(
"Do not claim that deterministic run 1 token normalization proves "
"exact original examples or boundaries."
),
(
"State that run 1 applied fim_rate 0.7 independently per tokenized "
"chunk, not per source document or sampled training window."
),
]
def file_sha256(path):
digest = hashlib.sha256()
with open(path, "rb") as f:
for block in iter(lambda: f.read(1024 * 1024), b""):
digest.update(block)
return digest.hexdigest()
def load_json(path):
with open(path, encoding="utf-8") as f:
value = json.load(f)
if not isinstance(value, dict):
raise ValueError(f"{path}: top-level JSON must be an object")
return value
def canonical_fields_sha256(fields):
payload = json.dumps(
sorted(fields),
separators=(",", ":"),
ensure_ascii=True,
).encode()
return hashlib.sha256(payload).hexdigest()
def _require(condition, message):
if not condition:
raise ValueError(message)
def _validate_artifact(artifact, label):
path = artifact.get("path")
expected = artifact.get("sha256")
_require(
isinstance(path, str) and isinstance(expected, str),
f"training-data {label} artifact is missing",
)
_require(
file_sha256(path) == expected,
f"training-data {label} hash differs from receipt",
)
def validate_publication_text(data_text, model_card_text):
data_text = " ".join(data_text.split())
model_card_text = " ".join(model_card_text.split())
combined = data_text + "\n" + model_card_text
for claim in PROHIBITED_PUBLICATION_TEXT:
_require(
claim not in combined,
f"prohibited training-data claim remains: {claim}",
)
required_data = (
"The final build log survived and is now preserved byte-for-byte",
"The realized train split is 92.006972 percent StarCoderData",
"neither extension parser activated during the run 1 Hub build",
"cannot support a local per-file licensing or attribution audit",
"trained on 400,000 documents drawn round-robin",
"not proof of exact original examples",
"FIM is selected independently for each chunk at prepare time",
"It was not applied once per source document",
)
required_model_card = (
"This is a source percentage, not a permissive-license percentage",
"did not activate for run 1",
"does not override training-source terms",
"sampled round-robin across the eleven source entries",
"cannot prove exact original example boundaries",
"Exact aggregate train-token totals survive",
"70 percent of those chunks were independently transformed",
"This is not a per-document or per-window rate",
)
for text in required_data:
_require(text in data_text, f"DATA.md disclosure is missing: {text}")
for text in required_model_card:
_require(
text in model_card_text,
f"MODEL_CARD.md disclosure is missing: {text}",
)
def _configured_repo_weights(config):
weights = {}
for source in config.get("sources", []):
repo = source.get("repo")
_require(isinstance(repo, str), "training source repo is missing")
weights[repo] = weights.get(repo, 0.0) + float(source["weight"])
return {repo: round(weight, 12) for repo, weight in weights.items()}
def validate_training_data_receipt(receipt, receipt_path=None):
_require(
receipt.get("schema_version") == 1,
"training-data receipt schema is not 1",
)
_require(
receipt.get("status") == "limitations_registered",
"training-data limitations are not registered",
)
run1_artifact = receipt.get("registered_run1_config", {})
run2_artifact = receipt.get("registered_run2_config", {})
_validate_artifact(run1_artifact, "run 1 config")
_validate_artifact(run2_artifact, "run 2 config")
run1 = load_json(run1_artifact["path"])
run2 = load_json(run2_artifact["path"])
run1_weights = _configured_repo_weights(run1)
_require(
"sources" not in run2,
"derived run 2 config contains executable training sources",
)
sources = receipt.get("sources")
_require(
sources == EXPECTED_SOURCES,
"training-data source evidence differs",
)
receipt_weights = {
repo: float(source.get("configured_weight"))
for repo, source in sources.items()
}
_require(
receipt_weights == run1_weights,
"training-data source weights differ from registered configs",
)
for repo, revision in EXPECTED_SOURCE_REVISIONS.items():
source = sources[repo]
_require(
source.get("post_build_cache_ref_revision") == revision
and source.get("current_main_revision_at_audit") == revision,
f"training-data {repo} revision evidence differs",
)
schema = source.get("observed_stream_row_schema", {})
_require(
canonical_fields_sha256(schema.get("fields", []))
== schema.get("canonical_fields_sha256"),
f"training-data {repo} row schema hash differs",
)
run1_gate = receipt.get("run1_quality_gate", {})
_require(
run1_gate == EXPECTED_RUN1_GATE,
"run 1 quality-gate limitation differs",
)
correction = receipt.get("post_run1_correction", {})
_require(
correction.get("effective_scope") == "future_source_streaming_only"
and correction.get("active_run1_process_or_shards_changed") is False
and correction.get("behavior") == EXPECTED_CORRECTION_BEHAVIOR,
"post-run 1 correction scope differs",
)
_validate_artifact(correction.get("corpus_script", {}), "corpus script")
_validate_artifact(
correction.get("quality_gate_test", {}),
"quality-gate test",
)
_require(
receipt.get("tokenizer_sampling") == EXPECTED_TOKENIZER_SAMPLING,
"tokenizer sampling limitation differs",
)
_validate_artifact(
receipt["tokenizer_sampling"]["build_log"],
"tokenizer build log",
)
_validate_artifact(
receipt["tokenizer_sampling"]["tokenizer"],
"tokenizer",
)
build_evidence = receipt.get("run1_final_build_evidence")
_require(
build_evidence == EXPECTED_FINAL_BUILD_EVIDENCE,
"run 1 final-build evidence differs",
)
_validate_artifact(build_evidence["log"], "run 1 final build log")
_validate_artifact(build_evidence["final_index"], "run 1 final index")
_require(
sum(build_evidence["realized_train_tokens"].values())
== build_evidence["total_train_tokens"],
"run 1 realized train-token totals do not sum",
)
fim_application = receipt.get("run1_fim_application")
_require(
fim_application == EXPECTED_RUN1_FIM_APPLICATION,
"run 1 FIM application evidence differs",
)
_validate_artifact(
fim_application["build_log"],
"run 1 FIM build log",
)
integrity = receipt.get("run1_shard_integrity", {})
integrity_artifact = integrity.get("receipt", {})
_validate_artifact(integrity_artifact, "run 1 shard integrity receipt")
_require(
{
key: value
for key, value in integrity.items()
if key != "receipt"
}
== EXPECTED_RUN1_INTEGRITY,
"run 1 shard integrity summary differs",
)
integrity_receipt = load_json(integrity_artifact["path"])
_require(
integrity_receipt.get("schema_version") == 1
and integrity_receipt.get("status") == "complete"
and integrity_receipt.get("algorithm")
== "run1_deterministic_no_fim_normalization_v1",
"run 1 shard integrity receipt contract differs",
)
_require(
integrity_receipt.get("source_index", {}).get("sha256")
== EXPECTED_RUN1_INTEGRITY["source_index_sha256"],
"run 1 source index hash differs",
)
_require(
{
key: integrity_receipt.get("boundary_recovery", {}).get(key)
for key in EXPECTED_RUN1_INTEGRITY["boundary_recovery"]
}
== EXPECTED_RUN1_INTEGRITY["boundary_recovery"],
"run 1 boundary-recovery evidence differs",
)
for split, expected in EXPECTED_RUN1_INTEGRITY[
"normalized_splits"
].items():
actual = integrity_receipt.get("splits", {}).get(split, {})
summary = {
"source_tokens": actual.get("source_tokens"),
"derived_tokens": actual.get("derived_tokens"),
"units": actual.get("units"),
"fim_units": (
actual.get("fim_psm_units", 0)
+ actual.get("fim_spm_units", 0)
),
}
_require(
summary == expected,
f"run 1 normalized {split} evidence differs",
)
expected_integrity_reference = {
"role": "current",
"receipt": integrity_artifact["path"],
"sha256": integrity_artifact["sha256"],
}
_require(
run1.get("data_integrity") == expected_integrity_reference,
"run 1 data-integrity config differs",
)
expected_integrity_reference["role"] = "source"
_require(
run2.get("data_integrity") == expected_integrity_reference,
"run 2 source-integrity config differs",
)
_require(
run2.get("data_build_contract") == EXPECTED_RUN2_BUILD_CONTRACT,
"run 2 deterministic normalization contract differs",
)
documents = receipt.get("publication_documents", {})
data_artifact = documents.get("data_document", {})
model_card_artifact = documents.get("model_card_template", {})
_validate_artifact(data_artifact, "data document")
_validate_artifact(model_card_artifact, "model-card template")
with open(data_artifact["path"], encoding="utf-8") as f:
data_text = f.read()
with open(model_card_artifact["path"], encoding="utf-8") as f:
model_card_text = f.read()
validate_publication_text(data_text, model_card_text)
limitations = receipt.get("run1_provenance_limitations", {})
_require(
limitations == EXPECTED_LIMITATIONS,
"run 1 provenance limitations differ",
)
_require(
receipt.get("required_publication_disclosures")
== EXPECTED_DISCLOSURES,
"training-data publication disclosures differ",
)
return {
"receipt_path": receipt_path,
"receipt_sha256": (
file_sha256(receipt_path) if receipt_path is not None else None
),
"source_weights": receipt_weights,
"source_revisions": EXPECTED_SOURCE_REVISIONS,
"future_parser_fix_bound": True,
"run1_shard_integrity_bound": True,
"deterministic_normalization_not_exact_original_units": True,
"publication_documents": {
key: value["sha256"]
for key, value in documents.items()
},
}
def main():
path = os.path.join("config", "training_data_receipt.json")
receipt = load_json(path)
evidence = validate_training_data_receipt(receipt, path)
print(
"Training-data limitations and future schema-aware parser fix: PASS "
f"({evidence['receipt_sha256']})"
)
if __name__ == "__main__":
main()