File size: 10,001 Bytes
818282c
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
{
  "audited_at": "2026-07-26T05:38:59Z",
  "post_run1_correction": {
    "active_run1_process_or_shards_changed": false,
    "behavior": "Known Hub schemas select their real path column and fail closed if it is missing or empty. A leading StarCoderData reponame metadata line is removed only for syntax parsing, while the original text remains the training payload.",
    "corpus_script": {
      "path": "scripts/corpus.py",
      "sha256": "05afba561a6258812d2baa802456c29c00d30955bec7f5c46389b90b2dad9e93"
    },
    "effective_scope": "future_source_streaming_only",
    "quality_gate_test": {
      "path": "scripts/test_corpus_quality.py",
      "sha256": "b90f72e30d0eb5040ded59aaf4ae5a7a4a90c8951bbc66fe4fceb0cdd2ba5eed"
    }
  },
  "publication_documents": {
    "data_document": {
      "path": "DATA.md",
      "sha256": "fac5dce1b46f29821e3ed509332922ba395d4f3a1660b2e95929e94393e25ec7"
    },
    "model_card_template": {
      "path": "MODEL_CARD.md",
      "sha256": "ecd9611c82d8a05da30f36ef41b5ec0642ec7a4e6137c20f5951d84a51d55d9f"
    }
  },
  "registered_run1_config": {
    "path": "config/run1.json",
    "sha256": "f1683a7b00bf1b1e93654ebb366fa0072ba6092b0dbad10272e9353196c3532d"
  },
  "registered_run2_config": {
    "path": "config/run2_no_fim.json",
    "sha256": "5692f15944dcef9a8a64a96ea5b5888b39024de321c2bb9a302f1c522bc3ea12"
  },
  "required_publication_disclosures": [
    "Do not claim that every Python or JSON training document passed an extension parser.",
    "Do not claim that the run 1 shards were locally verified as permissive-only.",
    "State that original StarCoderData repository terms and relevant attribution clauses still apply.",
    "State that FineWeb-Edu is ODC-By 1.0 and remains subject to Common Crawl terms.",
    "State that Apache 2.0 covers the Wisp artifact and does not override source-data or generated-code terms.",
    "State that the tokenizer sampled source entries round-robin by document rather than using the configured training-token weights.",
    "Do not claim that deterministic run 1 token normalization proves exact original examples or boundaries.",
    "State that run 1 applied fim_rate 0.7 independently per tokenized chunk, not per source document or sampled training window."
  ],
  "run1_fim_application": {
    "build_log": {
      "path": "evidence/run1_corpus_build.log",
      "sha256": "ef5de5c46aac1ff601158b43a3cde481ae6090ea04eefc772c531ff2ac78295e"
    },
    "build_source": {
      "git_blob_sha1": "18b7e158ecec3467be28e1b18a5bab72c0ee1c77",
      "git_commit": "a534de4d542167bdcea8adfda8fbf25d6cd0db44",
      "path": "scripts/prepare_data.py",
      "sha256": "6ebbd49a92de87582c429e2c0a5e2fd22792b7db1cbf44e37651b4eceaef7ff6"
    },
    "configured_rate_is_per_source_document": false,
    "configured_rate_is_per_training_window": false,
    "configured_transform_probability_per_chunk": 0.7,
    "maximum_chunk_tokens": 1024,
    "selected_orderings": {
      "psm_probability": 0.5,
      "spm_probability": 0.5
    },
    "selection_unit": "tokenized_chunk",
    "statement": "Run 1 split each tokenized source document into chunks of at most 1024 tokens and selected FIM independently for each chunk. The configured 0.7 is not a per-document or per-window rate.",
    "training_window_tokens": 2051
  },
  "run1_final_build_evidence": {
    "final_index": {
      "path": "data/shards/index.json",
      "sha256": "862b1a9b7cc6c3c0d31299e21b352e2b736de767a99bf7fa38213d6c60fc0db0"
    },
    "log": {
      "path": "evidence/run1_corpus_build.log",
      "sha256": "ef5de5c46aac1ff601158b43a3cde481ae6090ea04eefc772c531ff2ac78295e"
    },
    "realized_train_percent": {
      "fineweb_edu": 7.993028,
      "implementation_code": 84.997663,
      "markdown": 7.009309,
      "starcoderdata_including_markdown": 92.006972
    },
    "realized_train_tokens": {
      "HuggingFaceFW/fineweb-edu:sample-10BT": 400290897,
      "bigcode/starcoderdata:c": 250363883,
      "bigcode/starcoderdata:go": 451081764,
      "bigcode/starcoderdata:java": 400525337,
      "bigcode/starcoderdata:javascript": 650722190,
      "bigcode/starcoderdata:markdown": 351026251,
      "bigcode/starcoderdata:python": 1200566506,
      "bigcode/starcoderdata:rust": 451511705,
      "bigcode/starcoderdata:shell": 150445965,
      "bigcode/starcoderdata:sql": 100765073,
      "bigcode/starcoderdata:typescript": 600701168
    },
    "scope": "Exact aggregate train-token totals only. Row identities, rejection counts, per-source validation overshoot, and row-level obligations remain unavailable.",
    "total_train_tokens": 5008000739
  },
  "run1_provenance_limitations": {
    "exact_original_unit_boundaries_proven": false,
    "locally_verified_permissive_only": false,
    "per_row_attribution_index_available": false,
    "per_row_license_mapping_preserved": false,
    "per_source_validation_counts_recorded": false,
    "realized_per_source_train_tokens_recovered_from_final_log": true,
    "source_revisions_captured_during_build": false,
    "source_row_metadata_preserved_in_shards": false,
    "statement": "Run 1 preserves configured source weights, exact aggregate train-token totals from the recovered final build log, post-build revision evidence, and a post-build hash of every current shard. It does not preserve ordered raw rows, repository paths, rejection counts, per-source validation overshoot, per-row licenses, attribution mapping, or exact original unit boundaries needed for a local permissive-only and example-exact audit."
  },
  "run1_quality_gate": {
    "configured_path_field": null,
    "corpus_script_at_build_sha256": "7180e0d69a543fa2ddcf76ef6fa035a14bab2f7e0dfc8a31413b416ad891886e",
    "extension_parser_activated_for_hub_rows": false,
    "path_field_used": "path",
    "reason": "The run 1 iterator requested the absent path column, so Hub rows reached the structural filters without a file extension. The Python ast.parse and JSON json.loads branches therefore did not activate.",
    "starcoderdata_path_field": "max_stars_repo_path",
    "structural_filters_applied": true
  },
  "run1_shard_integrity": {
    "attestation_kind": "post_build_current_bytes_and_visible_grammar",
    "boundary_recovery": {
      "detectable_reassembly_groups": 27,
      "exact_original_units_proven": false,
      "mode": "deterministic_visible_grammar_normalization",
      "restored_internal_eos_tokens": 33
    },
    "normalized_splits": {
      "train": {
        "derived_tokens": 4992043184,
        "fim_units": 5319185,
        "source_tokens": 5008000739,
        "units": 7629643
      },
      "val": {
        "derived_tokens": 19949502,
        "fim_units": 18870,
        "source_tokens": 20006112,
        "units": 27087
      }
    },
    "receipt": {
      "path": "config/run1_shard_integrity_receipt.json",
      "sha256": "5831ecd4a471fbe07e19b212bc3de44bed0b0b6b456083e888f66802937bf471"
    },
    "scheduled_run2_training_positions": 4999872512,
    "source_bytes": 10056013702,
    "source_files": 52,
    "source_index_sha256": "862b1a9b7cc6c3c0d31299e21b352e2b736de767a99bf7fa38213d6c60fc0db0"
  },
  "schema_version": 1,
  "sources": {
    "HuggingFaceFW/fineweb-edu": {
      "configured_weight": 0.08,
      "current_main_revision_at_audit": "87f09149ef4734204d70ed1d046ddc9ca3f2b8f9",
      "dataset_card": {
        "license_label": "odc-by",
        "sha256": "a0cc8998a20499432b28b6575f3046b714938eb8e11b8d59a1d25ddf3716061e",
        "terms": "The dataset is distributed under ODC-By 1.0 and remains subject to Common Crawl terms.",
        "url": "https://huggingface.co/datasets/HuggingFaceFW/fineweb-edu/blob/87f09149ef4734204d70ed1d046ddc9ca3f2b8f9/README.md"
      },
      "observed_stream_row_schema": {
        "canonical_fields_sha256": "a7b0323d3e758514f936736e75a919bda456e98164299c1c1ce5970f65678f91",
        "fields": [
          "dump",
          "file_path",
          "id",
          "int_score",
          "language",
          "language_score",
          "score",
          "text",
          "token_count",
          "url"
        ],
        "subset": "sample-10BT"
      },
      "post_build_cache_ref_revision": "87f09149ef4734204d70ed1d046ddc9ca3f2b8f9"
    },
    "bigcode/starcoderdata": {
      "configured_weight": 0.92,
      "current_main_revision_at_audit": "9fc30b578cedaec69e47302df72cf00feed7c8c4",
      "dataset_card": {
        "license_label": "other",
        "sha256": "7a3e42cc82fb48b6b81f2ef06eab94af33e605eff743c6a4b8a3b1852ced7c0a",
        "terms": "Original repository licenses apply, including attribution clauses when relevant. Users must follow the source dataset update and removal terms.",
        "url": "https://huggingface.co/datasets/bigcode/starcoderdata/blob/9fc30b578cedaec69e47302df72cf00feed7c8c4/README.md"
      },
      "observed_stream_row_schema": {
        "canonical_fields_sha256": "ddfa03121c2f5e5766eada883df62dcaef2a04540a49d849692831ae81fbddd4",
        "fields": [
          "content",
          "id",
          "max_stars_count",
          "max_stars_repo_name",
          "max_stars_repo_path"
        ],
        "subset": "python"
      },
      "post_build_cache_ref_revision": "9fc30b578cedaec69e47302df72cf00feed7c8c4"
    }
  },
  "status": "limitations_registered",
  "tokenizer_sampling": {
    "build_log": {
      "path": "evidence/tokenizer_build.log",
      "sha256": "7c28f91dc527e0cc37d23c520ba183aef845f948b64ea822f3b3ca17de264467"
    },
    "configured_token_weights_applied": false,
    "documents_requested": 400000,
    "exact_row_manifest_preserved": false,
    "source_entries": 11,
    "statement": "The tokenizer sample included FineWeb-Edu and sampled source entries evenly by document, not according to the later training-token weights.",
    "strategy": "round_robin_by_configured_source_entry",
    "tokenizer": {
      "path": "tokenizer/code32k.json",
      "sha256": "401a28c1f079050c48f6438830ca772d161d897e3cf2f30588d9ddc587dc6081"
    }
  }
}