Hummingbird-V1 / training /corpus_contract.yaml
juinron's picture
Release Hummingbird-V1 500M natural continuation checkpoint
d03200e verified
Raw
History Blame Contribute Delete
4.53 kB
# Frost all-natural 20B-token pretraining corpus contract (pinned 2026-08-14).
#
# This contract is intentionally separate from final2b_corpus_contract.yaml: no
# generated arithmetic, rationale, tutorial, or benchmark-derived text is
# admitted. Public Open SLM tasks remain audit-only and are protected before
# global exact/near deduplication.
version: 1
project: frost-10m
purpose: natural_20b_unique_token_pretraining_corpus
frozen: 2026-08-14
seed: 42
target:
unique_train_tokens: 20000000000
train_presentations: 20000000000
gross_selected_tokens: 22000000000
split_policy: document_sha256_mod_100
validation_buckets: [0]
held_out_buckets: [1]
train_buckets: 2-99
# Shares are the desired packed-train mix. Gross quotas carry a 10% reserve for
# the two held-out hash buckets plus quality and cross-source dedup attrition.
source_mix:
- name: educational_web
share: 0.50
tokens_quota: 11000000000
fill: [natural_fineweb_edu]
- name: broad_web
share: 0.20
tokens_quota: 4400000000
fill: [natural_dclm]
- name: high_quality_web
share: 0.10
tokens_quota: 2200000000
fill: [natural_fineweb_hq]
- name: textbook_exposition
share: 0.15
tokens_quota: 3300000000
fill: [natural_cosmopedia_v2]
- name: mathematics
share: 0.05
tokens_quota: 1100000000
fill: [natural_finemath_4plus]
# Keep the most curated sources first so keep-first cross-source dedup retains
# the higher-quality copy. Every revision is an immutable Hub commit observed
# through the Hugging Face dataset API on 2026-08-14.
build_order:
- natural_finemath_4plus
- natural_cosmopedia_v2
- natural_fineweb_hq
- natural_fineweb_edu
- natural_dclm
sources:
- name: natural_fineweb_edu
row: educational_web
status: pinned
repo_id: HuggingFaceFW/fineweb-edu
revision: 87f09149ef4734204d70ed1d046ddc9ca3f2b8f9
config: sample-100BT
split: train
text_field: text
license: odc-by-1.0
shuffle_buffer: 10000
quality_filter: m9_admission
- name: natural_dclm
row: broad_web
status: pinned
repo_id: mlfoundations/dclm-baseline-1.0
revision: a3b142c183aebe5af344955ae20836eb34dcf69b
config: default
split: train
text_field: text
license: cc-by-4.0
shuffle_buffer: 10000
quality_filter: m9_admission
- name: natural_fineweb_hq
row: high_quality_web
status: pinned
repo_id: epfml/FineWeb-HQ
revision: e58199cdd52438d94405df1a4d8630cc5f13bf84
config: default
split: train
text_field: text
license: odc-by-1.0
shuffle_buffer: 10000
quality_filter: m9_admission
- name: natural_cosmopedia_v2
row: textbook_exposition
status: pinned
repo_id: HuggingFaceTB/smollm-corpus
revision: 3ba9d605774198c5868892d7a8deda78031a781f
config: cosmopedia-v2
split: train
text_field: text
license: odc-by-1.0
shuffle_buffer: 10000
quality_filter: m9_admission
- name: natural_finemath_4plus
row: mathematics
status: pinned
repo_id: HuggingFaceTB/finemath
revision: e92b25a616738fe95dc186b64dfb19f9c8525594
config: finemath-4plus
split: train
text_field: text
license: odc-by-1.0
shuffle_buffer: 10000
quality_filter: m9_admission
quality:
min_chars: 40
max_chars: 100000
max_symbol_ratio: 0.35
min_ascii_alpha_ratio: 0.45
reject_secrets: true
deduplication:
exact: document_sha256_canonical
near_duplicate:
method: 32-permutation MinHash / 8-band LSH
shingle_width: 5
threshold: 0.80
confirmation: exact Jaccard
cross_source: global_keep_first_in_build_order
cross_split: keep_first_over_union_before_hash_split
protection_index:
public_suite: artifacts/m9_protection/protection.manifest.json
internal_dev: artifacts/m9_internal_dev
terse_choice: artifacts/m10_terse/terse_dev.jsonl
pilot:
target_train_tokens: 1000000000
purpose: test mix, stability, loss slope, and internal-dev transfer before 20B
checkpoint_tokens: [250000000, 500000000, 750000000, 1000000000]
scaling_gate:
local_in_ram_minhash_max_tokens: 2000000000
full_20b_requirement: distributed or disk-sharded MinHash before packing
reason: avoid silently exhausting workstation RAM with a tens-of-millions-document index
selection_rule: >-
Choose LR and continuation only on training stability, held-out loss,
mathematics-dev loss, terse-choice accuracy, and non-public generation probes.
Public Open SLM results are reported only after the pilot decision is frozen.