TimeCapsuleLLM-ja-v3 / training_config.yaml
trtd56's picture
Release phase5 staging
0b2d408 verified
Raw
History Blame Contribute Delete
5.99 kB
extends: configs/base.yaml
phase: phase5
# Phase 5 moves the primary lever from data to model capacity: 345M -> ~730M
# (36L/20H/1280E). Rationale (2026-07-22): phase-4 held the model fixed and
# grew unique tokens 2.5x, moving historical-test BPB only 3.53%
# (1.0654 -> 1.0278) with a small train/val gap (0.127 nats) and no
# overfitting at 2.58 epochs. The phase-3 "data-bound" diagnosis is resolved;
# 345M is approaching this corpus's data-scaling knee. Data collection
# (below) is kept as a non-blocking supporting lever, not the main bet.
# See PLAN.md phase-5 section for the full 3-model comparison and budget math.
data:
ndl_metadata_dir: data/raw/ndl/metadata
ndl_metadata_glob: data/raw/ndl/metadata/books_*.zip
ndl_metadata_urls:
- https://dl.ndl.go.jp/static/files/dataset/dataset_202602_t_internet_01.zip
- https://dl.ndl.go.jp/static/files/dataset/dataset_202602_t_internet_02.zip
- https://dl.ndl.go.jp/static/files/dataset/dataset_202602_t_internet_03.zip
- https://dl.ndl.go.jp/static/files/dataset/dataset_202602_t_internet_04.zip
- https://dl.ndl.go.jp/static/files/dataset/dataset_202602_t_internet_05.zip
# Same frozen 241,430-candidate pool and selection order as phases 3-4. The
# head 186,000 zips are already on disk (phase-4 case B) and skipped by
# checksum; this run exhausts the remaining ~55,430 (case C, the pool ceiling).
ndl_selection_manifest: data/processed/ndl/selection.jsonl
ndl_download_manifest: data/processed/phase5/ndl_downloads.jsonl
ndl_raw_dir: data/raw/ndl/books
ndl_clean_dir: data/clean/ndl
ndl_clean_manifest: data/processed/phase5/ndl_clean.jsonl
# Reuse phase-4's manifest (already a superset carrying phases 1-3 forward)
# so ndl-clean only processes the newly downloaded candidates.
ndl_clean_reuse_manifest: data/processed/phase4/ndl_clean.jsonl
ndl_clean_report: phase5_ndl_cleaning.json
ndl_quality_sample_report: phase5_ndl_quality_sample.jsonl
# Exclude phase-4 documents from the fresh 200-page human review sample.
ndl_quality_exclude_manifest: data/processed/phase4/ndl_clean.jsonl
# Case C: exhaust the frozen pool. RESULT (2026-07-24, completed): only
# 8,202/55,430 new candidates accepted (14.8%, vs phase-4's 46.3%) for an
# estimated ~239M new train tokens -- below the +0.3B retokenize threshold.
# Per PLAN.md 5.2/5.3 this held data at phase-4's frozen .bin/tokenizer (see
# reports/phase5_corpus.json); the newly cleaned docs remain on disk for a
# possible future source expansion but are not merged into training.
ndl_target_documents: 241430
ndl_download_delay_seconds: 0.2
ndl_download_retries: 3
ndl_max_download_seconds: 600
# Held at phase-4's frozen split manifest (see tokenizer/tokenized notes
# below): evaluate() reads this for test-set clean_path byte sizes, so it
# must point at real data, not an unwritten phase5-only path.
split_manifest: data/processed/phase4/splits.jsonl
ndl_max_abnormal_ngram_rate: 0.38
ndl_max_latin_rate: 0.005
tokenizer:
# DECISION (2026-07-24, see reports/phase5_corpus.json): the residual pool's
# deepest ~55,430 candidates yielded only ~8,202 accepted docs / ~239M
# estimated new train tokens (~8.7% on top of phase-4's 2.742B) -- below the
# +0.3B retokenize threshold from PLAN.md 5.2/5.3. Data is held at phase-4's
# frozen tokenizer/tokens; point directly at the phase-4 artifacts rather
# than duplicating ~5.5GB of .bin data. Verified 2026-07-24: tokenizer.json
# SHA-256 c442ae590eed9f9da33bd3bdda3bcd991dde74b3e611fde021b31a6ed4545616
# matches reports/phase4_evaluation.json.
vocab_size: 16384
directory: artifacts/phase4/tokenizer
max_train_chars: 200000000
tokenized:
# Held at phase-4's frozen tokens (see tokenizer.directory note above).
# train.bin verified 2026-07-24: 5,484,978,736 bytes = 2,742,489,368 uint16
# tokens, matching reports/phase4_training.json.
directory: data/processed/phase4/tokenized
model:
# Case B from PLAN.md phase-5 comparison: ~730M params, 2.1x phase-4's 345M.
# head_dim = 1280/20 = 64 (GPT-2 style, same ratio as phase 3-4's config).
# Verify exact count with `tcja parameter-count --vocab-size 16384`.
n_layer: 36
n_head: 20
n_embd: 1280
block_size: 1024
dropout: 0.0
training:
device: cuda
dtype: bfloat16
# Provisional: 730M will not fit the L40S 46GB at phase-4's batch32/accum8
# shapes (345M already used 44.3/47.8GB there). First candidate GPU is A100
# 80GB. tokens_per_step kept at 262,144 (batch16 x accum16 x block1024).
# CONFIRM ON FLIGHT (PLAN.md 5.5) before any paid run; shrink batch/accum
# (e.g. 8x32) if measured VRAM headroom < 10%.
batch_size: 16
gradient_accumulation_steps: 16
# FINALIZED (2026-07-24, PLAN.md 5.4): data held at phase-4's 2,742,489,368
# train tokens (see tokenized.directory above) x 3 epochs = 8,227,468,104
# processed tokens / 262,144 tok/step = 31,383.9 -> 31,400 iters (~3.00
# epochs). 3 epochs keeps tokens/param at ~11.3 (730,164,480 params),
# within the data-constrained-scaling tolerance cited in PLAN.md. Batch/accum
# shapes above remain provisional pending the GPU flight (PLAN.md 5.5).
max_iters: 31400
learning_rate: 0.00025
min_lr: 0.000025
warmup_iters: 300
lr_decay_iters: 31400
eval_interval: 500
eval_iters: 50
checkpoint_interval: 500
grad_clip: 1.0
out_dir: artifacts/phase5
budget:
# PROVISIONAL upper bound from PLAN.md phase-5 section; confirm via flight
# (PLAN.md 5.5) before starting any paid GPU run.
minimum_train_tokens: 2700000000
maximum_hours: 90
maximum_usd: 130
gpu_hourly_usd: 1.39
huggingface:
# DECISION (2026-07-24, reports/phase5_corpus.json): data was held at
# phase-4's frozen corpus, so v3 is a model-only release. Point the model
# card's dataset link at the existing v2 dataset repo instead of the
# placeholder v3 dataset repo in base.yaml (which is never published).
dataset_repos:
phase5: "trtd56/TimeCapsuleLLM-ja-corpus-v2"