decoderstack-d24 / logs /run_full_d24_w8.sh
ChrisMcCormick's picture
Add model card, meta.json, logs, tokenizer, training source, and nanochat converter
11f07f9 verified
Raw
History Blame Contribute Delete
3.78 kB
#!/usr/bin/env bash
# THE FULL-HORIZON d24 RUN: 5,568 steps, world=8, bf16, total batch 2^20.
# ~1.87 h at the measured 1,208 ms/step. This is the run the checkpoints come
# from, so it is the deliverable.
#
# Config, and why each value:
# micro_batch_tokens 32768 -> 65536 grad accum 4 -> 2. Measured 1,208 vs
# 1,249 ms/step (~3.2% faster) at 71,568 MiB peak allocated against the
# card's 81,559 -- 8.8 GB spare. Confirmed end-to-end at this size
# including checkpoint capture and CORE eval (db64k_confirm.log).
# max_num_docs 96 -> 192 Worst-case packing over all 79 train
# shards this run reads is 162 docs at 65,536 tokens, so 163 is the
# minimum (the assert is `len(cum_lengths) < max_num_docs`). 192 gives
# 18% margin at zero measured cost: an A/B at 163/192/256 came out
# 1207.6/1209.7/1207.3 ms -- ghost cu_seqlens entries do not register at
# d24 (scan_max_docs_sweep.log).
# NUM_TRAIN_SHARDS 20 -> 80 Horizon is 5,838,471,168 tokens; shards
# 1..79 hold 7.9B raw ~= 6.7B usable after seq_len truncation, +876M
# margin. All 79 are already on disk (prefetch_train_shards.log).
#
# Left at the file's defaults deliberately: num_iterations 5568, warmup 40,
# val_loss_every 250, save_checkpoint True, save_steps (1950,) + the always-saved
# final -- the two capture points Chris chose (1950 is the last uncooled state;
# the cooldown covers steps 1950-5567).
#
# NOTE: val bpb is NOT comparable to the eval100a-e baselines -- the validation
# shard was rebuilt from shard_06542 to match nanochat's pinned val split
# (regen_val_shard_06542.py). It IS comparable to the upstream baseline curve in
# upstream_speedrun_d24_bf16_world8.log.
#
# Checkpoints land in $RUN_DIR/logs/<run_id>/ (under ~/.cache). Push them with
# python ~/agent-ops/stacks/2026-07-31_0822am_fable-rewrite-handoff/push_checkpoints.py \
# <run_dir>/logs/<run_id> <hf-model-repo>
# BEFORE releasing the box.
set -eu
source ~/nanochat/.venv/bin/activate
source ~/env.sh
export OMP_NUM_THREADS=1
export DATA_PATH="$HOME/.cache/stacks"
export PYTHONUNBUFFERED=1
SRC=~/agent-ops/stacks/2026-07-31_0822am_fable-rewrite-handoff
SESS=~/agent-ops/stacks/2026-07-31_0627pm_8xh100-d24-baseline-world8
RUN_DIR="$DATA_PATH/run_full_d24_w8"
mkdir -p "$RUN_DIR"
cp "$SRC/decoderstack_medium_pt-sft-fable-v3.py" "$RUN_DIR/run_full_d24_w8.py"
cd "$RUN_DIR"
sed -i 's/ micro_batch_tokens: int = 32768/ micro_batch_tokens: int = 65536/' run_full_d24_w8.py
sed -i 's/^ max_num_docs = 96$/ max_num_docs = 192/' run_full_d24_w8.py
# NB: no `$` anchor here -- this line carries a trailing comment, and a
# $-anchored pattern silently matched nothing (the earlier eval run-scripts
# have the same latent no-op; harmless there because NUM_TRAIN_SHARDS only
# gates DOWNLOADS, never the loader, which globs every train_*.bin on disk).
sed -i 's/^NUM_TRAIN_SHARDS = 20/NUM_TRAIN_SHARDS = 80/' run_full_d24_w8.py
echo "=== applied run-copy edits ==="
grep -n "micro_batch_tokens: int\|max_num_docs = \|NUM_TRAIN_SHARDS = \|num_iterations: int\|save_steps:\|save_checkpoint: bool" run_full_d24_w8.py
# Fail loudly on a sed that matched nothing, rather than starting a ~2 h run on
# a config nobody verified.
check() { grep -qE "$1" run_full_d24_w8.py || { echo "EDIT FAILED: $1"; exit 1; }; }
check '^ micro_batch_tokens: int = 65536'
check '^ max_num_docs = 192$'
check '^NUM_TRAIN_SHARDS = 80'
check '^ num_iterations: int = 5568$'
check '^ save_steps: tuple = \(1950,\)$'
check '^ save_checkpoint: bool = True$'
echo "=== all run-copy edits verified ==="
echo "=== run dir: $RUN_DIR ==="
torchrun --standalone --nproc_per_node=8 run_full_d24_w8.py 2>&1 | tee "$SESS/full_d24_w8.log"