"""Streaming encode of a large raw-text corpus to a uint16 .bin file. Memory-safe: encodes line-by-line, flushes chunks of ~16M tokens. Usage: .venv/bin/python data/encode_full.py --raw data/TinyStoriesV2-GPT4-train.txt \ --out data/train_full.bin --tok data/tokenizer.json """ import argparse, time from pathlib import Path import numpy as np from data.tokenizer import load_tokenizer def main(): ap = argparse.ArgumentParser() ap.add_argument("--raw", default="data/TinyStoriesV2-GPT4-train.txt") ap.add_argument("--out", default="data/train_full.bin") ap.add_argument("--tok", default="data/tokenizer.json") ap.add_argument("--chunk", type=int, default=16_000_000) ap.add_argument("--max-tokens", type=int, default=600_000_000) args = ap.parse_args() tok = load_tokenizer(args.tok) eot = tok.token_to_id("<|endoftext|>") out = Path(args.out) out.parent.mkdir(parents=True, exist_ok=True) total, buf = 0, [] t0 = time.time() with open(args.raw, "rb") as f, open(out, "wb") as g: for raw in f: line = raw.decode("utf-8", errors="replace").strip() if not line: continue ids = tok.encode(line).ids buf.extend(ids) buf.append(eot) total += len(ids) + 1 if len(buf) >= args.chunk or total >= args.max_tokens: np.asarray(buf, dtype=np.uint16).tofile(g) buf.clear() print(f"encoded {total:,} tokens in {time.time()-t0:.0f}s " f"({total/(time.time()-t0):,.0f} tok/s)", flush=True) if total >= args.max_tokens: break if buf: np.asarray(buf, dtype=np.uint16).tofile(g) print(f"done: {total:,} tokens -> {out} ({out.stat().st_size/1e9:.2f} GB)", flush=True) if __name__ == "__main__": main()