fsi-anomaly / data /encode_full.py
FerrellSyntheticIntelligence's picture
backup all: 100 files (batch)
1c0d385 verified
Raw
History Blame Contribute Delete
1.88 kB
"""Streaming encode of a large raw-text corpus to a uint16 .bin file.
Memory-safe: encodes line-by-line, flushes chunks of ~16M tokens.
Usage:
.venv/bin/python data/encode_full.py --raw data/TinyStoriesV2-GPT4-train.txt \
--out data/train_full.bin --tok data/tokenizer.json
"""
import argparse, time
from pathlib import Path
import numpy as np
from data.tokenizer import load_tokenizer
def main():
ap = argparse.ArgumentParser()
ap.add_argument("--raw", default="data/TinyStoriesV2-GPT4-train.txt")
ap.add_argument("--out", default="data/train_full.bin")
ap.add_argument("--tok", default="data/tokenizer.json")
ap.add_argument("--chunk", type=int, default=16_000_000)
ap.add_argument("--max-tokens", type=int, default=600_000_000)
args = ap.parse_args()
tok = load_tokenizer(args.tok)
eot = tok.token_to_id("<|endoftext|>")
out = Path(args.out)
out.parent.mkdir(parents=True, exist_ok=True)
total, buf = 0, []
t0 = time.time()
with open(args.raw, "rb") as f, open(out, "wb") as g:
for raw in f:
line = raw.decode("utf-8", errors="replace").strip()
if not line:
continue
ids = tok.encode(line).ids
buf.extend(ids)
buf.append(eot)
total += len(ids) + 1
if len(buf) >= args.chunk or total >= args.max_tokens:
np.asarray(buf, dtype=np.uint16).tofile(g)
buf.clear()
print(f"encoded {total:,} tokens in {time.time()-t0:.0f}s "
f"({total/(time.time()-t0):,.0f} tok/s)", flush=True)
if total >= args.max_tokens:
break
if buf:
np.asarray(buf, dtype=np.uint16).tofile(g)
print(f"done: {total:,} tokens -> {out} ({out.stat().st_size/1e9:.2f} GB)", flush=True)
if __name__ == "__main__":
main()