| # stream_data.py - rolling-shard streaming tokenizer (v13/v15 Stage 0). | |
| # Streams Cosmopedia + FineWeb-Edu, tokenizes with the project BPE, writes rolling | |
| # shards of SHARD_TOK tokens each; keeps at most KEEP shards on disk (delete oldest). | |
| # UNIQUE tokens forever - never re-reads (measured overfit wall, v12 8.2). | |
| import numpy as np, itertools, sys | |
| from pathlib import Path | |
| from datasets import load_dataset | |
| from tokenizers import Tokenizer | |
| OUT=Path('/root/dna/data'); OUT.mkdir(parents=True,exist_ok=True) | |
| tok=Tokenizer.from_file('/root/dna/fineweb-tokenizer.json') | |
| SHARD_TOK=25_000_000 | |
| KEEP=40 | |
| def streams(): | |
| fw=load_dataset('HuggingFaceFW/fineweb-edu','sample-100BT',split='train',streaming=True) | |
| co=load_dataset('HuggingFaceTB/cosmopedia','web_samples_v2',split='train',streaming=True) | |
| fi=iter(fw); ci=iter(co) | |
| while True: | |
| for _ in range(3): | |
| yield next(fi)['text'] | |
| try: yield next(ci)['text'] | |
| except Exception: pass | |
| def main(): | |
| it=streams(); si=0; buf=np.empty(SHARD_TOK,np.uint16); n=0; total=0 | |
| for txt in it: | |
| ids=tok.encode(txt).ids | |
| for x in ids: | |
| buf[n]=x; n+=1 | |
| if n==SHARD_TOK: | |
| tmp=OUT/f'shard_{si:05d}.tmp'; buf.tofile(tmp) | |
| tmp.rename(OUT/f'shard_{si:05d}.u16') | |
| total+=n; print(f'SHARD {si} done total_tokens {total:,}',flush=True) | |
| si+=1; n=0 | |
| old=sorted(OUT.glob('shard_*.u16')) | |
| while len(old)>KEEP: | |
| old[0].unlink(); old=old[1:] | |
| if __name__=='__main__': main() | |