File size: 2,998 Bytes
b296ad4 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 | """Build a Hugging Face viewer table plus lossless compressed source records."""
import argparse
import gzip
import hashlib
import json
from pathlib import Path
import shutil
import pyarrow as pa
import pyarrow.parquet as pq
def main():
p=argparse.ArgumentParser(); p.add_argument('--data',required=True);p.add_argument('--out',required=True)
p.add_argument('--teacher-source'); args=p.parse_args(); source=Path(args.data); out=Path(args.out)
(out/'data').mkdir(parents=True,exist_ok=True); (out/'raw').mkdir(exist_ok=True)
counts={}
for split in ['train','validation','test','manual']+(['development'] if (source/'development.jsonl').exists() else []):
writer=None; batch=[]; count=0
with (source/(split+'.jsonl')).open() as stream, gzip.open(out/'raw'/(split+'.jsonl.gz'),'wt',encoding='utf-8') as raw:
for line in stream:
raw.write(line); row=json.loads(line)
# Heterogeneous runtime JSON schemas remain lossless strings in Arrow.
flat={key:row.get(key,'') for key in ['id','scenario_id','language','backend','operation','question','prompt','response','provenance']}
flat['split']=split
flat['sample_weight']=float(row.get('sample_weight',1))
for key in ['context','target','slots']: flat[key]=json.dumps(row[key],ensure_ascii=False,separators=(',',':'))
batch.append(flat); count+=1
if len(batch)==2000:
table=pa.Table.from_pylist(batch)
if writer is None: writer=pq.ParquetWriter(out/'data'/(split+'.parquet'),table.schema,compression='zstd')
writer.write_table(table); batch=[]
if batch:
table=pa.Table.from_pylist(batch)
if writer is None: writer=pq.ParquetWriter(out/'data'/(split+'.parquet'),table.schema,compression='zstd')
writer.write_table(table)
if writer: writer.close()
counts[split]=count
(out/'provenance').mkdir(exist_ok=True)
for name in ['tokenization.json','grounding-stats.json','split-audit.json','full-audit.json']:
if (source/name).exists(): shutil.copy2(source/name,out/'provenance'/name)
if args.teacher_source:
teacher=Path(args.teacher_source)
for name in ['templates.jsonl','verified-concrete.jsonl']:
if (teacher/name).exists():
with (teacher/name).open('rb') as src,gzip.open(out/'provenance'/(name+'.gz'),'wb') as dest: shutil.copyfileobj(src,dest)
manifest={'splits':counts,'files':{}}
for file in sorted(out.rglob('*')):
if file.is_file() and file.name!='manifest.json':
digest=hashlib.sha256(file.read_bytes()).hexdigest()
manifest['files'][str(file.relative_to(out))]={'bytes':file.stat().st_size,'sha256':digest}
(out/'manifest.json').write_text(json.dumps(manifest,indent=2));print(json.dumps(counts))
if __name__=='__main__': main()
|