"""Build a Hugging Face viewer table plus lossless compressed source records.""" import argparse import gzip import hashlib import json from pathlib import Path import shutil import pyarrow as pa import pyarrow.parquet as pq def main(): p=argparse.ArgumentParser(); p.add_argument('--data',required=True);p.add_argument('--out',required=True) p.add_argument('--teacher-source'); args=p.parse_args(); source=Path(args.data); out=Path(args.out) (out/'data').mkdir(parents=True,exist_ok=True); (out/'raw').mkdir(exist_ok=True) counts={} for split in ['train','validation','test','manual']+(['development'] if (source/'development.jsonl').exists() else []): writer=None; batch=[]; count=0 with (source/(split+'.jsonl')).open() as stream, gzip.open(out/'raw'/(split+'.jsonl.gz'),'wt',encoding='utf-8') as raw: for line in stream: raw.write(line); row=json.loads(line) # Heterogeneous runtime JSON schemas remain lossless strings in Arrow. flat={key:row.get(key,'') for key in ['id','scenario_id','language','backend','operation','question','prompt','response','provenance']} flat['split']=split flat['sample_weight']=float(row.get('sample_weight',1)) for key in ['context','target','slots']: flat[key]=json.dumps(row[key],ensure_ascii=False,separators=(',',':')) batch.append(flat); count+=1 if len(batch)==2000: table=pa.Table.from_pylist(batch) if writer is None: writer=pq.ParquetWriter(out/'data'/(split+'.parquet'),table.schema,compression='zstd') writer.write_table(table); batch=[] if batch: table=pa.Table.from_pylist(batch) if writer is None: writer=pq.ParquetWriter(out/'data'/(split+'.parquet'),table.schema,compression='zstd') writer.write_table(table) if writer: writer.close() counts[split]=count (out/'provenance').mkdir(exist_ok=True) for name in ['tokenization.json','grounding-stats.json','split-audit.json','full-audit.json']: if (source/name).exists(): shutil.copy2(source/name,out/'provenance'/name) if args.teacher_source: teacher=Path(args.teacher_source) for name in ['templates.jsonl','verified-concrete.jsonl']: if (teacher/name).exists(): with (teacher/name).open('rb') as src,gzip.open(out/'provenance'/(name+'.gz'),'wb') as dest: shutil.copyfileobj(src,dest) manifest={'splits':counts,'files':{}} for file in sorted(out.rglob('*')): if file.is_file() and file.name!='manifest.json': digest=hashlib.sha256(file.read_bytes()).hexdigest() manifest['files'][str(file.relative_to(out))]={'bytes':file.stat().st_size,'sha256':digest} (out/'manifest.json').write_text(json.dumps(manifest,indent=2));print(json.dumps(counts)) if __name__=='__main__': main()