| """Build a Hugging Face viewer table plus lossless compressed source records.""" |
| import argparse |
| import gzip |
| import hashlib |
| import json |
| from pathlib import Path |
| import shutil |
| import pyarrow as pa |
| import pyarrow.parquet as pq |
|
|
|
|
| def main(): |
| p=argparse.ArgumentParser(); p.add_argument('--data',required=True);p.add_argument('--out',required=True) |
| p.add_argument('--teacher-source'); args=p.parse_args(); source=Path(args.data); out=Path(args.out) |
| (out/'data').mkdir(parents=True,exist_ok=True); (out/'raw').mkdir(exist_ok=True) |
| counts={} |
| for split in ['train','validation','test','manual']+(['development'] if (source/'development.jsonl').exists() else []): |
| writer=None; batch=[]; count=0 |
| with (source/(split+'.jsonl')).open() as stream, gzip.open(out/'raw'/(split+'.jsonl.gz'),'wt',encoding='utf-8') as raw: |
| for line in stream: |
| raw.write(line); row=json.loads(line) |
| |
| flat={key:row.get(key,'') for key in ['id','scenario_id','language','backend','operation','question','prompt','response','provenance']} |
| flat['split']=split |
| flat['sample_weight']=float(row.get('sample_weight',1)) |
| for key in ['context','target','slots']: flat[key]=json.dumps(row[key],ensure_ascii=False,separators=(',',':')) |
| batch.append(flat); count+=1 |
| if len(batch)==2000: |
| table=pa.Table.from_pylist(batch) |
| if writer is None: writer=pq.ParquetWriter(out/'data'/(split+'.parquet'),table.schema,compression='zstd') |
| writer.write_table(table); batch=[] |
| if batch: |
| table=pa.Table.from_pylist(batch) |
| if writer is None: writer=pq.ParquetWriter(out/'data'/(split+'.parquet'),table.schema,compression='zstd') |
| writer.write_table(table) |
| if writer: writer.close() |
| counts[split]=count |
| (out/'provenance').mkdir(exist_ok=True) |
| for name in ['tokenization.json','grounding-stats.json','split-audit.json','full-audit.json']: |
| if (source/name).exists(): shutil.copy2(source/name,out/'provenance'/name) |
| if args.teacher_source: |
| teacher=Path(args.teacher_source) |
| for name in ['templates.jsonl','verified-concrete.jsonl']: |
| if (teacher/name).exists(): |
| with (teacher/name).open('rb') as src,gzip.open(out/'provenance'/(name+'.gz'),'wb') as dest: shutil.copyfileobj(src,dest) |
| manifest={'splits':counts,'files':{}} |
| for file in sorted(out.rglob('*')): |
| if file.is_file() and file.name!='manifest.json': |
| digest=hashlib.sha256(file.read_bytes()).hexdigest() |
| manifest['files'][str(file.relative_to(out))]={'bytes':file.stat().st_size,'sha256':digest} |
| (out/'manifest.json').write_text(json.dumps(manifest,indent=2));print(json.dumps(counts)) |
|
|
|
|
| if __name__=='__main__': main() |
|
|