Download list_annotation_objects.py from HuggingFaceBio/carbon-a-database-explorer: direct link, hf CLI and curl.
- Browser
- Download file 1.36 kB
-
https://huggingface.co/spaces/HuggingFaceBio/carbon-a-database-explorer/resolve/refs%2Fpr%2F12/list_annotation_objects.py
- Command line
-
hf download hf://spaces/HuggingFaceBio/carbon-a-database-explorer@refs/pr/12/list_annotation_objects.py
-
curl -L -o list_annotation_objects.py https://huggingface.co/spaces/HuggingFaceBio/carbon-a-database-explorer/resolve/refs%2Fpr%2F12/list_annotation_objects.py
1.36 kB
| """Capture a dated, metadata-only inventory of published annotation objects.""" | |
| import argparse | |
| from datetime import datetime, timezone | |
| import hashlib | |
| import json | |
| from pathlib import Path | |
| from huggingface_hub import HfApi,get_token | |
| from remote_catalog import BUCKET | |
| if __name__=='__main__': | |
| p=argparse.ArgumentParser(description=__doc__) | |
| p.add_argument('--output',type=Path,default=Path('.cache/full-index/objects.jsonl')) | |
| args=p.parse_args(); args.output.parent.mkdir(parents=True,exist_ok=True) | |
| temporary=args.output.with_suffix('.jsonl.tmp'); started=datetime.now(timezone.utc).isoformat(); count=0 | |
| with temporary.open('w') as f: | |
| for x in HfApi(token=get_token()).list_bucket_tree(BUCKET,'annotations',recursive=True): | |
| if x.type=='file' and x.path.endswith('.parquet'): | |
| f.write(json.dumps({'path':x.path,'hash':x.xet_hash,'size':x.size})+'\n'); count+=1 | |
| if count%50000==0: print(count,flush=True) | |
| temporary.replace(args.output) | |
| args.output.with_suffix('.source.json').write_text(json.dumps({'bucket':BUCKET,'listing_started_at':started, | |
| 'listing_finished_at':datetime.now(timezone.utc).isoformat(),'objects':count, | |
| 'sha256':hashlib.sha256(args.output.read_bytes()).hexdigest()},indent=2)+'\n') | |
| print('Saved',count,'published annotation objects',flush=True) | |