"""Capture a dated, metadata-only inventory of published annotation objects.""" import argparse from datetime import datetime, timezone import hashlib import json from pathlib import Path from huggingface_hub import HfApi,get_token from remote_catalog import BUCKET if __name__=='__main__': p=argparse.ArgumentParser(description=__doc__) p.add_argument('--output',type=Path,default=Path('.cache/full-index/objects.jsonl')) args=p.parse_args(); args.output.parent.mkdir(parents=True,exist_ok=True) temporary=args.output.with_suffix('.jsonl.tmp'); started=datetime.now(timezone.utc).isoformat(); count=0 with temporary.open('w') as f: for x in HfApi(token=get_token()).list_bucket_tree(BUCKET,'annotations',recursive=True): if x.type=='file' and x.path.endswith('.parquet'): f.write(json.dumps({'path':x.path,'hash':x.xet_hash,'size':x.size})+'\n'); count+=1 if count%50000==0: print(count,flush=True) temporary.replace(args.output) args.output.with_suffix('.source.json').write_text(json.dumps({'bucket':BUCKET,'listing_started_at':started, 'listing_finished_at':datetime.now(timezone.utc).isoformat(),'objects':count, 'sha256':hashlib.sha256(args.output.read_bytes()).hexdigest()},indent=2)+'\n') print('Saved',count,'published annotation objects',flush=True)