File size: 1,361 Bytes
7034da5
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
"""Capture a dated, metadata-only inventory of published annotation objects."""
import argparse
from datetime import datetime, timezone
import hashlib
import json
from pathlib import Path
from huggingface_hub import HfApi,get_token
from remote_catalog import BUCKET

if __name__=='__main__':
    p=argparse.ArgumentParser(description=__doc__)
    p.add_argument('--output',type=Path,default=Path('.cache/full-index/objects.jsonl'))
    args=p.parse_args(); args.output.parent.mkdir(parents=True,exist_ok=True)
    temporary=args.output.with_suffix('.jsonl.tmp'); started=datetime.now(timezone.utc).isoformat(); count=0
    with temporary.open('w') as f:
        for x in HfApi(token=get_token()).list_bucket_tree(BUCKET,'annotations',recursive=True):
            if x.type=='file' and x.path.endswith('.parquet'):
                f.write(json.dumps({'path':x.path,'hash':x.xet_hash,'size':x.size})+'\n'); count+=1
                if count%50000==0: print(count,flush=True)
    temporary.replace(args.output)
    args.output.with_suffix('.source.json').write_text(json.dumps({'bucket':BUCKET,'listing_started_at':started,
        'listing_finished_at':datetime.now(timezone.utc).isoformat(),'objects':count,
        'sha256':hashlib.sha256(args.output.read_bytes()).hexdigest()},indent=2)+'\n')
    print('Saved',count,'published annotation objects',flush=True)