Spaces:
Running
Running
File size: 3,342 Bytes
da5cba1 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 | """Index datasets ahead of time: their indexes and packs, ready in STORAGE_DIR before anyone opens them.
uv run python -m app.precache --featured --top 40 # the collections and the 40 trending Harbor datasets
uv run python -m app.precache FineEnvs/some-dataset ... # these
hf buckets sync .local-data/indexes hf://buckets/<bucket>/indexes # then copy them to the Space's bucket
hf buckets sync .local-data/packs hf://buckets/<bucket>/packs
Public datasets only. With --use-token, files are downloaded with your HF token for its higher rate limits;
listings and access checks stay anonymous, so nothing private is read.
"""
from __future__ import annotations
import argparse
import os
import sys
import time
def main() -> int:
ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
ap.add_argument("specs", nargs="*", help="dataset ids, org/name")
ap.add_argument("--featured", action="store_true", help="every dataset in the collections")
ap.add_argument("--top", type=int, default=0, help="the N trending Harbor datasets on the Hub")
ap.add_argument("--use-token", action="store_true", help="download public files with your HF token")
ap.add_argument("--force", action="store_true", help="rebuild even when an index is current")
args = ap.parse_args()
if args.use_token:
from huggingface_hub import get_token
os.environ["RLX_INDEX_TOKEN"] = get_token() or ""
from . import catalog, config # after the token is in the environment
config.INDEX_DIR.mkdir(parents=True, exist_ok=True)
specs = list(args.specs)
if args.featured:
specs += catalog.featured_datasets()
if args.top:
rows = sorted(catalog.environments(), key=lambda d: d["trending"] * 1e9 + d["downloads"], reverse=True)
specs += [d["id"] for d in rows if d.get("kind") == "dataset" and d.get("framework", "harbor") == "harbor"][: args.top]
seen, failed = set(), 0
for spec in specs:
if spec in seen:
continue
seen.add(spec)
t = time.time()
try:
meta = catalog.info(spec)
if meta.get("framework") != "harbor" and not catalog.looks_harbor(spec, meta["sha"]):
print(f" rows {spec}: read row by row (app/envs), nothing to index")
continue
if not args.force and catalog._read_index(spec, meta["sha"]) and catalog._read_pack(spec, meta["sha"]):
print(f" current {spec}")
continue
job: dict = {}
idx, pack = catalog.build_index(spec, meta, job)
if pack is not None:
catalog._write_pack(spec, meta["sha"], pack)
catalog._write_index(idx)
print(f" indexed {spec}: {len(idx['tasks']):,} tasks in {time.time() - t:.0f}s"
+ ("" if idx["tasks"] else f" ({idx.get('note')})"), flush=True)
except Exception as exc: # noqa: BLE001 - one dataset failing leaves the rest
failed += 1
print(f" failed {spec}: {type(exc).__name__}: {str(exc)[:200]}", flush=True)
print(f"{len(seen) - failed} of {len(seen)} datasets ready in {config.STORAGE_DIR}")
return 1 if failed else 0
if __name__ == "__main__":
sys.exit(main())
|