hashundefined/full / upload_parts.py
hashundefined's picture
download
raw
2.04 kB
"""Upload split index parts to HF via the classic HTTP path (Xet disabled).
Windows Xet upload stalls on big files; the HTTP fallback caps at 50 GB/file,
so the <50 GB parts from split_idx.py are uploaded here. Resumable: parts
already present on HF are skipped. Watches the folder until all parts done.
"""
import glob
import os
import sys
import time
os.environ["HF_HUB_DISABLE_XET"] = "1"
from huggingface_hub import HfApi
BASE = os.path.dirname(os.path.abspath(__file__))
REPO = "Kzr0xx/icrm-hitek-full-db-mixed"
def get_token():
t = os.environ.get("HF_TOKEN")
if t:
return t
try:
with open(os.path.join(BASE, "hf_token.txt"), encoding="utf-8") as f:
t = f.read().strip()
if t:
return t
except OSError:
pass
sys.exit("HF_TOKEN env var or hf_token.txt required")
def main():
api = HfApi(token=get_token())
while True:
parts = sorted(glob.glob(os.path.join(BASE, "idx_*.?.parquet")))
pending = []
for p in parts:
name = os.path.basename(p)
try:
if api.file_exists(repo_id=REPO, filename=name,
repo_type="dataset"):
continue
except Exception as e:
print(f" check {name}: {e}", flush=True)
pending.append(p)
if not pending and parts:
print("ALL UPLOADED", flush=True)
break
for p in pending:
name = os.path.basename(p)
print(f"UPLOAD {name} ({os.path.getsize(p)/1073741824:.1f} GiB)",
flush=True)
t0 = time.time()
try:
api.upload_file(path_or_fileobj=p, path_in_repo=name,
repo_id=REPO, repo_type="dataset")
print(f"DONE {name} in {time.time()-t0:.0f}s", flush=True)
except Exception as e:
print(f"FAIL {name}: {e}", flush=True)
time.sleep(30)
if __name__ == "__main__":
main()

Xet Storage Details

Size:
2.04 kB
·
Xet hash:
e5829ab023ca13e0cb784fb1656ed64041fcf0a420bf4968847ff51477275a9d

Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.