Cion-lab commited on
Commit
0a2e67e
·
verified ·
1 Parent(s): 80f5734

add p1_push_bench: Hub upload/download rate at checkpoint sizes

Browse files
Files changed (1) hide show
  1. probes/p1_push_bench.py +105 -0
probes/p1_push_bench.py ADDED
@@ -0,0 +1,105 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Phase 1: how fast can a Kaggle job push a checkpoint to the Hub, and does it survive verification?
2
+ #
3
+ # Why this is a Gate 1 question and not a Phase 3 discovery. §3.13's cycle is
4
+ # write -> upload -> VERIFY from the Hub independently -> delete local -> confirm free space, and it runs
5
+ # at every one of the 10 checkpoints plus the rolling `latest`. Two numbers decide whether that cycle is
6
+ # affordable: the upload rate, and the size of an exact-resume checkpoint. If a 1.6 GB checkpoint pushes in
7
+ # 25 s the cycle is free; if it takes 20 min then 11 of them are ~3.7 h of billed GPU time against a
8
+ # 30 h week, which changes the checkpoint cadence decision. Neither is knowable from a card or a blog.
9
+ #
10
+ # Sizes chosen from the real arithmetic, not convenience: 100M params in an exact-resume checkpoint is
11
+ # fp32 weights (400 MB) + fp32 Adam m and v (800 MB) + fp32 master/grad copies depending on recipe
12
+ # (~400 MB) + scheduler/RNG/scaler state (negligible) ~= 1.5-1.8 GB. 200 MB is the floor case (weights
13
+ # only, i.e. what a final export costs).
14
+ #
15
+ # Uses the credential helper, so no token appears in this file or in any kernel source. The probe repo is
16
+ # deleted at the end; nothing here is prior work worth keeping.
17
+
18
+ import hashlib
19
+ import json
20
+ import os
21
+ import time
22
+ import urllib.error
23
+ import urllib.request
24
+
25
+ R = {}
26
+ REPO = "Cion-lab/ounce100m-pushbench-DELETEME"
27
+
28
+
29
+ def guard(name, fn):
30
+ try:
31
+ R[name] = fn()
32
+ except Exception as e:
33
+ R[name] = {"error": f"{type(e).__name__}: {e}"[:280]}
34
+
35
+
36
+ import ounce100m_credentials # noqa: E402
37
+
38
+ guard("credentials", lambda: ounce100m_credentials.install(verify=True))
39
+ tok = os.environ.get("HF_TOKEN", "")
40
+
41
+ from huggingface_hub import HfApi, create_repo # noqa: E402
42
+
43
+ api = HfApi(token=tok)
44
+
45
+
46
+ def make_repo():
47
+ create_repo(repo_id=REPO, token=tok, repo_type="model", private=False, exist_ok=True)
48
+ return REPO
49
+
50
+
51
+ guard("repo", make_repo)
52
+
53
+
54
+ def push(mb):
55
+ """Time an upload, then verify it by reading back over plain anonymous HTTPS."""
56
+ n = int(mb * 1024**2)
57
+ # Upload a deterministic buffer rather than drawing 1.6 GB of urandom: urandom is CPU-bound and
58
+ # would measure the RNG, not the network. Size and byte-count verification still transfer.
59
+ buf = b"\x5a" * n
60
+ t0 = time.monotonic()
61
+ api.upload_file(path_or_fileobj=buf, path_in_repo=f"blob_{mb}mb.bin", repo_id=REPO,
62
+ repo_type="model", commit_message=f"push bench {mb} MB")
63
+ up = time.monotonic() - t0
64
+ url = f"https://huggingface.co/{REPO}/resolve/main/blob_{mb}mb.bin"
65
+ t1 = time.monotonic()
66
+ read = 0
67
+ with urllib.request.urlopen(url, timeout=600) as r:
68
+ while True:
69
+ chunk = r.read(4 * 1024**2)
70
+ if not chunk:
71
+ break
72
+ read += len(chunk)
73
+ down = time.monotonic() - t1
74
+ return {"MB": mb, "upload_s": round(up, 1), "upload_MB_per_s": round(n / up / 1024**2, 2),
75
+ "download_s": round(down, 1), "download_MB_per_s": round(read / down / 1024**2, 2),
76
+ "bytes_read_back": read, "size_matches": read == n}
77
+
78
+
79
+ for mb in (200, 800, 1600):
80
+ guard(f"push_{mb}MB", lambda mb=mb: push(mb))
81
+
82
+
83
+ def cleanup():
84
+ from huggingface_hub import delete_repo
85
+
86
+ delete_repo(repo_id=REPO, repo_type="model", token=tok)
87
+ try:
88
+ urllib.request.urlopen(f"https://huggingface.co/{REPO}", timeout=30)
89
+ return {"deleted": False, "still_resolves": True}
90
+ except urllib.error.HTTPError as e:
91
+ return {"deleted": True, "status_after": e.code}
92
+
93
+
94
+ guard("cleanup", cleanup)
95
+
96
+ R["disk_note"] = {
97
+ "kaggle_working_GB_total": 19.5,
98
+ "rule": "a checkpoint must be written somewhere, pushed, verified, then deleted before the next "
99
+ "one is produced; at 1.6 GB each, 10 retained checkpoints on the Hub is 16 GB of Hub "
100
+ "storage and at most 1-2 copies may exist on the instance at a time",
101
+ }
102
+
103
+ print("PROBE_JSON_BEGIN")
104
+ print(json.dumps(R, indent=1, default=str))
105
+ print("PROBE_JSON_END")