Cion-lab commited on
Commit
80f5734
·
verified ·
1 Parent(s): 90c1cde

add credential helper + Hub write/bulk-upload roundtrip probe

Browse files
ounce100m_credentials.py ADDED
@@ -0,0 +1,88 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Runtime credentials for ounce100m jobs, fetched from the account's own private Kaggle dataset.
2
+
3
+ Why this exists. §2 forbids the HF token reaching a public artifact -- job logs, model cards, or any
4
+ code mirrored to the Hub -- and requires it to come "from the environment or a secret store". Internet-
5
+ enabled Kaggle sessions arrive already authenticated against the Kaggle API (verified: CLI 2.0.2
6
+ preinstalled, `auth_method: ACCESS_TOKEN`), and `dodosoomro/ounce100m-secret-store` is a private dataset
7
+ that is anonymously unreachable and absent from public search. So a job can retrieve its own credential
8
+ at run time: no token in kernel source, no token in the public code repo, no token in any log.
9
+
10
+ Note that `datasetDataSources` mounting was tried and did NOT populate /kaggle/input (see
11
+ memory/ERRORS.md), so this module downloads instead of mounting. Download also has the advantage of
12
+ working in a freshly created kernel with no metadata to keep in sync.
13
+
14
+ Rules for callers:
15
+ * never print, log, or write the token to /kaggle/working -- working-dir files become kernel outputs
16
+ * call `install()` once at process start, before any huggingface_hub import
17
+ * the staging directory is under /tmp, which is not published as an artifact
18
+ """
19
+
20
+ import json
21
+ import os
22
+ import subprocess
23
+ import zipfile
24
+
25
+ DS_SLUG = "ounce100m-secret-store"
26
+ DS_ID = f"dodosoomro/{DS_SLUG}"
27
+ STAGE = "/tmp/.ounce100m" # deliberately not under /kaggle/working
28
+ CREDS = "credentials.json"
29
+
30
+
31
+ class CredentialError(RuntimeError):
32
+ pass
33
+
34
+
35
+ def _download():
36
+ os.makedirs(STAGE, exist_ok=True)
37
+ os.chmod(STAGE, 0o700)
38
+ r = subprocess.run(
39
+ ["kaggle", "datasets", "download", "-d", DS_ID, "-p", STAGE, "--unzip"],
40
+ capture_output=True, text=True, timeout=300)
41
+ if r.returncode != 0:
42
+ # Scrubbed: an error string from the CLI could otherwise echo the token into a retained log.
43
+ raise CredentialError(f"kaggle datasets download rc={r.returncode}")
44
+ path = os.path.join(STAGE, CREDS)
45
+ if not os.path.exists(path):
46
+ # --unzip flattens differently across CLI versions; fall back to opening the archive directly.
47
+ zipped = os.path.join(STAGE, f"{DS_SLUG}.zip")
48
+ if os.path.exists(zipped):
49
+ with zipfile.ZipFile(zipped) as z:
50
+ z.extract(CREDS, STAGE)
51
+ path = os.path.join(STAGE, CREDS)
52
+ if not os.path.exists(path):
53
+ raise CredentialError(f"{CREDS} not found after download; staged: {sorted(os.listdir(STAGE))}")
54
+ return path
55
+
56
+
57
+ def token():
58
+ """The HF token as a string. Callers must not print it."""
59
+ path = os.environ.get("OUNCE100M_CRED_PATH") # lets a mounting job hand us a path instead
60
+ if not path:
61
+ path = _download()
62
+ with open(path) as f:
63
+ blob = json.load(f)
64
+ tok = blob.get("HF_TOKEN", "")
65
+ if not tok.startswith("hf_") or len(tok) < 20:
66
+ raise CredentialError("credential file present but HF_TOKEN malformed -- refusing to continue")
67
+ return tok
68
+
69
+
70
+ def install(verify=False):
71
+ """Put the token where huggingface_hub expects it. Returns a *safe to print* summary only."""
72
+ import hashlib
73
+
74
+ tok = token()
75
+ os.environ["HF_TOKEN"] = tok
76
+ os.environ["HUGGING_FACE_HUB_TOKEN"] = tok
77
+ out = {"source": DS_ID, "length": len(tok),
78
+ "sha256_prefix": hashlib.sha256(tok.encode()).hexdigest()[:12]}
79
+ if verify:
80
+ from huggingface_hub import HfApi
81
+
82
+ me = HfApi(token=tok).whoami()
83
+ out["hub_user"] = me.get("name")
84
+ return out
85
+
86
+
87
+ if __name__ == "__main__":
88
+ print(json.dumps(install(verify=True), indent=1))
probes/p1_credential_roundtrip.py ADDED
@@ -0,0 +1,123 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Phase 1: prove the credential path end to end, then prove a job can WRITE to the Hub.
2
+ #
3
+ # This is the test that closes D-002. The token arrives via `ounce100m_credentials`, which downloads it
4
+ # from the account's own private Kaggle dataset using the container's built-in Kaggle auth -- so nothing
5
+ # here, and nothing in any kernel source or mirrored repo, contains a credential.
6
+ #
7
+ # The write test matters more than it looks: §3.13's whole cycle (write -> upload -> verify from the Hub
8
+ # -> delete local -> confirm free space) is unimplementable if a job cannot push. It is verified by
9
+ # reading the file back over plain anonymous HTTPS rather than through the client that wrote it, so a
10
+ # library-level false success cannot be mistaken for a working upload path.
11
+ #
12
+ # Zero GPU quota. No token printed anywhere.
13
+
14
+ import hashlib
15
+ import json
16
+ import os
17
+ import subprocess
18
+ import time
19
+ import urllib.error
20
+ import urllib.request
21
+
22
+ R = {}
23
+ TEST_REPO = "Cion-lab/ounce100m-authtest-DELETEME"
24
+
25
+
26
+ def guard(name, fn):
27
+ try:
28
+ R[name] = fn()
29
+ except Exception as e:
30
+ R[name] = {"error": f"{type(e).__name__}: {e}"[:300]}
31
+
32
+
33
+ guard("credentials_module", lambda: __import__("ounce100m_credentials").install(verify=True))
34
+
35
+ tok = os.environ.get("HF_TOKEN", "")
36
+ R["token_in_env_after_install"] = {"present": bool(tok), "length": len(tok)}
37
+
38
+
39
+ def staging_is_clean():
40
+ # /kaggle/working files are published as kernel output artifacts; a credential must never land there.
41
+ listing = sorted(os.listdir("/kaggle/working"))
42
+ leaked = [f for f in listing if "credential" in f.lower() or "token" in f.lower()]
43
+ return {"working_listing": listing, "credential_in_working_dir": leaked}
44
+
45
+
46
+ guard("artifact_hygiene", staging_is_clean)
47
+
48
+
49
+ def write_roundtrip():
50
+ from huggingface_hub import HfApi, create_repo, delete_repo
51
+
52
+ api = HfApi(token=tok)
53
+ t0 = time.monotonic()
54
+ # hub 1.11 (the image's version) has no `namespace` kwarg on create_repo -- the owner comes from
55
+ # the repo_id. Passing namespace= was what broke the earlier attempt (E-013).
56
+ create_repo(repo_id=TEST_REPO, token=tok, repo_type="model", private=False, exist_ok=True)
57
+ payload = b"ounce100m credential round-trip probe\n"
58
+ api.upload_file(path_or_fileobj=payload, path_in_repo="authcheck.txt",
59
+ repo_id=TEST_REPO, repo_type="model", commit_message="auth write check")
60
+ url = f"https://huggingface.co/{TEST_REPO}/resolve/main/authcheck.txt"
61
+ readback = urllib.request.urlopen(url, timeout=60).read()
62
+ verified = readback.strip() == payload.strip()
63
+ sha = hashlib.sha256(readback).hexdigest()[:12]
64
+ info = api.repo_info(TEST_REPO, repo_type="model")
65
+ delete_repo(repo_id=TEST_REPO, repo_type="model", token=tok)
66
+ gone = False
67
+ try:
68
+ urllib.request.urlopen(url, timeout=30)
69
+ except urllib.error.HTTPError as e:
70
+ gone = e.code in (404, 401, 403)
71
+ except Exception:
72
+ pass
73
+ return {"created": True, "bytes": len(payload), "readback_matches": verified,
74
+ "readback_sha256_prefix": sha, "revision": getattr(info, "sha", None),
75
+ "seconds": round(time.monotonic() - t0, 1), "deleted": True,
76
+ "delete_confirmed_gone": gone}
77
+
78
+
79
+ guard("hub_write_roundtrip", write_roundtrip)
80
+
81
+ # A throughput datapoint for the thing the main run actually does: push a ~200 MB blob and time it.
82
+ # Checkpoint size arithmetic (§Phase 1 open item) is meaningless without knowing the upload rate.
83
+ def bulk_upload():
84
+ from huggingface_hub import HfApi
85
+
86
+ api = HfApi(token=tok)
87
+ blob = os.urandom(200 * 1024**2)
88
+ sha = hashlib.sha256(blob).hexdigest()
89
+ t0 = time.monotonic()
90
+ api.upload_file(path_or_fileobj=blob, path_in_repo="bulk.bin", repo_id=TEST_REPO,
91
+ repo_type="model", commit_message="bulk upload timing probe")
92
+ up = time.monotonic() - t0
93
+ url = f"https://huggingface.co/{TEST_REPO}/resolve/main/bulk.bin"
94
+ t1 = time.monotonic()
95
+ got = urllib.request.urlopen(url, timeout=300).read()
96
+ down = time.monotonic() - t1
97
+ return {"MB": round(len(blob) / 1024**2), "upload_s": round(up, 1),
98
+ "upload_MB_per_s": round(len(blob) / up / 1024**2, 1),
99
+ "download_s": round(down, 1), "download_MB_per_s": round(len(got) / down / 1024**2, 1),
100
+ "sha_matches": hashlib.sha256(got).hexdigest() == sha,
101
+ "note": "LFS threshold may route this differently than a real multi-GB checkpoint"}
102
+
103
+
104
+ guard("bulk_upload_rate", bulk_upload)
105
+
106
+
107
+ def cleanup():
108
+ from huggingface_hub import delete_repo
109
+
110
+ delete_repo(repo_id=TEST_REPO, repo_type="model", token=tok)
111
+ return "deleted"
112
+
113
+
114
+ guard("cleanup_delete", cleanup)
115
+
116
+ R["staging_dir"] = {"path": "/tmp/.ounce100m",
117
+ "exists": os.path.isdir("/tmp/.ounce100m"),
118
+ "mode": oct(os.stat("/tmp/.ounce100m").st_mode & 0o777)
119
+ if os.path.isdir("/tmp/.ounce100m") else None}
120
+
121
+ print("PROBE_JSON_BEGIN")
122
+ print(json.dumps(R, indent=1, default=str))
123
+ print("PROBE_JSON_END")