# Phase 2 blocker check: can this token write DATASETS, or only models? # # Why this exists. `p1-credential-roundtrip` proved a job can create and delete a **model** repo. The # cold-resume rehearsal then failed to create a **dataset** repo with `401 Unauthorized` from # POST /api/repos/create, using the same token and the same code path. Phase 2 requires publishing the mix # as a public `ounce100m-*` *dataset*, and the stage repo is a dataset too -- so if the hypothesis is # right, this is a hard blocker on Gate 2 that only the user can clear by adjusting the token. # # The token is fine-grained (`repo.write` scoped to the user). HF splits "write to models" and "write to # datasets" into separate scopes, and `repo.write` has historically meant the former, so the hypothesis is # plausible rather than a mystery. # # Test design: for each repo type, create → upload → read back anonymously → delete, and report the # outcome per step. Anything created here is deleted in the same run; nothing lingers in the account. # It also dumps the token's own declared capabilities, which is the authoritative answer rather than an # inference from a 401. # # CPU session, zero GPU quota. No credentials in this file (D-006 helper). import json import os import urllib.error import urllib.request import ounce100m_credentials R = {} STAMP = "scopetest" NAMES = {"model": f"Cion-lab/ounce100m-{STAMP}-model-DELETEME", "dataset": f"Cion-lab/ounce100m-{STAMP}-dataset-DELETEME"} def guard(name, fn): try: R[name] = fn() except Exception as e: detail = "" resp = getattr(e, "response", None) if resp is not None: try: detail = resp.text[:400] except Exception: pass R[name] = {"error": f"{type(e).__name__}: {str(e)[:280]}", "body": detail} guard("install", lambda: ounce100m_credentials.install(verify=True)) TOKEN = os.environ.get("HF_TOKEN", "") def whoami_full(): req = urllib.request.Request("https://huggingface.co/whoami-v2", headers={"Authorization": f"Bearer {TOKEN}"}) with urllib.request.urlopen(req, timeout=45) as r: d = json.loads(r.read().decode()) # report capability *shape*, never the token itself out = {"name": d.get("name"), "type": d.get("type")} full = d.get("full") or {} out["full_keys"] = sorted(full.keys()) auth = full.get("auth") or d.get("auth") or {} out["auth"] = {k: v for k, v in auth.items() if k != "accessToken"} for k in ("capabilities", "scopes", "permissions"): if k in full: out[k] = full[k] elif k in d: out[k] = d[k] if "accessToken" in full: at = full["accessToken"] out["accessToken_type"] = type(at).__name__ if isinstance(at, dict): out["accessToken_scopes"] = at.get("scopes") or at.get("permission") return out guard("whoami_v2", whoami_full) def trial(kind): from huggingface_hub import HfApi, create_repo, delete_repo repo = NAMES[kind] api = HfApi(token=TOKEN) steps = {} create_repo(repo_id=repo, repo_type=kind, private=False, exist_ok=True) steps["create"] = "ok" payload = f"ounce100m scope probe {kind}\n".encode() api.upload_file(path_or_fileobj=payload, path_in_repo="probe.txt", repo_id=repo, repo_type=kind, commit_message="scope probe") steps["upload"] = "ok" url = (f"https://huggingface.co/{repo}/resolve/main/probe.txt" if kind == "model" else f"https://huggingface.co/datasets/{repo}/resolve/main/probe.txt") try: with urllib.request.urlopen(url, timeout=60) as r: steps["anonymous_readback_matches"] = r.read().strip() == payload.strip() except urllib.error.HTTPError as e: steps["anonymous_readback"] = f"HTTP {e.code}" delete_repo(repo_id=repo, repo_type=kind) steps["delete"] = "ok" return steps for kind in ("model", "dataset"): guard(f"trial_{kind}", lambda k=kind: trial(k)) # confirm the deletes really took, so nothing is left behind in the account def listing(): out = {} for kind, q in (("model", "models"), ("dataset", "datasets")): try: with urllib.request.urlopen( f"https://huggingface.co/api/{q}?author=Cion-lab", timeout=45) as r: out[kind] = sorted(m["id"] for m in json.loads(r.read().decode())) except Exception as e: out[kind] = f"{type(e).__name__}" return out guard("namespace_after", listing) R["verdict"] = { "model_write": "ok" if isinstance(R.get("trial_model"), dict) and R["trial_model"].get("create") == "ok" else R.get("trial_model"), "dataset_write": "ok" if isinstance(R.get("trial_dataset"), dict) and R["trial_dataset"].get("create") == "ok" else R.get("trial_dataset"), } R["BLOCKS_GATE_2"] = R["verdict"]["dataset_write"] != "ok" print("SCOPE_JSON_BEGIN") print(json.dumps(R, indent=1, default=str)[:5000]) print("SCOPE_JSON_END")