RL-Explorer / app /mimo /runner /opencode.py
AdithyaSK's picture
AdithyaSK HF Staff
Deploy HF RL Explorer
6abb7be verified
Raw History Blame Contribute Delete
13.5 kB
"""The agent harness: OpenCode, run headless inside the task's sandbox.
`opencode run --format json` prints one JSON event per line. Those stream back through
`Sandbox.run(on_stdout=...)` and become the rollout's own events as they arrive.
Configured the way mimoagent's OpenCode adapter runs it (agents/blackbox/opencode.py): the injected
config is the only config (a repo's own opencode.json is ignored), no auto-update, no share, no LSP
download, no title-generation call, OpenCode's own state kept outside the workspace, and the output
ceiling raised past 32k tokens when asked.
Where this differs, and why: Xiaomi's pods had no route to the internet, so their agents could not look
answers up. An HF Sandbox does, so web search/fetch tools are off and the hosts where answers live (code
hosting, bug trackers, source-serving package proxies, search engines) are blackholed in /etc/hosts, the
mechanism mimoagent itself uses (``answer_leak_blocklist``). Everything else, including curl to local
services, works as usual. (A first test run on a Code task found the upstream fix with ``webfetch``.)
"""
from __future__ import annotations
import json
import shlex
from .. import config
VERSION = "1.18.32" # pinned: a rollout should not change because OpenCode shipped overnight
# The release binary, fetched with whatever the image has (some task images have no curl), and the
# same build variants the official installer picks: -baseline without AVX2, -musl on musl libc.
INSTALL = r"""command -v opencode >/dev/null 2>&1 || {
t=linux-x64; grep -qw avx2 /proc/cpuinfo || t=$t-baseline; (ldd --version 2>&1 | grep -qi musl) && t=$t-musl
u=https://github.com/anomalyco/opencode/releases/download/v%s/opencode-$t.tar.gz
{ curl -fsSL "$u" -o /tmp/.oc.tgz || wget -qO /tmp/.oc.tgz "$u" ||
python3 -c "import sys,urllib.request; urllib.request.urlretrieve(sys.argv[1], '/tmp/.oc.tgz')" "$u"; } &&
tar -xzf /tmp/.oc.tgz -C /tmp opencode && install -m 755 /tmp/opencode /usr/local/bin/opencode && rm -f /tmp/.oc.tgz /tmp/opencode
}; opencode --version""" % VERSION
# Blackholed for the agent (mimoagent's answer_leak_blocklist mechanism). The model endpoint and HF are untouched.
ANSWER_HOSTS = [
"github.com", "www.github.com", "api.github.com", "raw.githubusercontent.com", "codeload.github.com",
"objects.githubusercontent.com", "gist.github.com", "gist.githubusercontent.com", "gitlab.com", "bitbucket.org",
"sourceforge.net", "git.kernel.org", "chromium.googlesource.com", "android.googlesource.com", "sourcegraph.com",
"bugs.chromium.org", "issues.oss-fuzz.com", "oss-fuzz.com", "osv.dev", "api.osv.dev", "bugzilla.mozilla.org",
"proxy.golang.org", "sum.golang.org", "pkg.go.dev", "google.com", "www.google.com", "bing.com", "www.bing.com",
"duckduckgo.com", "html.duckduckgo.com", "search.brave.com", "stackoverflow.com",
]
def block_answer_hosts(r) -> None:
lines = "\n".join(f"0.0.0.0 {h}" for h in ANSWER_HOSTS)
res = r.sh(f"printf '%b\\n' '# mimo-explorer answer-leak blocklist\\n{lines}' >> /etc/hosts && grep -c '^0.0.0.0' /etc/hosts")
if res.exit_code != 0: # fail closed, as mimoagent does
raise RuntimeError("could not install the answer-leak blocklist in /etc/hosts: " + (res.stderr or res.stdout or "")[-300:])
def install(r) -> str:
res = r.sh(INSTALL, timeout=300)
version = (res.stdout or "").strip().splitlines()[-1:] or ["?"]
if res.exit_code != 0:
raise RuntimeError("could not install OpenCode in the sandbox: " + (res.stderr or res.stdout or "")[-300:])
prov = r.run.get("provenance") or {}
if prov: # what actually ran, not only what was pinned
r.update(provenance={**prov, "harness": {**prov.get("harness", {}), "installed": version[0]}})
return version[0]
# Per-reply output ceiling. mimoagent's example_configs/opencode.yaml runs OpenCode with output_limit 65536, unlocked
# past OpenCode's own 32000 through OPENCODE_EXPERIMENTAL_OUTPUT_TOKEN_MAX. At 32000 a model that thinks at length
# can spend the whole reply thinking, get cut off before its first tool call, and end the session with nothing done.
OUTPUT_LIMIT = 65536
def output_limit(params: dict) -> int:
return params.get("max_tokens") or params.get("output_limit") or OUTPUT_LIMIT
def config_for(model: str, provider: str | None, steps: int, mcp: list[dict] | None = None,
endpoint: dict | None = None, params: dict | None = None, proxy: str | None = None) -> dict:
"""With `proxy` (the Space), every model call goes to this server's per-rollout proxy and the sandbox holds no
credential at all; the capability in the URL is the only secret, and it dies with the rollout."""
params = params or {}
if endpoint: # your own OpenAI-compatible endpoint; without the proxy, the key arrives as AGENT_API_KEY
mid, prov = model, {"byo": {"npm": "@ai-sdk/openai-compatible", "name": "Your endpoint",
"options": {"baseURL": proxy or endpoint["base_url"],
"apiKey": "proxied" if proxy else "{env:AGENT_API_KEY}"},
"models": {model: {"name": model, "tool_call": True}}}}
ref = f"byo/{model}"
else:
mid = f"{model}:{provider}" if provider else model
prov = {"hf": {"npm": "@ai-sdk/openai-compatible", "name": "Hugging Face Inference Providers",
"options": {"baseURL": proxy or config.ROUTER, "apiKey": "proxied" if proxy else "{env:HF_TOKEN}"},
"models": {mid: {"name": model, "tool_call": True}}}}
ref = f"hf/{mid}"
entry = next(iter(next(iter(prov.values()))["models"].values()))
if params.get("thinking") and params["thinking"] != "default":
entry["options"] = {"reasoningEffort": params["thinking"]} # sent as reasoning_effort ("none" turns it off)
entry["limit"] = {"context": 1_000_000, "output": output_limit(params)}
build = {"steps": steps}
if params.get("temperature") is not None:
build["temperature"] = params["temperature"]
return {
"$schema": "https://opencode.ai/config.json", "autoupdate": False, "share": "disabled",
"model": ref,
"provider": prov,
"agent": {"build": build},
"mcp": {s["name"]: {"type": "remote", "url": s["url"], "enabled": True} for s in (mcp or [])},
# mimoagent: "lsp/formatter are coding capability, and snapshot is opencode's own undo history"
"snapshot": True, "lsp": True, "formatter": True,
"permission": {"webfetch": "deny", "websearch": "deny", "bash": "allow", "edit": "allow"},
}
def run(r, prompt: str, cwd: str, *, steps: int, timeout: float, user: str | None = None,
home: str | None = None, mcp: list[dict] | None = None) -> str:
"""Run the agent to completion. Returns its final message."""
# the exact user message the agent starts from; OpenCode adds its own system prompt and tool definitions
r.emit("prompt", text=prompt, harness="opencode")
if r.run.get("scripted") is not None: # validation only (app/validate.py): a fixed probe instead of a model
return _scripted(r, cwd, user)
ep, params = r.run.get("endpoint"), r.run.get("params") or {}
steps = params.get("steps") or steps
timeout = params["timeout_min"] * 60 if params.get("timeout_min") else timeout
proxy = r.llm_base()
r.check_model_network()
cfg = config_for(r.run["model"], r.run.get("provider"), steps, mcp, ep, params, proxy)
r.sandbox.files.write("/tmp/opencode.json", json.dumps(cfg))
r.sandbox.files.write("/tmp/prompt.txt", prompt)
r.sh("chmod 644 /tmp/opencode.json /tmp/prompt.txt")
key_env = ""
if ep and not proxy:
# a root-only file, read by the root shell when it builds the command: the key is never on a command line
r.sandbox.files.write("/root/.agent_key", r.agent_key or "")
r.sh("chmod 600 /root/.agent_key")
key_env = 'AGENT_API_KEY="$(cat /root/.agent_key)" '
xdg = "/tmp/.opencode-state" # OpenCode's own sessions/caches: outside the workspace, which is graded state
r.sh(f"mkdir -p {xdg} && chmod 777 {xdg}")
flags = {"OPENCODE_CONFIG": "/tmp/opencode.json", "OPENCODE_DISABLE_PROJECT_CONFIG": "1",
"OPENCODE_DISABLE_MODELS_FETCH": "1", "OPENCODE_DISABLE_AUTOUPDATE": "1", "OPENCODE_DISABLE_LSP_DOWNLOAD": "1",
"OPENCODE_DISABLE_SHARE": "1", "XDG_DATA_HOME": f"{xdg}/data", "XDG_CONFIG_HOME": f"{xdg}/config",
"XDG_CACHE_HOME": f"{xdg}/cache", "XDG_STATE_HOME": f"{xdg}/state"}
if output_limit(params) > 32000: # upstream silently caps output at 32k otherwise
flags["OPENCODE_EXPERIMENTAL_OUTPUT_TOKEN_MAX"] = str(output_limit(params))
token_env = "" if proxy else 'HF_TOKEN="$HF_TOKEN" '
env = f"env {key_env}{token_env}HOME={home or '$HOME'} " + " ".join(f"{k}={v}" for k, v in flags.items())
# --thinking streams the model's reasoning as its own events, so the trace shows why, not only what;
# --title skips the extra model call that would otherwise name the session
inner = f'{env} opencode run --format json --thinking --auto --title mimo-rollout "$(cat /tmp/prompt.txt)"'
cmd = f"cd {shlex.quote(cwd)} && " + (f"runuser -u {user} -- {inner}" if user else inner)
buf = [""]
texts: list[str] = []
def on_out(chunk: str) -> None:
buf[0] += chunk
*lines, buf[0] = buf[0].split("\n")
for line in lines:
if line.strip():
_event(r, line, texts)
def on_err(chunk: str) -> None:
s = chunk.strip()
if s and "DeprecationWarning" not in s:
r.log(s[-1500:])
try:
res = r.sh(cmd, timeout=timeout, on_stdout=on_out, on_stderr=on_err)
finally:
if ep and not proxy:
r.sandbox.run("rm -f /root/.agent_key", shell=True, check=False, timeout=30)
if buf[0].strip():
_event(r, buf[0], texts)
if getattr(r, "agent_failure", None):
raise RuntimeError("The agent's model call failed; this run is not graded. " + r.agent_failure[:500])
if res.exit_code and not res.timed_out:
raise RuntimeError(f"The agent exited with code {res.exit_code}; this run is not graded.")
if res.timed_out:
r.emit("error", text=f"The agent hit the {int(timeout // 60)}-minute limit; grading what it left.")
return texts[-1] if texts else ""
def _event(r, line: str, texts: list[str]) -> None:
try:
ev = json.loads(line)
except ValueError:
r.log(line[:1500])
return
typ, part = ev.get("type"), ev.get("part") or {}
if typ == "text" and part.get("text"):
texts.append(part["text"])
r.emit("text", text=part["text"])
elif typ == "reasoning" and (part.get("text") or "").strip():
t = part["text"].strip()
r.emit("thinking", text=t[:20000], chars=len(t))
elif typ == "tool_use":
st = part.get("state") or {}
tm = st.get("time") or {}
out = st.get("output")
if not isinstance(out, str):
out = json.dumps(out, ensure_ascii=False) if out is not None else ""
r.emit("tool", tool=part.get("tool"), title=st.get("title") or "", input=st.get("input"),
output=out[:8000], truncated=len(out) > 8000, status=st.get("status"),
ms=(tm.get("end") - tm.get("start")) if tm.get("end") and tm.get("start") else None)
elif typ == "step_finish":
t = part.get("tokens") or {}
cache = t.get("cache") or {}
tok = {"input": t.get("input"), "output": t.get("output"), "reasoning": t.get("reasoning"),
"cache_read": cache.get("read"), "cache_write": cache.get("write")}
r.add_tokens(tok)
reason = part.get("reason")
r.emit("step", tokens=tok, cost=r.cost_now()["model"], reason=reason)
limit = output_limit(r.run.get("params") or {})
if reason == "length" or (tok["output"] or 0) + (tok["reasoning"] or 0) >= limit * 0.99:
r.update(truncated=True)
r.emit("error", text=f"The model's reply hit the {limit:,}-token output limit before it finished "
f"({tok['reasoning'] or 0:,} tokens of it thinking). A cut-off reply has no tool call, "
"so OpenCode ends the session there, as it does in Xiaomi's harness.")
elif typ == "error":
err = ev.get("error") or part
r.agent_failure = ((err.get("data") or {}).get("message") or json.dumps(err)[:800]) if isinstance(err, dict) else str(err)[:800]
r.emit("error", text=r.agent_failure)
def _scripted(r, cwd: str, user: str | None) -> str:
"""Stand-in for the model in health checks: run a probe as the agent user, in the agent's directory."""
probe = r.run["scripted"]
if probe:
r.sandbox.files.write("/tmp/.probe.sh", probe)
r.sh("chmod 755 /tmp/.probe.sh")
cmd = f"cd {shlex.quote(cwd)} && " + (f"runuser -u {user} -- bash /tmp/.probe.sh" if user else "bash /tmp/.probe.sh")
res = r.sh(cmd, timeout=600)
r.emit("tool", tool="bash", title="probe", input={"command": probe[:2000]}, output=((res.stdout or "") + (res.stderr or ""))[:8000],
truncated=False, status="completed" if res.exit_code == 0 else "error", ms=None)
final = r.run.get("scripted_final", "")
if final:
r.emit("text", text=final)
return final