Spaces:
Running
Running
File size: 13,515 Bytes
da5cba1 6abb7be da5cba1 6abb7be da5cba1 6abb7be da5cba1 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 | """The agent harness: OpenCode, run headless inside the task's sandbox.
`opencode run --format json` prints one JSON event per line. Those stream back through
`Sandbox.run(on_stdout=...)` and become the rollout's own events as they arrive.
Configured the way mimoagent's OpenCode adapter runs it (agents/blackbox/opencode.py): the injected
config is the only config (a repo's own opencode.json is ignored), no auto-update, no share, no LSP
download, no title-generation call, OpenCode's own state kept outside the workspace, and the output
ceiling raised past 32k tokens when asked.
Where this differs, and why: Xiaomi's pods had no route to the internet, so their agents could not look
answers up. An HF Sandbox does, so web search/fetch tools are off and the hosts where answers live (code
hosting, bug trackers, source-serving package proxies, search engines) are blackholed in /etc/hosts, the
mechanism mimoagent itself uses (``answer_leak_blocklist``). Everything else, including curl to local
services, works as usual. (A first test run on a Code task found the upstream fix with ``webfetch``.)
"""
from __future__ import annotations
import json
import shlex
from .. import config
VERSION = "1.18.32" # pinned: a rollout should not change because OpenCode shipped overnight
# The release binary, fetched with whatever the image has (some task images have no curl), and the
# same build variants the official installer picks: -baseline without AVX2, -musl on musl libc.
INSTALL = r"""command -v opencode >/dev/null 2>&1 || {
t=linux-x64; grep -qw avx2 /proc/cpuinfo || t=$t-baseline; (ldd --version 2>&1 | grep -qi musl) && t=$t-musl
u=https://github.com/anomalyco/opencode/releases/download/v%s/opencode-$t.tar.gz
{ curl -fsSL "$u" -o /tmp/.oc.tgz || wget -qO /tmp/.oc.tgz "$u" ||
python3 -c "import sys,urllib.request; urllib.request.urlretrieve(sys.argv[1], '/tmp/.oc.tgz')" "$u"; } &&
tar -xzf /tmp/.oc.tgz -C /tmp opencode && install -m 755 /tmp/opencode /usr/local/bin/opencode && rm -f /tmp/.oc.tgz /tmp/opencode
}; opencode --version""" % VERSION
# Blackholed for the agent (mimoagent's answer_leak_blocklist mechanism). The model endpoint and HF are untouched.
ANSWER_HOSTS = [
"github.com", "www.github.com", "api.github.com", "raw.githubusercontent.com", "codeload.github.com",
"objects.githubusercontent.com", "gist.github.com", "gist.githubusercontent.com", "gitlab.com", "bitbucket.org",
"sourceforge.net", "git.kernel.org", "chromium.googlesource.com", "android.googlesource.com", "sourcegraph.com",
"bugs.chromium.org", "issues.oss-fuzz.com", "oss-fuzz.com", "osv.dev", "api.osv.dev", "bugzilla.mozilla.org",
"proxy.golang.org", "sum.golang.org", "pkg.go.dev", "google.com", "www.google.com", "bing.com", "www.bing.com",
"duckduckgo.com", "html.duckduckgo.com", "search.brave.com", "stackoverflow.com",
]
def block_answer_hosts(r) -> None:
lines = "\n".join(f"0.0.0.0 {h}" for h in ANSWER_HOSTS)
res = r.sh(f"printf '%b\\n' '# mimo-explorer answer-leak blocklist\\n{lines}' >> /etc/hosts && grep -c '^0.0.0.0' /etc/hosts")
if res.exit_code != 0: # fail closed, as mimoagent does
raise RuntimeError("could not install the answer-leak blocklist in /etc/hosts: " + (res.stderr or res.stdout or "")[-300:])
def install(r) -> str:
res = r.sh(INSTALL, timeout=300)
version = (res.stdout or "").strip().splitlines()[-1:] or ["?"]
if res.exit_code != 0:
raise RuntimeError("could not install OpenCode in the sandbox: " + (res.stderr or res.stdout or "")[-300:])
prov = r.run.get("provenance") or {}
if prov: # what actually ran, not only what was pinned
r.update(provenance={**prov, "harness": {**prov.get("harness", {}), "installed": version[0]}})
return version[0]
# Per-reply output ceiling. mimoagent's example_configs/opencode.yaml runs OpenCode with output_limit 65536, unlocked
# past OpenCode's own 32000 through OPENCODE_EXPERIMENTAL_OUTPUT_TOKEN_MAX. At 32000 a model that thinks at length
# can spend the whole reply thinking, get cut off before its first tool call, and end the session with nothing done.
OUTPUT_LIMIT = 65536
def output_limit(params: dict) -> int:
return params.get("max_tokens") or params.get("output_limit") or OUTPUT_LIMIT
def config_for(model: str, provider: str | None, steps: int, mcp: list[dict] | None = None,
endpoint: dict | None = None, params: dict | None = None, proxy: str | None = None) -> dict:
"""With `proxy` (the Space), every model call goes to this server's per-rollout proxy and the sandbox holds no
credential at all; the capability in the URL is the only secret, and it dies with the rollout."""
params = params or {}
if endpoint: # your own OpenAI-compatible endpoint; without the proxy, the key arrives as AGENT_API_KEY
mid, prov = model, {"byo": {"npm": "@ai-sdk/openai-compatible", "name": "Your endpoint",
"options": {"baseURL": proxy or endpoint["base_url"],
"apiKey": "proxied" if proxy else "{env:AGENT_API_KEY}"},
"models": {model: {"name": model, "tool_call": True}}}}
ref = f"byo/{model}"
else:
mid = f"{model}:{provider}" if provider else model
prov = {"hf": {"npm": "@ai-sdk/openai-compatible", "name": "Hugging Face Inference Providers",
"options": {"baseURL": proxy or config.ROUTER, "apiKey": "proxied" if proxy else "{env:HF_TOKEN}"},
"models": {mid: {"name": model, "tool_call": True}}}}
ref = f"hf/{mid}"
entry = next(iter(next(iter(prov.values()))["models"].values()))
if params.get("thinking") and params["thinking"] != "default":
entry["options"] = {"reasoningEffort": params["thinking"]} # sent as reasoning_effort ("none" turns it off)
entry["limit"] = {"context": 1_000_000, "output": output_limit(params)}
build = {"steps": steps}
if params.get("temperature") is not None:
build["temperature"] = params["temperature"]
return {
"$schema": "https://opencode.ai/config.json", "autoupdate": False, "share": "disabled",
"model": ref,
"provider": prov,
"agent": {"build": build},
"mcp": {s["name"]: {"type": "remote", "url": s["url"], "enabled": True} for s in (mcp or [])},
# mimoagent: "lsp/formatter are coding capability, and snapshot is opencode's own undo history"
"snapshot": True, "lsp": True, "formatter": True,
"permission": {"webfetch": "deny", "websearch": "deny", "bash": "allow", "edit": "allow"},
}
def run(r, prompt: str, cwd: str, *, steps: int, timeout: float, user: str | None = None,
home: str | None = None, mcp: list[dict] | None = None) -> str:
"""Run the agent to completion. Returns its final message."""
# the exact user message the agent starts from; OpenCode adds its own system prompt and tool definitions
r.emit("prompt", text=prompt, harness="opencode")
if r.run.get("scripted") is not None: # validation only (app/validate.py): a fixed probe instead of a model
return _scripted(r, cwd, user)
ep, params = r.run.get("endpoint"), r.run.get("params") or {}
steps = params.get("steps") or steps
timeout = params["timeout_min"] * 60 if params.get("timeout_min") else timeout
proxy = r.llm_base()
r.check_model_network()
cfg = config_for(r.run["model"], r.run.get("provider"), steps, mcp, ep, params, proxy)
r.sandbox.files.write("/tmp/opencode.json", json.dumps(cfg))
r.sandbox.files.write("/tmp/prompt.txt", prompt)
r.sh("chmod 644 /tmp/opencode.json /tmp/prompt.txt")
key_env = ""
if ep and not proxy:
# a root-only file, read by the root shell when it builds the command: the key is never on a command line
r.sandbox.files.write("/root/.agent_key", r.agent_key or "")
r.sh("chmod 600 /root/.agent_key")
key_env = 'AGENT_API_KEY="$(cat /root/.agent_key)" '
xdg = "/tmp/.opencode-state" # OpenCode's own sessions/caches: outside the workspace, which is graded state
r.sh(f"mkdir -p {xdg} && chmod 777 {xdg}")
flags = {"OPENCODE_CONFIG": "/tmp/opencode.json", "OPENCODE_DISABLE_PROJECT_CONFIG": "1",
"OPENCODE_DISABLE_MODELS_FETCH": "1", "OPENCODE_DISABLE_AUTOUPDATE": "1", "OPENCODE_DISABLE_LSP_DOWNLOAD": "1",
"OPENCODE_DISABLE_SHARE": "1", "XDG_DATA_HOME": f"{xdg}/data", "XDG_CONFIG_HOME": f"{xdg}/config",
"XDG_CACHE_HOME": f"{xdg}/cache", "XDG_STATE_HOME": f"{xdg}/state"}
if output_limit(params) > 32000: # upstream silently caps output at 32k otherwise
flags["OPENCODE_EXPERIMENTAL_OUTPUT_TOKEN_MAX"] = str(output_limit(params))
token_env = "" if proxy else 'HF_TOKEN="$HF_TOKEN" '
env = f"env {key_env}{token_env}HOME={home or '$HOME'} " + " ".join(f"{k}={v}" for k, v in flags.items())
# --thinking streams the model's reasoning as its own events, so the trace shows why, not only what;
# --title skips the extra model call that would otherwise name the session
inner = f'{env} opencode run --format json --thinking --auto --title mimo-rollout "$(cat /tmp/prompt.txt)"'
cmd = f"cd {shlex.quote(cwd)} && " + (f"runuser -u {user} -- {inner}" if user else inner)
buf = [""]
texts: list[str] = []
def on_out(chunk: str) -> None:
buf[0] += chunk
*lines, buf[0] = buf[0].split("\n")
for line in lines:
if line.strip():
_event(r, line, texts)
def on_err(chunk: str) -> None:
s = chunk.strip()
if s and "DeprecationWarning" not in s:
r.log(s[-1500:])
try:
res = r.sh(cmd, timeout=timeout, on_stdout=on_out, on_stderr=on_err)
finally:
if ep and not proxy:
r.sandbox.run("rm -f /root/.agent_key", shell=True, check=False, timeout=30)
if buf[0].strip():
_event(r, buf[0], texts)
if getattr(r, "agent_failure", None):
raise RuntimeError("The agent's model call failed; this run is not graded. " + r.agent_failure[:500])
if res.exit_code and not res.timed_out:
raise RuntimeError(f"The agent exited with code {res.exit_code}; this run is not graded.")
if res.timed_out:
r.emit("error", text=f"The agent hit the {int(timeout // 60)}-minute limit; grading what it left.")
return texts[-1] if texts else ""
def _event(r, line: str, texts: list[str]) -> None:
try:
ev = json.loads(line)
except ValueError:
r.log(line[:1500])
return
typ, part = ev.get("type"), ev.get("part") or {}
if typ == "text" and part.get("text"):
texts.append(part["text"])
r.emit("text", text=part["text"])
elif typ == "reasoning" and (part.get("text") or "").strip():
t = part["text"].strip()
r.emit("thinking", text=t[:20000], chars=len(t))
elif typ == "tool_use":
st = part.get("state") or {}
tm = st.get("time") or {}
out = st.get("output")
if not isinstance(out, str):
out = json.dumps(out, ensure_ascii=False) if out is not None else ""
r.emit("tool", tool=part.get("tool"), title=st.get("title") or "", input=st.get("input"),
output=out[:8000], truncated=len(out) > 8000, status=st.get("status"),
ms=(tm.get("end") - tm.get("start")) if tm.get("end") and tm.get("start") else None)
elif typ == "step_finish":
t = part.get("tokens") or {}
cache = t.get("cache") or {}
tok = {"input": t.get("input"), "output": t.get("output"), "reasoning": t.get("reasoning"),
"cache_read": cache.get("read"), "cache_write": cache.get("write")}
r.add_tokens(tok)
reason = part.get("reason")
r.emit("step", tokens=tok, cost=r.cost_now()["model"], reason=reason)
limit = output_limit(r.run.get("params") or {})
if reason == "length" or (tok["output"] or 0) + (tok["reasoning"] or 0) >= limit * 0.99:
r.update(truncated=True)
r.emit("error", text=f"The model's reply hit the {limit:,}-token output limit before it finished "
f"({tok['reasoning'] or 0:,} tokens of it thinking). A cut-off reply has no tool call, "
"so OpenCode ends the session there, as it does in Xiaomi's harness.")
elif typ == "error":
err = ev.get("error") or part
r.agent_failure = ((err.get("data") or {}).get("message") or json.dumps(err)[:800]) if isinstance(err, dict) else str(err)[:800]
r.emit("error", text=r.agent_failure)
def _scripted(r, cwd: str, user: str | None) -> str:
"""Stand-in for the model in health checks: run a probe as the agent user, in the agent's directory."""
probe = r.run["scripted"]
if probe:
r.sandbox.files.write("/tmp/.probe.sh", probe)
r.sh("chmod 755 /tmp/.probe.sh")
cmd = f"cd {shlex.quote(cwd)} && " + (f"runuser -u {user} -- bash /tmp/.probe.sh" if user else "bash /tmp/.probe.sh")
res = r.sh(cmd, timeout=600)
r.emit("tool", tool="bash", title="probe", input={"command": probe[:2000]}, output=((res.stdout or "") + (res.stderr or ""))[:8000],
truncated=False, status="completed" if res.exit_code == 0 else "error", ms=None)
final = r.run.get("scripted_final", "")
if final:
r.emit("text", text=final)
return final
|