Spaces:
Running
Running
File size: 1,428 Bytes
78b04f0 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 | """What the harness actually sends to the model in this bench."""
from __future__ import annotations
IMAGE_HARNESSES = frozenset({"qwen3-vl", "uitars", "jedi", "opencua", "evocua", "fara"})
TEXT_HARNESSES = frozenset({"openhands"})
TEXT_IMAGE_HARNESSES = frozenset(
{
"browser-use",
"openmanus",
"ouroboros-cut",
"ouroboros-full-isolated",
"ouroboros-full-evolving",
}
)
TEXT_ONLY_MODELS = frozenset({"z-ai/glm-5.2", "minimax/minimax-m2.7"})
# browser-use / openmanus in this bench drop vision for DeepSeek — LLM sees DOM only.
DOM_TEXT_EVEN_IF_VISION_HARNESS = frozenset({"deepseek/deepseek-v4.1-flash"})
INPUT_HELP = (
"Harness input: text (DOM/text only), text+image (page text plus screenshots), "
"or image (GUI screenshot, no DOM). DeepSeek on browser-use/openmanus is DOM-only. "
"Text-only models on Ouroboros do not get a screenshot injected into the LLM."
)
def harness_input(model: str | None, harness: str | None) -> str:
if not harness:
return "—"
if harness in IMAGE_HARNESSES:
return "image"
if harness in TEXT_HARNESSES:
return "text"
if harness in TEXT_IMAGE_HARNESSES:
if model in TEXT_ONLY_MODELS:
return "text"
if model in DOM_TEXT_EVEN_IF_VISION_HARNESS and harness in {"browser-use", "openmanus"}:
return "text"
return "text+image"
return "—"
|