File size: 18,269 Bytes
3b09ba9 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292 293 294 295 296 297 298 299 300 301 302 303 304 305 306 307 308 309 310 311 312 313 314 315 316 317 318 319 320 321 322 323 324 325 326 327 328 329 330 331 332 333 334 335 336 337 338 339 340 341 342 343 344 345 346 347 348 349 350 351 352 353 354 355 356 357 358 359 360 361 362 363 364 365 366 367 368 369 370 371 372 373 374 375 376 377 378 379 380 381 382 383 384 385 386 387 388 389 390 391 392 393 394 395 396 397 398 399 400 401 402 403 404 405 406 407 | """
Tool implementations for the headless grounding agent.
These mirror, as closely as practical, the interactive tools a Claude Code
session already uses to build a case by hand: a shell (grep/find/pdftotext),
a way to *see* PDF pages as images (the multimodal equivalent of Read with
`pages`), a way to read source files verbatim, a way to produce a real
deterministic crop of a figure, and a final "submit" call that ends the
agent loop with the completed narrative.json.
No network calls happen here except what the model's own bash commands
make (none should, for this task) -- everything is local subprocess/file
I/O against the case directory.
"""
from __future__ import annotations
import base64
import io
import json
import subprocess
from pathlib import Path
from typing import Any
from PIL import Image
MAX_BASH_OUTPUT = 12_000 # chars; keep tool results bounded like a real terminal would be
BASH_TIMEOUT_S = 30
PDF_RENDER_DPI = 200
class ToolError(Exception):
pass
def _truncate(s: str, limit: int = MAX_BASH_OUTPUT) -> str:
if len(s) <= limit:
return s
return s[:limit] + f"\n...[truncated, {len(s) - limit} more chars]"
class CaseWorkspace:
"""All tool calls for one case are scoped to this directory."""
def __init__(self, case_dir: Path):
self.case_dir = case_dir.resolve()
self.repo_dir = self.case_dir / "repo"
self.pdf_path = self.case_dir / "paper.pdf"
if not self.pdf_path.is_file():
raise ToolError(f"paper.pdf not found at {self.pdf_path}")
if not self.repo_dir.is_dir():
raise ToolError(f"repo/ not found at {self.repo_dir}")
self.submitted_narrative: dict | None = None
self._crops: dict[str, Image.Image] = {}
# ---------- bash ----------
def bash(self, command: str) -> str:
"""Run a shell command with cwd = the case directory (so `repo/...`
and `paper.pdf` are directly reachable). Meant for grep/find/ls/
pdftotext -- not for arbitrary system administration."""
try:
proc = subprocess.run(
command,
shell=True,
cwd=self.case_dir,
capture_output=True,
text=True,
timeout=BASH_TIMEOUT_S,
)
except subprocess.TimeoutExpired:
return f"[error] command timed out after {BASH_TIMEOUT_S}s"
out = proc.stdout or ""
err = proc.stderr or ""
combined = out
if err:
combined += ("\n" if combined else "") + f"[stderr]\n{err}"
if proc.returncode != 0:
combined += f"\n[exit code {proc.returncode}]"
return _truncate(combined) if combined.strip() else "[no output]"
# ---------- read_pdf_page ----------
def read_pdf_page(self, page: int) -> list[dict[str, Any]]:
"""Render one 1-indexed PDF page to an image and return it as a
multimodal content block, the same way the interactive Read tool's
`pages` parameter works. This is the ONLY way this agent ever sees
the paper -- there is no OCR/text-extraction shortcut for the
Method section, matching the existing protocol's discipline."""
with subprocess_tmpdir() as tmpdir:
prefix = tmpdir / "page"
proc = subprocess.run(
[
"pdftoppm", "-png", "-r", str(PDF_RENDER_DPI),
"-f", str(page), "-l", str(page),
str(self.pdf_path), str(prefix),
],
capture_output=True, text=True, timeout=60,
)
if proc.returncode != 0:
raise ToolError(f"pdftoppm failed: {proc.stderr}")
matches = sorted(tmpdir.glob("page-*.png"))
if not matches:
raise ToolError(f"page {page} did not render (out of range?)")
img = Image.open(matches[0]).convert("RGB")
# Cap width for token economy -- this is a reading aid for the
# model, not the final artifact, so 1400px is plenty of detail.
if img.width > 1400:
ratio = 1400 / img.width
img = img.resize((1400, int(img.height * ratio)), Image.LANCZOS)
buf = io.BytesIO()
img.save(buf, format="PNG")
b64 = base64.b64encode(buf.getvalue()).decode("ascii")
return [
{"type": "text", "text": f"paper.pdf, page {page}:"},
{"type": "image", "source": {"type": "base64", "media_type": "image/png", "data": b64}},
]
# ---------- read_file ----------
def read_file(self, path: str, offset: int = 1, limit: int = 400) -> str:
"""Read a text file relative to the case directory (usually
something under repo/), 1-indexed line offset, verbatim -- the
source of truth for every code_ref snippet."""
target = (self.case_dir / path).resolve()
if self.case_dir not in target.parents and target != self.case_dir:
raise ToolError("path escapes the case directory")
if not target.is_file():
raise ToolError(f"no such file: {path}")
lines = target.read_text(errors="replace").splitlines()
start = max(1, offset)
end = min(len(lines), start + limit - 1)
numbered = "\n".join(f"{i:>6}\t{lines[i - 1]}" for i in range(start, end + 1))
return numbered or "[empty range]"
# ---------- crop_figure ----------
def crop_figure(self, page: int, left: int, top: int, right: int, bottom: int) -> list[dict[str, Any]]:
"""Deterministic crop of a figure's own artwork from a specific
page, at PDF_RENDER_DPI, in pixel coordinates -- exactly the
render->crop->verify process CLAUDE.md describes for figure_refs
and case.overview.figure. Returns the crop for the model to look
at AND a crop_id string. The model never sees or handles the
actual image bytes as text -- to use a confirmed crop, it writes
the literal placeholder string "CROP:<crop_id>" into the
narrative's `image` field; the orchestrator substitutes the real
base64 data-uri after submit_narrative, so a multi-KB image never
has to be retyped as output tokens (which is slow, expensive, and
was observed to get truncated mid-string against max_tokens)."""
with subprocess_tmpdir() as tmpdir:
prefix = tmpdir / "page"
proc = subprocess.run(
[
"pdftoppm", "-png", "-r", str(PDF_RENDER_DPI),
"-f", str(page), "-l", str(page),
str(self.pdf_path), str(prefix),
],
capture_output=True, text=True, timeout=60,
)
if proc.returncode != 0:
raise ToolError(f"pdftoppm failed: {proc.stderr}")
matches = sorted(tmpdir.glob("page-*.png"))
if not matches:
raise ToolError(f"page {page} did not render (out of range?)")
img = Image.open(matches[0]).convert("RGB")
crop = img.crop((left, top, right, bottom))
if crop.width < 10 or crop.height < 10:
raise ToolError("crop box is degenerate (too small) -- check coordinates")
buf = io.BytesIO()
crop.save(buf, format="PNG")
b64 = base64.b64encode(buf.getvalue()).decode("ascii")
crop_id = f"crop_{len(self._crops) + 1}"
self._crops[crop_id] = crop
return [
{"type": "text", "text": (
f"crop_id={crop_id} (page {page}, box=({left},{top},{right},{bottom})). "
f"If this looks correct, reference it in submit_narrative as the "
f"literal string \"CROP:{crop_id}\" -- do not copy any image data yourself."
)},
{"type": "image", "source": {"type": "base64", "media_type": "image/png", "data": b64}},
]
def resolve_crop_placeholders(self, narrative: dict, max_width: int = 900, quality: int = 88) -> dict:
"""Walk the submitted narrative and replace any string field whose
value is exactly "CROP:<crop_id>" with the real compressed
data:image/jpeg;base64,... URI for that crop. This is the only
place base64 image text is ever produced -- deterministically, by
this code, never generated by the model."""
crops: dict[str, Image.Image] = getattr(self, "_crops", {})
def uri_for(crop_id: str) -> str:
if crop_id not in crops:
raise ToolError(
f"submit_narrative referenced CROP:{crop_id}, but no such crop "
f"was produced via crop_figure this session (have: {list(crops)})"
)
img = crops[crop_id].convert("RGB")
if img.width > max_width:
ratio = max_width / img.width
img = img.resize((max_width, int(img.height * ratio)), Image.LANCZOS)
buf = io.BytesIO()
img.save(buf, format="JPEG", quality=quality, optimize=True)
b64 = base64.b64encode(buf.getvalue()).decode("ascii")
return f"data:image/jpeg;base64,{b64}"
def walk(node):
if isinstance(node, dict):
return {k: walk(v) for k, v in node.items()}
if isinstance(node, list):
return [walk(v) for v in node]
if isinstance(node, str) and node.startswith("CROP:"):
return uri_for(node[len("CROP:"):])
return node
return walk(narrative)
def resolve_code_snippets(self, narrative: dict) -> dict:
"""Overwrite every code_ref's `snippet` with the exact verbatim text
read from disk at file:start_line-end_line, ignoring whatever the
model typed there. Mirrors scripts/verify_case.py's own extraction
exactly, so a resolved snippet always passes that check by
construction -- this is the same fix as CROP: placeholders, applied
to code: never trust the model to retype content character-for-
character when it can instead be read mechanically. `file` is
interpreted relative to case.repo_path (default 'repo')."""
repo_rel = ((narrative.get("case") or {}).get("repo_path") or "repo")
repo_root = (self.case_dir / repo_rel).resolve()
def resolve_ref(ref: dict) -> dict:
file_rel = ref.get("file")
start, end = ref.get("start_line"), ref.get("end_line")
if not file_rel or not isinstance(start, int) or not isinstance(end, int):
return ref
target = (repo_root / file_rel).resolve()
if not target.is_file():
# Same tolerance as verify_case.py's check_code_ref: accept
# a file path that's case-dir-relative (includes the repo
# dir's own name) instead of repo_path-relative.
fallback = (self.case_dir / file_rel).resolve()
target = fallback if fallback.is_file() else target
if self.case_dir not in target.parents and target != self.case_dir:
raise ToolError(f"code_ref file escapes the case directory: {file_rel}")
if not target.is_file():
return ref # let verify_case.py report the missing-file error
lines = target.read_text(errors="replace").splitlines()
if end > len(lines) or start < 1:
return ref # let verify_case.py report the bad-range error
return {**ref, "snippet": "\n".join(lines[start - 1:end])}
def walk(node):
if isinstance(node, dict):
if "code_refs" in node and isinstance(node["code_refs"], list):
node = {**node, "code_refs": [
resolve_ref(r) if isinstance(r, dict) else r for r in node["code_refs"]
]}
return {k: (walk(v) if k != "code_refs" else v) for k, v in node.items()}
if isinstance(node, list):
return [walk(v) for v in node]
return node
return walk(narrative)
# ---------- submit_narrative ----------
def submit_narrative(self, narrative: dict) -> str:
self.submitted_narrative = narrative
return "received."
import contextlib
import shutil
import tempfile
@contextlib.contextmanager
def subprocess_tmpdir():
d = Path(tempfile.mkdtemp(prefix="pcd_"))
try:
yield d
finally:
shutil.rmtree(d, ignore_errors=True)
# ---------- Anthropic tool schemas ----------
TOOL_SCHEMAS = [
{
"name": "bash",
"description": (
"Run a shell command with cwd set to this case's directory. "
"Use for grep/find/ls/pdftotext against repo/ and paper.pdf -- "
"the same commands you'd run interactively to search code or "
"pull the bibliography. Not a general-purpose shell; keep "
"commands read-only and scoped to this case."
),
"input_schema": {
"type": "object",
"required": ["command"],
"properties": {"command": {"type": "string"}},
},
},
{
"name": "read_pdf_page",
"description": (
"Render one 1-indexed page of paper.pdf as an image and view "
"it. This is the ONLY way to read the paper -- there is no "
"text extraction shortcut for the Method section, equations, "
"or figures. Call once per page you need to actually read."
),
"input_schema": {
"type": "object",
"required": ["page"],
"properties": {"page": {"type": "integer", "minimum": 1}},
},
},
{
"name": "read_file",
"description": (
"Read a text file relative to the case directory verbatim, "
"with 1-indexed line numbers (e.g. 'repo/model/mapfns.py'). "
"This is the only legitimate source for a code_ref snippet -- "
"never write a snippet you have not read this way."
),
"input_schema": {
"type": "object",
"required": ["path"],
"properties": {
"path": {"type": "string"},
"offset": {"type": "integer", "minimum": 1, "default": 1},
"limit": {"type": "integer", "minimum": 1, "default": 400},
},
},
},
{
"name": "crop_figure",
"description": (
"Render `page` at 200 DPI and crop pixel box "
"(left, top, right, bottom), then show you the result so you "
"can confirm it's the right figure, tightly framed, before "
"using it. Call this to check a crop; if it's wrong, call "
"again with adjusted coordinates. Only use the result once "
"you've actually looked at it and it shows the intended "
"figure, not cut off, not a neighboring figure's caption. "
"Returns a crop_id -- to actually use the crop, write the "
"literal string \"CROP:<crop_id>\" as the value of a "
"figure_refs[].image or case.overview.figure.image field in "
"submit_narrative. NEVER write out base64 image data yourself "
"-- always use this placeholder string instead; the real "
"bytes are substituted in automatically after you submit."
),
"input_schema": {
"type": "object",
"required": ["page", "left", "top", "right", "bottom"],
"properties": {
"page": {"type": "integer", "minimum": 1},
"left": {"type": "integer"},
"top": {"type": "integer"},
"right": {"type": "integer"},
"bottom": {"type": "integer"},
},
},
},
{
"name": "submit_narrative",
"description": (
"Submit the completed narrative.json for this case. Ends the "
"session -- only call this once, when the case is fully done "
"and you believe it would pass scripts/verify_case.py. Pass "
"the entire narrative object (case + nodes), schema-shaped. "
"For any figure image, set the field to the literal string "
"\"CROP:<crop_id>\" from a prior crop_figure call -- never "
"write actual base64 image data as part of this call; it is "
"slow, expensive, and has been observed to get truncated "
"against the output length limit. If you haven't called "
"crop_figure for a figure, omit `image` for it rather than "
"inventing one. For every code_ref, `snippet` is automatically "
"re-read from disk at file:start_line-end_line after you "
"submit and OVERWRITES whatever text you put there -- so don't "
"spend effort hand-copying it character-for-character (that's "
"slow and error-prone); a short placeholder is fine. What "
"actually has to be exactly right is `file`, `start_line`, and "
"`end_line` -- get the range tight around the real logic, not "
"off by a few lines and not spanning boilerplate."
),
"input_schema": {
"type": "object",
"required": ["narrative"],
"properties": {"narrative": {"type": "object"}},
},
},
]
def dispatch(ws: CaseWorkspace, name: str, tool_input: dict) -> Any:
if name == "bash":
return ws.bash(tool_input["command"])
if name == "read_pdf_page":
return ws.read_pdf_page(tool_input["page"])
if name == "read_file":
return ws.read_file(
tool_input["path"],
tool_input.get("offset", 1),
tool_input.get("limit", 400),
)
if name == "crop_figure":
return ws.crop_figure(
tool_input["page"], tool_input["left"], tool_input["top"],
tool_input["right"], tool_input["bottom"],
)
if name == "submit_narrative":
return ws.submit_narrative(tool_input["narrative"])
raise ToolError(f"unknown tool: {name}")
|