File size: 18,269 Bytes
3b09ba9
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
"""
Tool implementations for the headless grounding agent.

These mirror, as closely as practical, the interactive tools a Claude Code
session already uses to build a case by hand: a shell (grep/find/pdftotext),
a way to *see* PDF pages as images (the multimodal equivalent of Read with
`pages`), a way to read source files verbatim, a way to produce a real
deterministic crop of a figure, and a final "submit" call that ends the
agent loop with the completed narrative.json.

No network calls happen here except what the model's own bash commands
make (none should, for this task) -- everything is local subprocess/file
I/O against the case directory.
"""
from __future__ import annotations

import base64
import io
import json
import subprocess
from pathlib import Path
from typing import Any

from PIL import Image

MAX_BASH_OUTPUT = 12_000  # chars; keep tool results bounded like a real terminal would be
BASH_TIMEOUT_S = 30
PDF_RENDER_DPI = 200


class ToolError(Exception):
    pass


def _truncate(s: str, limit: int = MAX_BASH_OUTPUT) -> str:
    if len(s) <= limit:
        return s
    return s[:limit] + f"\n...[truncated, {len(s) - limit} more chars]"


class CaseWorkspace:
    """All tool calls for one case are scoped to this directory."""

    def __init__(self, case_dir: Path):
        self.case_dir = case_dir.resolve()
        self.repo_dir = self.case_dir / "repo"
        self.pdf_path = self.case_dir / "paper.pdf"
        if not self.pdf_path.is_file():
            raise ToolError(f"paper.pdf not found at {self.pdf_path}")
        if not self.repo_dir.is_dir():
            raise ToolError(f"repo/ not found at {self.repo_dir}")
        self.submitted_narrative: dict | None = None
        self._crops: dict[str, Image.Image] = {}

    # ---------- bash ----------
    def bash(self, command: str) -> str:
        """Run a shell command with cwd = the case directory (so `repo/...`
        and `paper.pdf` are directly reachable). Meant for grep/find/ls/
        pdftotext -- not for arbitrary system administration."""
        try:
            proc = subprocess.run(
                command,
                shell=True,
                cwd=self.case_dir,
                capture_output=True,
                text=True,
                timeout=BASH_TIMEOUT_S,
            )
        except subprocess.TimeoutExpired:
            return f"[error] command timed out after {BASH_TIMEOUT_S}s"
        out = proc.stdout or ""
        err = proc.stderr or ""
        combined = out
        if err:
            combined += ("\n" if combined else "") + f"[stderr]\n{err}"
        if proc.returncode != 0:
            combined += f"\n[exit code {proc.returncode}]"
        return _truncate(combined) if combined.strip() else "[no output]"

    # ---------- read_pdf_page ----------
    def read_pdf_page(self, page: int) -> list[dict[str, Any]]:
        """Render one 1-indexed PDF page to an image and return it as a
        multimodal content block, the same way the interactive Read tool's
        `pages` parameter works. This is the ONLY way this agent ever sees
        the paper -- there is no OCR/text-extraction shortcut for the
        Method section, matching the existing protocol's discipline."""
        with subprocess_tmpdir() as tmpdir:
            prefix = tmpdir / "page"
            proc = subprocess.run(
                [
                    "pdftoppm", "-png", "-r", str(PDF_RENDER_DPI),
                    "-f", str(page), "-l", str(page),
                    str(self.pdf_path), str(prefix),
                ],
                capture_output=True, text=True, timeout=60,
            )
            if proc.returncode != 0:
                raise ToolError(f"pdftoppm failed: {proc.stderr}")
            matches = sorted(tmpdir.glob("page-*.png"))
            if not matches:
                raise ToolError(f"page {page} did not render (out of range?)")
            img = Image.open(matches[0]).convert("RGB")
            # Cap width for token economy -- this is a reading aid for the
            # model, not the final artifact, so 1400px is plenty of detail.
            if img.width > 1400:
                ratio = 1400 / img.width
                img = img.resize((1400, int(img.height * ratio)), Image.LANCZOS)
            buf = io.BytesIO()
            img.save(buf, format="PNG")
            b64 = base64.b64encode(buf.getvalue()).decode("ascii")
        return [
            {"type": "text", "text": f"paper.pdf, page {page}:"},
            {"type": "image", "source": {"type": "base64", "media_type": "image/png", "data": b64}},
        ]

    # ---------- read_file ----------
    def read_file(self, path: str, offset: int = 1, limit: int = 400) -> str:
        """Read a text file relative to the case directory (usually
        something under repo/), 1-indexed line offset, verbatim -- the
        source of truth for every code_ref snippet."""
        target = (self.case_dir / path).resolve()
        if self.case_dir not in target.parents and target != self.case_dir:
            raise ToolError("path escapes the case directory")
        if not target.is_file():
            raise ToolError(f"no such file: {path}")
        lines = target.read_text(errors="replace").splitlines()
        start = max(1, offset)
        end = min(len(lines), start + limit - 1)
        numbered = "\n".join(f"{i:>6}\t{lines[i - 1]}" for i in range(start, end + 1))
        return numbered or "[empty range]"

    # ---------- crop_figure ----------
    def crop_figure(self, page: int, left: int, top: int, right: int, bottom: int) -> list[dict[str, Any]]:
        """Deterministic crop of a figure's own artwork from a specific
        page, at PDF_RENDER_DPI, in pixel coordinates -- exactly the
        render->crop->verify process CLAUDE.md describes for figure_refs
        and case.overview.figure. Returns the crop for the model to look
        at AND a crop_id string. The model never sees or handles the
        actual image bytes as text -- to use a confirmed crop, it writes
        the literal placeholder string "CROP:<crop_id>" into the
        narrative's `image` field; the orchestrator substitutes the real
        base64 data-uri after submit_narrative, so a multi-KB image never
        has to be retyped as output tokens (which is slow, expensive, and
        was observed to get truncated mid-string against max_tokens)."""
        with subprocess_tmpdir() as tmpdir:
            prefix = tmpdir / "page"
            proc = subprocess.run(
                [
                    "pdftoppm", "-png", "-r", str(PDF_RENDER_DPI),
                    "-f", str(page), "-l", str(page),
                    str(self.pdf_path), str(prefix),
                ],
                capture_output=True, text=True, timeout=60,
            )
            if proc.returncode != 0:
                raise ToolError(f"pdftoppm failed: {proc.stderr}")
            matches = sorted(tmpdir.glob("page-*.png"))
            if not matches:
                raise ToolError(f"page {page} did not render (out of range?)")
            img = Image.open(matches[0]).convert("RGB")
            crop = img.crop((left, top, right, bottom))
            if crop.width < 10 or crop.height < 10:
                raise ToolError("crop box is degenerate (too small) -- check coordinates")
            buf = io.BytesIO()
            crop.save(buf, format="PNG")
            b64 = base64.b64encode(buf.getvalue()).decode("ascii")
            crop_id = f"crop_{len(self._crops) + 1}"
            self._crops[crop_id] = crop
        return [
            {"type": "text", "text": (
                f"crop_id={crop_id}  (page {page}, box=({left},{top},{right},{bottom})). "
                f"If this looks correct, reference it in submit_narrative as the "
                f"literal string \"CROP:{crop_id}\" -- do not copy any image data yourself."
            )},
            {"type": "image", "source": {"type": "base64", "media_type": "image/png", "data": b64}},
        ]

    def resolve_crop_placeholders(self, narrative: dict, max_width: int = 900, quality: int = 88) -> dict:
        """Walk the submitted narrative and replace any string field whose
        value is exactly "CROP:<crop_id>" with the real compressed
        data:image/jpeg;base64,... URI for that crop. This is the only
        place base64 image text is ever produced -- deterministically, by
        this code, never generated by the model."""
        crops: dict[str, Image.Image] = getattr(self, "_crops", {})

        def uri_for(crop_id: str) -> str:
            if crop_id not in crops:
                raise ToolError(
                    f"submit_narrative referenced CROP:{crop_id}, but no such crop "
                    f"was produced via crop_figure this session (have: {list(crops)})"
                )
            img = crops[crop_id].convert("RGB")
            if img.width > max_width:
                ratio = max_width / img.width
                img = img.resize((max_width, int(img.height * ratio)), Image.LANCZOS)
            buf = io.BytesIO()
            img.save(buf, format="JPEG", quality=quality, optimize=True)
            b64 = base64.b64encode(buf.getvalue()).decode("ascii")
            return f"data:image/jpeg;base64,{b64}"

        def walk(node):
            if isinstance(node, dict):
                return {k: walk(v) for k, v in node.items()}
            if isinstance(node, list):
                return [walk(v) for v in node]
            if isinstance(node, str) and node.startswith("CROP:"):
                return uri_for(node[len("CROP:"):])
            return node

        return walk(narrative)

    def resolve_code_snippets(self, narrative: dict) -> dict:
        """Overwrite every code_ref's `snippet` with the exact verbatim text
        read from disk at file:start_line-end_line, ignoring whatever the
        model typed there. Mirrors scripts/verify_case.py's own extraction
        exactly, so a resolved snippet always passes that check by
        construction -- this is the same fix as CROP: placeholders, applied
        to code: never trust the model to retype content character-for-
        character when it can instead be read mechanically. `file` is
        interpreted relative to case.repo_path (default 'repo')."""
        repo_rel = ((narrative.get("case") or {}).get("repo_path") or "repo")
        repo_root = (self.case_dir / repo_rel).resolve()

        def resolve_ref(ref: dict) -> dict:
            file_rel = ref.get("file")
            start, end = ref.get("start_line"), ref.get("end_line")
            if not file_rel or not isinstance(start, int) or not isinstance(end, int):
                return ref
            target = (repo_root / file_rel).resolve()
            if not target.is_file():
                # Same tolerance as verify_case.py's check_code_ref: accept
                # a file path that's case-dir-relative (includes the repo
                # dir's own name) instead of repo_path-relative.
                fallback = (self.case_dir / file_rel).resolve()
                target = fallback if fallback.is_file() else target
            if self.case_dir not in target.parents and target != self.case_dir:
                raise ToolError(f"code_ref file escapes the case directory: {file_rel}")
            if not target.is_file():
                return ref  # let verify_case.py report the missing-file error
            lines = target.read_text(errors="replace").splitlines()
            if end > len(lines) or start < 1:
                return ref  # let verify_case.py report the bad-range error
            return {**ref, "snippet": "\n".join(lines[start - 1:end])}

        def walk(node):
            if isinstance(node, dict):
                if "code_refs" in node and isinstance(node["code_refs"], list):
                    node = {**node, "code_refs": [
                        resolve_ref(r) if isinstance(r, dict) else r for r in node["code_refs"]
                    ]}
                return {k: (walk(v) if k != "code_refs" else v) for k, v in node.items()}
            if isinstance(node, list):
                return [walk(v) for v in node]
            return node

        return walk(narrative)

    # ---------- submit_narrative ----------
    def submit_narrative(self, narrative: dict) -> str:
        self.submitted_narrative = narrative
        return "received."


import contextlib
import shutil
import tempfile


@contextlib.contextmanager
def subprocess_tmpdir():
    d = Path(tempfile.mkdtemp(prefix="pcd_"))
    try:
        yield d
    finally:
        shutil.rmtree(d, ignore_errors=True)


# ---------- Anthropic tool schemas ----------

TOOL_SCHEMAS = [
    {
        "name": "bash",
        "description": (
            "Run a shell command with cwd set to this case's directory. "
            "Use for grep/find/ls/pdftotext against repo/ and paper.pdf -- "
            "the same commands you'd run interactively to search code or "
            "pull the bibliography. Not a general-purpose shell; keep "
            "commands read-only and scoped to this case."
        ),
        "input_schema": {
            "type": "object",
            "required": ["command"],
            "properties": {"command": {"type": "string"}},
        },
    },
    {
        "name": "read_pdf_page",
        "description": (
            "Render one 1-indexed page of paper.pdf as an image and view "
            "it. This is the ONLY way to read the paper -- there is no "
            "text extraction shortcut for the Method section, equations, "
            "or figures. Call once per page you need to actually read."
        ),
        "input_schema": {
            "type": "object",
            "required": ["page"],
            "properties": {"page": {"type": "integer", "minimum": 1}},
        },
    },
    {
        "name": "read_file",
        "description": (
            "Read a text file relative to the case directory verbatim, "
            "with 1-indexed line numbers (e.g. 'repo/model/mapfns.py'). "
            "This is the only legitimate source for a code_ref snippet -- "
            "never write a snippet you have not read this way."
        ),
        "input_schema": {
            "type": "object",
            "required": ["path"],
            "properties": {
                "path": {"type": "string"},
                "offset": {"type": "integer", "minimum": 1, "default": 1},
                "limit": {"type": "integer", "minimum": 1, "default": 400},
            },
        },
    },
    {
        "name": "crop_figure",
        "description": (
            "Render `page` at 200 DPI and crop pixel box "
            "(left, top, right, bottom), then show you the result so you "
            "can confirm it's the right figure, tightly framed, before "
            "using it. Call this to check a crop; if it's wrong, call "
            "again with adjusted coordinates. Only use the result once "
            "you've actually looked at it and it shows the intended "
            "figure, not cut off, not a neighboring figure's caption. "
            "Returns a crop_id -- to actually use the crop, write the "
            "literal string \"CROP:<crop_id>\" as the value of a "
            "figure_refs[].image or case.overview.figure.image field in "
            "submit_narrative. NEVER write out base64 image data yourself "
            "-- always use this placeholder string instead; the real "
            "bytes are substituted in automatically after you submit."
        ),
        "input_schema": {
            "type": "object",
            "required": ["page", "left", "top", "right", "bottom"],
            "properties": {
                "page": {"type": "integer", "minimum": 1},
                "left": {"type": "integer"},
                "top": {"type": "integer"},
                "right": {"type": "integer"},
                "bottom": {"type": "integer"},
            },
        },
    },
    {
        "name": "submit_narrative",
        "description": (
            "Submit the completed narrative.json for this case. Ends the "
            "session -- only call this once, when the case is fully done "
            "and you believe it would pass scripts/verify_case.py. Pass "
            "the entire narrative object (case + nodes), schema-shaped. "
            "For any figure image, set the field to the literal string "
            "\"CROP:<crop_id>\" from a prior crop_figure call -- never "
            "write actual base64 image data as part of this call; it is "
            "slow, expensive, and has been observed to get truncated "
            "against the output length limit. If you haven't called "
            "crop_figure for a figure, omit `image` for it rather than "
            "inventing one. For every code_ref, `snippet` is automatically "
            "re-read from disk at file:start_line-end_line after you "
            "submit and OVERWRITES whatever text you put there -- so don't "
            "spend effort hand-copying it character-for-character (that's "
            "slow and error-prone); a short placeholder is fine. What "
            "actually has to be exactly right is `file`, `start_line`, and "
            "`end_line` -- get the range tight around the real logic, not "
            "off by a few lines and not spanning boilerplate."
        ),
        "input_schema": {
            "type": "object",
            "required": ["narrative"],
            "properties": {"narrative": {"type": "object"}},
        },
    },
]


def dispatch(ws: CaseWorkspace, name: str, tool_input: dict) -> Any:
    if name == "bash":
        return ws.bash(tool_input["command"])
    if name == "read_pdf_page":
        return ws.read_pdf_page(tool_input["page"])
    if name == "read_file":
        return ws.read_file(
            tool_input["path"],
            tool_input.get("offset", 1),
            tool_input.get("limit", 400),
        )
    if name == "crop_figure":
        return ws.crop_figure(
            tool_input["page"], tool_input["left"], tool_input["top"],
            tool_input["right"], tool_input["bottom"],
        )
    if name == "submit_narrative":
        return ws.submit_narrative(tool_input["narrative"])
    raise ToolError(f"unknown tool: {name}")