File size: 13,180 Bytes
f0634fb
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
import type { ContentBlock, McpServer, ToolCallContent } from '@agentclientprotocol/sdk';
import {
  buildImageCompressionCaption,
  compressBase64ForModel,
  type ContentPart,
  type McpServerConfig,
  parseImageDataUrl,
  persistOriginalImage,
} from '@moonshot-ai/agent-core-v2';
import type { ToolResultEvent } from '@moonshot-ai/agent-core-v2/events';
import type { ToolInputDisplay } from '@moonshot-ai/agent-core-v2/tool/toolInputDisplay';

import { log } from './log';
import { isHideOutputMarker } from './marker';

/**
 * Convert an array of ACP {@link ContentBlock}s into agent-core-v2
 * {@link ContentPart}s suitable for a user `ContextMessage`'s `content`.
 *
 * Image parts are built from the client-declared MIME verbatim; run the
 * result through {@link compressPromptImageParts} before submitting so
 * unsupported formats are dropped and MIME aliases canonicalized. Audio and
 * blob embedded resources are dropped with a warning (ACP
 * `promptCapabilities` currently advertise audio as unsupported).
 */
export function acpBlocksToContentParts(blocks: readonly ContentBlock[]): readonly ContentPart[] {
  const out: ContentPart[] = [];
  for (const block of blocks) {
    if (block.type === 'text') {
      out.push({ type: 'text', text: block.text });
      continue;
    }
    if (block.type === 'image') {
      const url = `data:${block.mimeType};base64,${block.data}`;
      out.push({ type: 'image_url', imageUrl: { url } });
      continue;
    }
    if (block.type === 'audio') {
      log.warn('acp: dropping unsupported audio prompt block', {
        mimeType: block.mimeType,
      });
      continue;
    }
    if (block.type === 'resource_link') {
      const fileRef = fileLinkToTextRef(block.uri);
      if (fileRef !== null) {
        out.push({ type: 'text', text: fileRef });
        continue;
      }
      const text = `<resource_link uri="${escapeXmlAttr(block.uri)}" name="${escapeXmlAttr(
        block.name,
      )}" />`;
      out.push({ type: 'text', text });
      continue;
    }
    if (block.type === 'resource') {
      const resource = block.resource;
      if ('text' in resource) {
        // TextResourceContents β€” wrap as a `<resource>` element so the
        // model sees the uri provenance alongside the text body.
        const text = `<resource uri="${escapeXmlAttr(resource.uri)}">${resource.text}</resource>`;
        out.push({ type: 'text', text });
        continue;
      }
      // BlobResourceContents β€” drop+warn.
      log.warn('acp: dropping blob embedded resource', {
        uri: resource.uri,
        mimeType: resource.mimeType,
      });
      continue;
    }
    // Future-proof: anything else (new ACP block kinds) β†’ warn and drop.
    log.warn('acp: dropping unsupported prompt content block', {
      type: (block as { type: string }).type,
    });
  }
  return out;
}

/**
 * Shrink oversized inline images in a prompt-part list β€” the ACP ingestion
 * point's input-stage compression, mirroring kap-server's upload-time step
 * (`resolvePromptMediaFiles`). Best effort: a part that cannot be compressed
 * is passed through unchanged.
 *
 * Compression is NOT duplicated by the engine: agent-core-v2's prompt pipeline
 * (`agent/prompt/promptService.ts`) only *extracts* pre-existing compression
 * captions from user text (rerouting them to system reminders) β€” it never
 * compresses images at the prompt entry, so the edge ingestion point owns
 * that step.
 *
 * Format gating is deliberately left to the engine: the accepted image
 * formats depend on the provider the agent is bound to, which this edge does
 * not know. The engine's prompt pipeline gates every image part against that
 * provider's set (dropping rejected parts for a text notice and rewriting
 * accepted MIME aliases to their canonical form) before anything reaches the
 * session history, so parts in formats we cannot re-encode pass through here
 * untouched.
 *
 * Compression is never silent: a re-encoded image gains a caption text part
 * immediately before it stating what the original was, and the original bytes
 * are persisted (into `originalsDir` β€” typically the session's
 * media-originals dir β€” or the shared temp-dir fallback) so the model can
 * read fine detail back via ReadMediaFile + region.
 */
export async function compressPromptImageParts(
  parts: readonly ContentPart[],
  options: {
    readonly originalsDir?: string | undefined;
    /**
     * Longest-edge ceiling (px) override. The ACP server runs the engine
     * in-process, so the Agent-scope `ImageConfigBridge` has already pushed
     * the env-resolved `[image]` config section into the compression module's
     * global seam β€” leave this `undefined` (the default) and the configured /
     * built-in cap applies. The override exists for tests.
     */
    readonly maxImageEdgePx?: number | undefined;
  } = {},
): Promise<ContentPart[]> {
  const out: ContentPart[] = [];
  for (const part of parts) {
    if (part.type === 'image_url') {
      const parsed = parseImageDataUrl(part.imageUrl.url);
      if (parsed !== null) {
        const result = await compressBase64ForModel(parsed.base64, parsed.mimeType, {
          maxEdge: options.maxImageEdgePx,
        });
        if (result.changed) {
          const originalPath = await persistOriginalImage(
            Buffer.from(parsed.base64, 'base64'),
            parsed.mimeType,
            { dir: options.originalsDir },
          );
          out.push({
            type: 'text',
            text: buildImageCompressionCaption({
              original: {
                width: result.originalWidth,
                height: result.originalHeight,
                byteLength: result.originalByteLength,
                mimeType: parsed.mimeType,
              },
              final: {
                width: result.width,
                height: result.height,
                byteLength: result.finalByteLength,
                mimeType: result.mimeType,
              },
              originalPath,
            }),
          });
          out.push({
            type: 'image_url',
            imageUrl: { ...part.imageUrl, url: `data:${result.mimeType};base64,${result.base64}` },
          });
          continue;
        }
      }
    }
    out.push(part);
  }
  return out;
}

/**
 * Convert ACP `session/new` / `session/load` `mcpServers` β€” a named array
 * discriminated by `type` (absent = stdio) β€” into the engine's name-keyed
 * {@link McpServerConfig} record. Returns `undefined` for an absent/empty
 * list (or when every entry was dropped) so the engine builds no session
 * overlay. The unstable `type: 'acp'` transport is unsupported and dropped
 * with a warning.
 */
export function acpMcpServersToConfigRecord(
  servers: readonly McpServer[] | undefined,
): Record<string, McpServerConfig> | undefined {
  if (servers === undefined || servers.length === 0) return undefined;
  const out: Record<string, McpServerConfig> = {};
  for (const server of servers) {
    if (!('type' in server)) {
      out[server.name] = {
        transport: 'stdio',
        command: server.command,
        args: server.args,
        env: namedPairsToRecord(server.env),
        runtime_id: 'local',
      };
      continue;
    }
    if (server.type === 'http' || server.type === 'sse') {
      out[server.name] = {
        transport: server.type,
        url: server.url,
        headers: namedPairsToRecord(server.headers),
      };
      continue;
    }
    log.warn('acp: dropping unsupported MCP server transport', {
      name: server.name,
      type: server.type,
    });
  }
  return Object.keys(out).length === 0 ? undefined : out;
}

/** ACP env/header lists are `{name, value}` arrays; the engine wants a record. */
function namedPairsToRecord(
  pairs: readonly { readonly name: string; readonly value: string }[],
): Record<string, string> | undefined {
  if (pairs.length === 0) return undefined;
  return Object.fromEntries(pairs.map((p) => [p.name, p.value]));
}

/**
 * Minimum-viable XML-attribute escaping for prompt-embedded resource
 * wrappers. The output is consumed by an LLM, not parsed by a canonical
 * XML parser, so we only escape the five characters that would change the
 * apparent tag structure: `&`, `<`, `>`, `"`, `'`. `&` must run
 * first to avoid double-escaping the entities introduced by the others.
 */
function escapeXmlAttr(s: string): string {
  return s
    .replaceAll('&', '&amp;')
    .replaceAll('<', '&lt;')
    .replaceAll('>', '&gt;')
    .replaceAll('"', '&quot;')
    .replaceAll("'", '&apos;');
}

function fileLinkToTextRef(uri: string): string | null {
  let url: URL;
  try {
    url = new URL(uri);
  } catch {
    return null;
  }
  if (url.protocol !== 'file:') return null;

  let path: string;
  try {
    path = decodeURIComponent(url.pathname);
  } catch {
    return null;
  }

  // `file://server/share/a.ts` is the URI form of a Windows UNC path
  // (`\\server\share\a.ts`). `URL.pathname` only carries `/share/a.ts`; the
  // host is part of the file location, so keep it in the projected text ref.
  // `file://localhost/...` is still treated as local. Host is lower-cased so
  // `file://Server/...` and `file://server/...` collapse to one ref.
  const host = url.hostname.toLowerCase();
  const isUncHost = host !== '' && host !== 'localhost';

  // Drive-letter normalization is local-only: a UNC URI never legitimately
  // carries `/C:/...` in its path, so we leave such inputs untouched rather
  // than stripping a leading slash that would alter the UNC payload.
  if (!isUncHost && /^\/[A-Za-z]:/.test(path)) path = path.slice(1);

  if (isUncHost) {
    path = `//${host}${path.startsWith('/') ? path : `/${path}`}`;
  }

  const range = parseLineRange(url.hash) ?? parseLineRange(url.search);
  return range !== null ? `${path}:${range}` : path;
}

function parseLineRange(suffix: string): string | null {
  if (!suffix) return null;
  const body = suffix.replace(/^[#?]/, '');
  const match = /^(?:lines?=|L)(\d+)(?:[-:]L?(\d+))?/i.exec(body);
  if (!match) return null;
  return match[2] !== undefined ? `${match[1]}-${match[2]}` : match[1]!;
}

/**
 * Project a {@link ToolInputDisplay} block into an ACP {@link ToolCallContent}
 * entry for the tool-call card. Diff/file_io blocks become inline diffs;
 * plan_review becomes a text content entry; everything else yields `null`
 * (the caller drops it).
 */
export function displayBlockToAcpContent(block: ToolInputDisplay): ToolCallContent | null {
  if (block.kind === 'diff') {
    return {
      type: 'diff',
      path: block.path,
      oldText: block.before,
      newText: block.after,
    };
  }
  if (block.kind === 'file_io' && block.before !== undefined && block.after !== undefined) {
    return {
      type: 'diff',
      path: block.path,
      oldText: block.before,
      newText: block.after,
    };
  }
  if (block.kind === 'plan_review') {
    const text = composePlanContent(block);
    if (text === null) return null;
    return { type: 'content', content: { type: 'text', text } };
  }
  return null;
}

/**
 * Render the text body of a `plan_review` display block. Empty plan β†’ `null`
 * (caller drops the entry). When `block.path` is set, prefix with the on-disk
 * location so the client can show it alongside the markdown body.
 */
function composePlanContent(
  block: Extract<ToolInputDisplay, { kind: 'plan_review' }>,
): string | null {
  if (block.plan.trim().length === 0) return null;
  if (block.path !== undefined) {
    return `Plan saved to: ${block.path}\n\n${block.plan}`;
  }
  return block.plan;
}

/**
 * Convert a {@link ToolResultEvent}'s `output` into ACP
 * {@link ToolCallContent} entries.
 *
 * A non-empty string is passed through as a text block; objects/arrays are
 * JSON-stringified (best-effort β€” falls back to a placeholder on circular
 * structures). Empty/undefined/null output yields an empty array β€” the caller
 * still emits a `tool_call_update` so the client sees the status transition
 * to completed/failed.
 *
 * Diff content does NOT come from this function: `ToolResultEvent` has no
 * `display` field; diffs attach to `ToolCallStartedEvent.display` and are
 * emitted by `toolCallStartToSessionUpdate`.
 */
export function toolResultToAcpContent(event: ToolResultEvent): ToolCallContent[] {
  const out = event.output;
  // Array output containing the HideOutputMarker tells the adapter to suppress
  // this tool's textual content entirely (e.g. terminal output routed through
  // its own reverse-RPC channel). Detected before any other processing so
  // mark-bearing outputs never leak even a stringified preview.
  if (Array.isArray(out) && out.some(isHideOutputMarker)) {
    return [];
  }
  if (out === undefined || out === null) return [];
  if (typeof out === 'string') {
    if (out.length === 0) return [];
    return [{ type: 'content', content: { type: 'text', text: out } }];
  }
  // Best-effort stringify for object/array outputs.
  let text: string;
  try {
    text = JSON.stringify(out);
  } catch {
    text = '[object]';
  }
  if (!text) return [];
  return [{ type: 'content', content: { type: 'text', text } }];
}