File size: 8,660 Bytes
4e23b01
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
import type { Tool } from './tool';

export type Role = 'system' | 'user' | 'assistant' | 'tool';

export interface TextPart {
  type: 'text';
  text: string;
}

export interface ThinkPart {
  type: 'think';
  think: string;
  encrypted?: string; // Provider-specific reasoning signature
  hidden?: boolean;
}

export interface ImageURLPart {
  type: 'image_url';
  imageUrl: { url: string; id?: string; name?: string };
}

export interface AudioURLPart {
  type: 'audio_url';
  audioUrl: { url: string; id?: string };
}

export interface VideoURLPart {
  type: 'video_url';
  videoUrl: { url: string; id?: string | undefined; name?: string };
}

/**
 * A single piece of content within a {@link Message}.
 *
 * The union covers text, model reasoning ("think"), images, audio, and video.
 * Providers convert these to their native content-block format during
 * {@link ChatProvider.generate}.
 */
export type ContentPart = TextPart | ThinkPart | ImageURLPart | AudioURLPart | VideoURLPart;

export interface ToolCall {
  type: 'function';
  id: string;
  name: string;
  arguments: string | null;
  extras?: Record<string, unknown>;
  /**
   * Provider-specific streaming index used to route argument deltas to the
   * correct parallel tool call. Set by streaming providers (OpenAI Chat
   * Completions `index`, Responses API `item_id`). Consumed internally by
   * {@link generate} and stripped before the ToolCall is stored on a Message.
   *
   * @internal
   */
  _streamIndex?: number | string;
}

/** Streaming delta for tool call arguments. */
export interface ToolCallPart {
  type: 'tool_call_part';
  argumentsPart: string | null;
  /**
   * Provider-specific index for routing this streaming delta to the correct
   * parallel tool call. Used by OpenAI Chat Completions (`index`) and
   * Responses API (`item_id`/`output_index`). When absent, the delta is
   * appended to the most-recently-seen ToolCall (single-tool-call fallback).
   */
  index?: number | string;
}

/**
 * A single chunk yielded by {@link StreamedMessage}'s async iterator.
 *
 * During streaming, the generate loop receives a sequence of these parts and
 * merges compatible consecutive parts (e.g. TextPart + TextPart) in-place so
 * the final {@link Message} contains fully-assembled content.
 *
 * Tool-call completion is inferred from merge boundaries (a non-merging next
 * part flushes the pending tool call) and from stream end. Provider adapters
 * are responsible for translating their native "done" signals into this
 * shape; they do not emit a separate done event.
 */
export type StreamedMessagePart = ContentPart | ToolCall | ToolCallPart;

/**
 * A single message in a conversation.
 *
 * Messages carry a {@link role} (system, user, assistant, or tool), an array
 * of {@link ContentPart} content blocks, and optional {@link ToolCall} entries.
 * Tool result messages set {@link toolCallId} to correlate with the originating
 * call.
 */
export interface Message {
  /** The role of the message sender. */
  readonly role: Role;
  /** Optional display name for the sender (used by some providers). */
  readonly name?: string;
  /** Ordered content parts (text, images, thinking, etc.). */
  readonly content: ContentPart[];
  /** Tool calls requested by the assistant in this message. */
  readonly toolCalls: ToolCall[];
  /** For `tool` role messages, the ID of the tool call this result answers. */
  readonly toolCallId?: string;
  /** When `true`, indicates the message was not fully received (e.g. stream interrupted). */
  readonly partial?: boolean;
  /**
   * Full tool definitions carried by this message. Meaningful only on
   * `role: 'system'` messages: it is the append-only primitive for loading a
   * tool mid-conversation without touching the request's top-level `tools[]`
   * (which must stay byte-stable to preserve the provider's prompt cache).
   * Providers that support message-level tool declarations (Kimi
   * `messages[].tools`) serialize it; callers must not send such a message to
   * a provider without that capability.
   */
  readonly tools?: readonly Tool[] | undefined;
}

/** Check if a streamed part is a ContentPart (text, think, image_url, audio_url, video_url). */
export function isContentPart(part: StreamedMessagePart): part is ContentPart {
  const t = part.type;
  return (
    t === 'text' || t === 'think' || t === 'image_url' || t === 'audio_url' || t === 'video_url'
  );
}

/**
 * True for a message whose only payload is `tools` — the dynamic tool-loading
 * primitive (see {@link Message.tools}). Message-level tool declarations are a
 * Kimi wire feature; every other provider must skip such a message entirely:
 * their explicit field construction already keeps the `tools` field off the
 * wire, but the leftover empty message would be rejected (OpenAI: system
 * message without content) or serialized as a garbage `<system></system>`
 * turn (Anthropic/Google system-to-user wrapping).
 */
export function isToolDeclarationOnlyMessage(message: Message): boolean {
  return (
    message.tools !== undefined &&
    message.tools.length > 0 &&
    message.content.length === 0 &&
    message.toolCalls.length === 0
  );
}

/** Check if a streamed part is a ToolCall. */
export function isToolCall(part: StreamedMessagePart): part is ToolCall {
  return part.type === 'function';
}

/** Check if a streamed part is a ToolCallPart (streaming argument delta). */
export function isToolCallPart(part: StreamedMessagePart): part is ToolCallPart {
  return part.type === 'tool_call_part';
}

/**
 * Merge `source` into `target` in-place for streaming accumulation.
 *
 * Supported combinations:
 * - TextPart + TextPart -> concatenate text
 * - ThinkPart + ThinkPart -> concatenate think (refuse if target.encrypted already set)
 * - ToolCall + ToolCallPart -> append arguments
 *
 * **Routing for parallel tool calls**: When OpenAI (or compatible) APIs stream
 * multiple tool calls in parallel, argument deltas may interleave across calls.
 * To handle this, {@link generate} routes ToolCallParts by their optional
 * {@link ToolCallPart.index} field (mirroring the provider's streaming index)
 * to the correct pending ToolCall, rather than relying on sequential ordering.
 * This function still performs sequential merging as a fallback when the
 * pending part matches the incoming one.
 *
 * Returns `true` if the merge was performed, `false` otherwise.
 */
export function mergeInPlace(target: StreamedMessagePart, source: StreamedMessagePart): boolean {
  // TextPart + TextPart
  if (target.type === 'text' && source.type === 'text') {
    target.text += source.text;
    return true;
  }

  // ThinkPart + ThinkPart
  if (target.type === 'think' && source.type === 'think') {
    if (target.encrypted !== undefined) {
      return false;
    }
    target.think += source.think;
    if (source.encrypted !== undefined) {
      target.encrypted = source.encrypted;
    }
    return true;
  }

  // ToolCall + ToolCallPart
  if (target.type === 'function' && source.type === 'tool_call_part') {
    if (source.argumentsPart !== null) {
      target.arguments =
        target.arguments === null
          ? source.argumentsPart
          : target.arguments + source.argumentsPart;
    }
    return true;
  }

  return false;
}

/**
 * Extract the concatenated text from a message's content parts.
 *
 * @param message The message to extract text from.
 * @param sep Separator between text parts. Defaults to empty string.
 */
export function extractText(message: Message, sep: string = ''): string {
  return message.content
    .filter((part): part is TextPart => part.type === 'text')
    .map((part) => part.text)
    .join(sep);
}

/**
 * @deprecated Use `extractText` instead.
 */
export function getTextContent(message: Message): string {
  return extractText(message);
}

/** Create a simple user message with a single text part. */
export function createUserMessage(content: string): Message {
  return {
    role: 'user',
    content: [{ type: 'text', text: content }],
    toolCalls: [],
  };
}

/** Create an assistant message from content parts and optional tool calls. */
export function createAssistantMessage(content: ContentPart[], toolCalls?: ToolCall[]): Message {
  return {
    role: 'assistant',
    content,
    toolCalls: toolCalls ?? [],
  };
}

/** Create a tool result message. */
export function createToolMessage(toolCallId: string, output: string | ContentPart[]): Message {
  const content: ContentPart[] =
    typeof output === 'string' ? [{ type: 'text', text: output }] : output;
  return {
    role: 'tool',
    content,
    toolCalls: [],
    toolCallId,
  };
}