openclaw / src /agents /failover /message-patterns.ts
SaylorTwift's picture
SaylorTwift HF Staff
Add files using upload-large-folder tool
e249c6d verified
Raw History Blame Contribute Delete
18.5 kB
import { normalizeLowercaseStringOrEmpty } from "@openclaw/normalization-core/string-coerce";
type ErrorPattern = RegExp | string;
// Both figures must be denominated in tokens and come from one clause. A message can state an RPM
// limit and mention TPM elsewhere, and reading the pair on its own would compare a request count
// against a token budget; requiring the unit to lead the clause keeps the numbers commensurable.
const STATED_TOKEN_SIZES_RE =
/(?:\btpm\b|tokens per minute)[^.\n]*?\blimit\s+([\d,]+)[^.\n]*?\brequested\s+([\d,]+)/i;
function readStatedTokenCount(digits: string | undefined): number | undefined {
const parsed = Number(digits?.replaceAll(",", ""));
return Number.isFinite(parsed) && parsed > 0 ? parsed : undefined;
}
/**
* Groq denominates a per-request size ceiling per minute: an oversized single request is refused
* with a 413 naming TPM that states both `Limit <n>` and `Requested <m>`. A request larger than
* the whole limit does not fit even an empty bucket, so waiting can never admit it. Ordinary
* throttling states a requested size within the limit and remains a rate limit.
*
* The ceiling belongs to the request and to the refusing provider's quota, not to the model's
* context window, so compaction budgeted against that window cannot satisfy it either.
* Embedded recovery surfaces reset guidance without retrying. If a transport-owning harness
* bypasses that recovery, model failover may advance to a differently provisioned candidate.
*/
export function isProviderRequestSizeCeilingError(errorMessage?: string): boolean {
if (!errorMessage) {
return false;
}
const stated = STATED_TOKEN_SIZES_RE.exec(errorMessage);
const limit = readStatedTokenCount(stated?.[1]);
const requested = readStatedTokenCount(stated?.[2]);
return limit !== undefined && requested !== undefined && requested > limit;
}
// First-party model transports use these terminal-contract forms when EOF arrives
// before a response is complete; keep non-model stream lifecycle errors out.
// The completions transport throws `Stream ended without finish_reason` (no
// "terminal" wording); Mistral/Google keep the longer form. Plugin lifecycle
// strings such as `opencode-go stream ended without a terminal event` must not
// match — those are not assistant-stream contracts.
// Responses EOF can leave a tool call open; a completed response with unresolved
// calls instead indicates an inconsistent terminal payload, not a disconnect.
export const INCOMPLETE_ASSISTANT_STREAM_RE =
/^(?:[\w -]*stream ended (?:before (?:message_?stop|(?:a )?terminal (?:finish reason|response event|event))|without (?:a terminal )?finish[_ ]reason)|Responses stream ended with unresolved tool calls)[.!]?$/i;
// Undici ends a stream body with this exact bare transport message. Keep it
// anchored so unrelated failures that merely contain the word do not match.
export const TERMINATED_TRANSPORT_MESSAGE_RE = /^terminated$/i;
// These exact transport diagnostics identify rejection, not whether replay is safe.
const PRE_DISPATCH_TOOL_CALL_REJECTION_MESSAGES = new Set([
"Provider completed tool call with malformed JSON arguments",
"Provider completed stream with an incomplete tool call",
"Provider returned an incomplete or malformed tool call",
"Mistral completed tool call has invalid JSON arguments",
"Responses stream completed tool call with invalid JSON arguments",
]);
export function isPreDispatchToolCallRejectionMessage(errorMessage?: string): boolean {
return errorMessage !== undefined && PRE_DISPATCH_TOOL_CALL_REJECTION_MESSAGES.has(errorMessage);
}
const PERIODIC_USAGE_LIMIT_RE =
/\b(?:daily|weekly|monthly)(?:\/(?:daily|weekly|monthly))* (?:usage )?limit(?:s)?(?: (?:exhausted|reached|exceeded))?\b/i;
const HIGH_CONFIDENCE_AUTH_PERMANENT_PATTERNS = [
/api[_ ]?key[_ ]?(?:revoked|deactivated|deleted)/i,
/deactivated[_ ]workspace/i,
"key has been disabled",
"key has been revoked",
"account has been deactivated",
"not allowed for this organization",
] as const satisfies readonly ErrorPattern[];
// Providers use both "invalid API key" and "API key is/not valid" word order.
// Keep them in one matcher so every result/exception classifier agrees on auth failover.
const INVALID_API_KEY_RE =
/(?:invalid[_ ]?api[_ ]?key(?![a-z0-9])|api[_ ]?key(?:[_ ]?(?:is[_ ]?)?(?:invalid(?![a-z0-9])|not[_ ]?valid(?![a-z0-9]))))/i;
const AMBIGUOUS_AUTH_ERROR_PATTERNS = [
INVALID_API_KEY_RE,
/could not (?:authenticate|validate).*(?:api[_ ]?key|credentials)/i,
"permission_error",
] as const satisfies readonly ErrorPattern[];
const COMMON_AUTH_ERROR_PATTERNS = [
"incorrect api key",
"invalid token",
"authentication",
"re-authenticate",
"oauth token refresh failed",
"unauthorized",
"forbidden",
"access denied",
"insufficient permissions",
"insufficient permission",
/missing scopes?:/i,
"expired",
"token has expired",
/\b401\b/,
/\b403\b/,
"no credentials found",
"no api key found",
/\bfailed to (?:extract|parse|validate|decode)\b.*\btoken\b/,
] as const satisfies readonly ErrorPattern[];
const CJK_AUTH_ERROR_PATTERNS = [
"无权访问",
"认证失败",
"鉴权失败",
"密钥无效",
"apikey 无效",
/(?:当前\s*ak|ce-011).*?(?:违规请求|禁止访问)|(?:违规请求|禁止访问).*?(?:当前\s*ak|ce-011)/i,
/\bce-011\b/i,
] as const satisfies readonly ErrorPattern[];
const ZAI_BILLING_CODE_1311_RE = /"code"\s*:\s*1311\b/;
const ZAI_AUTH_CODE_1113_RE = /"code"\s*:\s*1113\b/;
const VOLCENGINE_INVALID_SUBSCRIPTION_RE = /"code"\s*:\s*"InvalidSubscription"/i;
const STATUS_INTERNAL_SERVER_ERROR_RE = /\bstatus:\s*internal server error\b/i;
const STATUS_INTERNAL_SERVER_ERROR_WITH_500_RE =
/^(?=[\s\S]*\bstatus:\s*internal server error\b)(?=[\s\S]*\bcode["']?\s*[:=]\s*500\b)/i;
const HTTP_5XX_STATUS_RE = /\bHTTP\s+5\d\d\b/i;
const BILLING_ERROR_HARD_402_RE =
/["']?(?:status|code)["']?\s*[:=]\s*402\b|\bhttp\s*402\b|\berror(?:\s+code)?\s*[:=]?\s*402\b|^\s*402\s+payment/i;
// Numeric ids and token counts are not HTTP throttling signals. Require a
// standalone status token, HTTP/status context, or a structured status/code shape.
const RATE_LIMIT_429_RE =
/^\s*429\b|\b(?:https?|status(?:[ _-]?code)?|response(?:[ _-]?code)?|http(?:[ _-]?status)?)\b[\s:=#"'(]{0,6}429\b|["'](?:status|code)["']\s*:\s*429\b|\b429\b[\s:)\].,-]*(?:rate[_ -]?limit(?:ed|ing)?|too many requests|resource has been exhausted|quota(?:\s+(?:exceeded|exhausted|depleted|reached))?)\b/i;
const ZAI_AUTH_ERROR_PATTERNS = [
// Z.ai: error 1113 = wrong endpoint or invalid credentials (#48988)
ZAI_AUTH_CODE_1113_RE,
] as const satisfies readonly ErrorPattern[];
const ERROR_PATTERNS = {
rateLimit: [
/rate[_ ]limit|too many requests/i,
RATE_LIMIT_429_RE,
/too many (?:concurrent )?requests/i,
// Accepted tradeoff: unrealistic provider text such as "throttling disabled"
// still matches this intentionally broad provider-error signal.
/\bthrottl(?:ing[_]?exception|ing|ed)\b/i,
/\bconcurrency limit\b.*\b(?:breached|reached)\b/i,
"model_cooldown",
"exceeded your current quota",
/\bresource[_ -]?exhausted\b/i,
/\bquota[_ -]?exceeded\b/i,
"usage limit",
/\btpm\b/i,
"tokens per minute",
"tokens per day",
// Chinese provider rate-limit messages
"请求过于频繁",
"调用频率",
"频率限制",
"配额不足",
"配额已用尽",
"额度不足",
"额度已用尽",
],
overloaded: [
/overloaded_error|"type"\s*:\s*"overloaded_error"/i,
"overloaded",
/\b(?:selected\s+)?model\s+(?:is\s+)?at capacity\b/i,
/\bservice(?:[_ ]temporarily)?[_ ]unavailable\b/i,
"high demand",
"high load",
// Chinese provider overloaded messages
"服务过载",
"当前负载过高",
"访问量过大",
],
serverError: [
"an error occurred while processing",
"internal server error",
"internal_error",
"server_error",
"bad gateway",
"gateway timeout",
"upstream error",
"upstream connect error",
"connection reset",
// Chinese provider server error messages
"内部错误",
"服务器错误",
"服务器内部错误",
"系统错误",
"系统繁忙",
"系统异常",
],
timeout: [
"timeout",
"timed out",
"deadline exceeded",
"context deadline exceeded",
/^(?=[\s\S]*\bgot status:\s*internal\b)(?=[\s\S]*\bcode["']?\s*[:=]\s*500\b)/i,
/^(?=[\s\S]*["']status["']\s*:\s*["']internal["'])(?=[\s\S]*["']code["']\s*:\s*500\b)/i,
"connection error",
"network error",
"network request failed",
"fetch failed",
"socket hang up",
// Codex and node-fetch expose these exact terminal transport messages.
// Keep them anchored so unrelated local stream failures do not trigger model failover.
/^stream disconnected before completion(?::[\s\S]*)?$/i,
/^premature close of server response while trying to fetch\b/i,
INCOMPLETE_ASSISTANT_STREAM_RE,
// Chinese provider error messages (ZhipuAI/GLM, Bailian, Kimi/Moonshot, DeepSeek, etc.)
"网络错误",
"网络异常",
"服务暂时不可用",
"服务繁忙",
"请求超时",
"连接超时",
"连接错误",
/\beconn(?:refused|reset|aborted)\b/i,
/\benetunreach\b/i,
/\behostunreach\b/i,
/\behostdown\b/i,
/\benetreset\b/i,
/\betimedout\b/i,
/\besockettimedout\b/i,
/\bepipe\b/i,
/\benotfound\b/i,
/\beai_again\b/i,
/without sending (?:any )?chunks?/i,
// Bare `error` is a provider-completed failure, not a hang — classified as
// server_error separately so diagnostics stay accurate while fallback still
// runs (#109218). Keep abort / network / malformed as timeout-like transients.
/\bstop reason:\s*(?:abort|malformed_response|network_error)\b/i,
/\breason:\s*(?:abort|malformed_response|network_error)\b/i,
/\bunhandled stop reason:\s*(?:abort|malformed_response|network_error)\b/i,
// `\breason:` does not match provider payloads like `finish_reason: network_error` (#61281).
/\bfinish_reason:\s*(?:abort|malformed_response|network_error)\b/i,
// AbortError messages from fetch/stream aborts (Ollama NDJSON stream
// timeouts, signal aborts, etc.) — without these the flattened message
// falls through to reason=unknown (#58315).
/\boperation was aborted\b/i,
/\bstream (?:was )?(?:closed|aborted)\b/i,
// Undici transport-level failures during CDN/provider outages (Cloudflare
// 502 served with an empty body, socket reset mid-response, body-stream
// aborted). These arrive as bare strings on the outer error and, without
// an explicit match, the fallback chain is never attempted (#69368).
TERMINATED_TRANSPORT_MESSAGE_RE,
/^stream_read_error$/i,
/\bund_err_(?:socket|connect|headers?|body|req_content_length_mismatch|aborted|closed)\b/i,
// shared model runtime's openai provider surfaces `Request failed` when the HTTP
// response has no body and no status text (typical of Cloudflare 502s
// from the upstream Codex service). Treat it as a transport failure so
// the configured fallback chain runs instead of surfacing the error.
/^request failed$/i,
/\brequest failed after repeated internal retries\b/i,
// The generic assistant error text "LLM request failed." is produced by
// formatUserFacingAssistantErrorText when the underlying provider error
// cannot be formatted into a specific category. For local providers (LM
// Studio, Ollama) this wraps connection/availability failures when the
// model is not loaded or the endpoint is unreachable. Without this match,
// cron retry and payload.fallbacks never engage because the error is not
// classified as any transient type (#93931).
// Use a strict exact-match regex so variants like
// "LLM request failed: provider rejected the request schema or tool payload."
// (a format/schema error, not transient) are NOT caught here — they
// fall through to their own pattern classifications.
/^llm request failed\.$/i,
],
billing: [
BILLING_ERROR_HARD_402_RE,
/\b(?:got|returned|received)\s+(?:a\s+)?402\b(?!\s+records\b)/i,
"payment required",
"insufficient credits",
/used\s+all\s+available\s+credits/i,
/(?:monthly\s+)?spend(?:ing)?\s+limit/i,
/insufficient[_ ]quota/i,
/\b(?:go|free)usagelimiterror\b/i,
"available balance",
"out of budget",
"credit balance",
"plans & billing",
/insufficient[_ ]balance/i,
// Fuzzy: "Insufficient MBT balance", "Insufficient token balance", etc.
// Exactly one intervening word — avoids false positives like
// "insufficient to reconcile the final balance"
/\binsufficient\s+\w+\s+balance\b/i,
"insufficient usd or diem balance",
/requires?\s+more\s+credits/i,
/out of extra usage/i,
/draw from your extra usage/i,
/extra usage is required(?: for long context requests)?/i,
// Chinese provider billing messages
"余额不足",
"账户余额不足",
"欠费",
"账户已欠费",
// Volcengine Coding Plan entitlement failure. Official Ark error code:
// HTTP 400 + InvalidSubscription means the plan is missing or expired.
VOLCENGINE_INVALID_SUBSCRIPTION_RE,
/\bdoes not have a valid coding\s*plan subscription\b/i,
// Z.ai: error 1311 = model not included in current subscription plan (#48988)
ZAI_BILLING_CODE_1311_RE,
/\bcurrent\s+subscription\s+plan\b.*\b(?:does\s+not|doesn't|not)\b.*\binclude\s+access\b/i,
/\bmodel\b.*\bnot\s+available\b.*\bcurrent\s+plan\b/i,
],
authPermanent: HIGH_CONFIDENCE_AUTH_PERMANENT_PATTERNS,
auth: [
...AMBIGUOUS_AUTH_ERROR_PATTERNS,
...COMMON_AUTH_ERROR_PATTERNS,
...ZAI_AUTH_ERROR_PATTERNS,
...CJK_AUTH_ERROR_PATTERNS,
],
format: [
"string should match pattern",
"tool_use.id",
"tool_use_id",
"messages.1.content.1.tool_use.id",
"invalid request format",
/tool call id was.*must be/i,
// Prefill-strict models (e.g. claude-opus-4-7) reject requests that end
// with an assistant turn. The lane must not re-queue these — the same
// payload will fail identically on every retry, causing an infinite loop
// (#79688).
"does not support assistant message prefill",
"conversation must end with a user message",
// Agent harness provider mismatch: the harness rejects the model because
// the provider id is not in its supported set. Retrying the same model
// will fail identically — classify so the fallback notice is informative
// instead of "unknown" (#91710).
/agent harness .* does not support .*provider is not one of/i,
],
} as const;
const BILLING_ERROR_HEAD_RE =
/^(?:error[:\s-]+)?billing(?:\s+error)?(?:[:\s-]+|$)|^(?:error[:\s-]+)?(?:credit balance|insufficient credits?|payment required|http\s*402\b)/i;
function matchesErrorPatterns(raw: string, patterns: readonly ErrorPattern[]): boolean {
if (!raw) {
return false;
}
const value = normalizeLowercaseStringOrEmpty(raw);
return patterns.some((pattern) =>
pattern instanceof RegExp ? pattern.test(value) : value.includes(pattern),
);
}
function matchesErrorPatternGroups(
raw: string,
groups: readonly (readonly ErrorPattern[])[],
): boolean {
return groups.some((patterns) => matchesErrorPatterns(raw, patterns));
}
export function matchesFormatErrorPattern(raw: string): boolean {
return matchesErrorPatterns(raw, ERROR_PATTERNS.format);
}
export function isRateLimitErrorMessage(raw: string): boolean {
return matchesErrorPatterns(raw, ERROR_PATTERNS.rateLimit);
}
export function isTimeoutErrorMessage(raw: string): boolean {
return matchesErrorPatterns(raw, ERROR_PATTERNS.timeout);
}
/**
* Provider stream completed with an explicit error finish/stop reason.
* These are not request timeouts: the transport finished quickly with a
* provider-side error. Keep them failover-eligible as server_error (#109218).
*/
const PROVIDER_COMPLETED_ERROR_FINISH_REASON_PATTERNS = [
/\bfinish_reason:\s*error\b/i,
/\bstop reason:\s*error\b/i,
/\bunhandled stop reason:\s*error\b/i,
// Symmetric with the timeout stop-reason family; scoped to the word `error`
// only so `reason: network_error` stays in the timeout lane.
/\breason:\s*error\b/i,
] as const satisfies readonly ErrorPattern[];
export function isProviderCompletedErrorFinishReasonMessage(raw: string): boolean {
return matchesErrorPatterns(raw, PROVIDER_COMPLETED_ERROR_FINISH_REASON_PATTERNS);
}
export function isPeriodicUsageLimitErrorMessage(raw: string): boolean {
return PERIODIC_USAGE_LIMIT_RE.test(raw);
}
export function isBillingErrorMessage(raw: string): boolean {
const value = normalizeLowercaseStringOrEmpty(raw);
if (!value) {
return false;
}
// Multi-section Markdown is explanatory content, not a provider error body.
// Without the former length cliff, examples discussing billing would otherwise match soft hints.
if ([...raw.matchAll(/(?:^|\n)##\s+\S/g)].length > 1) {
return false;
}
if (matchesErrorPatterns(value, ERROR_PATTERNS.billing)) {
return true;
}
if (!BILLING_ERROR_HEAD_RE.test(raw)) {
return false;
}
return (
value.includes("upgrade") ||
value.includes("credits") ||
value.includes("payment") ||
value.includes("purchase") ||
value.includes("subscription") ||
value.includes("plan")
);
}
export function isAuthPermanentErrorMessage(raw: string): boolean {
return matchesErrorPatternGroups(raw, [HIGH_CONFIDENCE_AUTH_PERMANENT_PATTERNS]);
}
export function isAuthErrorMessage(raw: string): boolean {
return matchesErrorPatternGroups(raw, [
AMBIGUOUS_AUTH_ERROR_PATTERNS,
COMMON_AUTH_ERROR_PATTERNS,
ZAI_AUTH_ERROR_PATTERNS,
CJK_AUTH_ERROR_PATTERNS,
]);
}
export function isOverloadedErrorMessage(raw: string): boolean {
return matchesErrorPatterns(raw, ERROR_PATTERNS.overloaded);
}
export function isServerErrorMessage(raw: string): boolean {
const value = normalizeLowercaseStringOrEmpty(raw);
if (!value) {
return false;
}
if (STATUS_INTERNAL_SERVER_ERROR_WITH_500_RE.test(value) || HTTP_5XX_STATUS_RE.test(value)) {
return true;
}
const scrubbed = value.replace(STATUS_INTERNAL_SERVER_ERROR_RE, "").trim();
if (scrubbed === "") {
return true;
}
return matchesErrorPatterns(scrubbed, ERROR_PATTERNS.serverError);
}