File size: 9,334 Bytes
67d18ac | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 | /**
* Configuration types for zcode-proxy.
* @see .omo/plans/zcode-proxy.md Task 2
*/
/** Provider endpoint configuration (one per upstream provider). */
export interface ProviderEndpoints {
/** Base URL for Anthropic-format API, e.g. "https://api.z.ai/api/anthropic". */
anthropicBase: string;
/** Base URL for OpenAI-format API, e.g. "https://api.z.ai/api/coding/paas/v4". */
openaiBase: string;
}
/** Auth section of the proxy configuration. */
interface AuthConfig {
/**
* Key that clients must provide to use the proxy (via `Authorization: Bearer {proxyApiKey}`).
* If unset, the proxy does not require client auth.
*/
proxyApiKey?: string;
/** Path to stored OAuth credentials created by `auth login`. */
oauthCredentialsPath?: string;
}
/**
* Identity headers injected on every upstream request to mimic the ZCode
* desktop client. Mirrors the `pio` builder in the reverse-engineered bundle
* (`_reverse/zcode.cjs`); see `_reverse/NOTEPAD.md` "How Credential is Used".
*
* Resolution: env var (matches ZCode's own convention) β YAML override β default.
* `appVersion` must be printable ASCII (`/^[\x20-\x7e]+$/`); non-conforming
* values are silently dropped and fall back to the default (current ZCode
* release), exactly like `fio` in the bundle.
*/
export interface ProxyIdentity {
appVersion: string;
sourceTitle: string;
refererOrigin: string;
/**
* Device identity for `X-Device-Mid` (mirrors ZCode's telemetry deviceMid:
* a random UUIDv4 generated ONCE and reused forever β no hardware values).
* Empty/undefined omits the header. Desktop: persisted in config.yaml
* (`ensureDeviceMidInConfig`). Android: injected via the
* `ZCODE_IDENTITY_DEVICE_MID` env var (NodeRunner, app-private file) β env
* wins over YAML. Must stay stable per anti-pattern #13; never randomize
* per-request.
*/
deviceMid?: string;
}
/** Local client-session inference mode for upstream session affinity. */
export interface ClientIdentityConfig {
/** "observe" logs/instruments only; "enforce" reuses upstream x-session-id; "off" disables inference. */
mode: "off" | "observe" | "enforce";
/** In-memory session TTL in seconds. */
ttlSeconds: number;
/** Maximum number of inferred sessions retained in memory. */
maxSessions: number;
}
/**
* Responses-API (`/v1/responses`) configuration. When `enabled`, the proxy
* translates Codex-style Responses requests to the GLM Chat Completions upstream.
*/
export interface ResponsesConfig {
/** Enable the `/v1/responses` route. Default `true`. */
enabled: boolean;
/** Max stored responses (LRU). Default 1000. */
storeMaxEntries: number;
/** Stored-response TTL in ms. Default 24h (in-memory; cleared on restart). */
storeTtlMs: number;
}
/** GLM MCP hosted-tool configuration. Endpoints are derived from the active provider. */
export interface McpConfig {
/** Enable MCP interception (web_search) and function-tool injection (web_reader/zread). Default `true`. */
enabled: boolean;
/** Intercept `web_search` / `web_search_preview` hosted tools via GLM `web_search_prime` MCP. Default `true`. */
webSearch: boolean;
/** Inject `webReader` as a function tool the model can call. Default `false` (off by default to limit scope). */
webReader: boolean;
/** Inject the three `zread` tools as function tools. Default `false`. */
zread: boolean;
}
/**
* Async (off-peak / idle-plan) bridge configuration. When `enabled`, exposes
* `/async/v1/messages` and `/async/v1/chat/completions` that route to ZCode's
* off-peak ticket-queue backend. The proxy keeps the client connection alive
* with SSE comments during ticket-queue wait, forwards the LLM stream once
* the ticket is `ready`, and auto-retries on ticket-expired (up to `maxRetries`).
*
* Requires a logged-in oauth credential (off-peak needs both
* `Authorization: Bearer ${jwt}` and `X-Coding-Plan-Api-Key` headers). A
* credential lacking the JWT makes the route entry return 400
* `async_credentials_unavailable`.
*
* @see _reverse/NOTEPAD.md "Off-Peak / Idle Plan" section for full upstream protocol.
*/
export interface AsyncConfig {
/** Enable the `/async/*` routes. Default `false`. */
enabled: boolean;
/** Base origin for off-peak endpoints. Default `"https://zcode.z.ai"`. */
origin: string;
/** Ticket-status poll interval in ms. Default `5000`. */
pollIntervalMs: number;
/** SSE keepalive comment interval during ticket-queue wait, in ms. Default `3000`. */
keepAliveIntervalMs: number;
/** Maximum total wait time for a ticket to become `ready`, in ms. `0` = unlimited. Default `0`. */
maxWaitMs: number;
/** Maximum auto-retry count on `off-peak-ticket-expired`. Default `3`. */
maxRetries: number;
/** Settle call timeout in ms (best-effort close-out on completion/abort). Default `8000`. */
settleTimeoutMs: number;
/** Control-plane call (takeTicket/pollStatus) timeout in ms. Default `15000`. */
controlTimeoutMs: number;
/** Optional model override; empty string uses the request's `model`. Default `""`. */
defaultModel: string;
}
/**
* Manual claim ("weekend plan") β mirrors the ZCode 3.10 desktop client's
* `manualClaimPlan` feature: periodically list claimable trial plans
* (`GET {origin}/api/v1/zcode-plan/billing/preview`) and, when one is
* available, claim it (`POST {origin}/api/v1/zcode-plan/billing/claim`) with
* the OAuth JWT and an Aliyun captcha token. Claimed plans grant Start-Plan
* style quota with delayed activation (`effective_at` / `starts_at`).
*
* Requires `auth.mode: oauth` (claim uses `Authorization: Bearer ${jwt}`).
*
* @see _reverse/NOTEPAD.md "Manual Claim Plan" section for the protocol.
*/
export interface ClaimConfig {
/** Enable the claim subsystem (CLI `claim` command + auto scheduler). Default `true`. */
enabled: boolean;
/** Auto-claim in the background while the proxy is serving. Default `true` (effective when `enabled`). */
auto: boolean;
/** Base origin of the zcode-plan billing endpoints. Default `"https://zcode.z.ai"`. */
origin: string;
/** Preview poll interval in ms. Default `300000` (5 min). */
pollIntervalMs: number;
/** Backoff after a failed claim attempt in ms. Default `600000` (10 min). */
cooldownMs: number;
/** Optional `plan_id` to claim; empty string claims the highest-priority preview. Default `""`. */
planId: string;
}
/**
* Provider endpoint routing β mirrors the ZCode client's
* `ProviderEndpointRoutingService`: periodically fetch
* `GET {configUrl}/api/v1/agent/configs` and rewrite matching upstream URLs
* per the server-controlled `data.proxyEndpoint.mapping` table. As of
* 2026-08-19 only the coding-plan Anthropic endpoints are mapped (to
* `zcode.z.ai/api/v1/ultra[-zai]/...`); resolution is generic so future
* entries apply automatically. Always fail-open.
*/
export interface EndpointRoutingConfig {
/** Enable URL remapping. Default `true`. */
enabled: boolean;
/** Base origin of the agent-configs endpoint. Default `"https://zcode.z.ai"`. */
origin: string;
}
/**
* Client request signing V4 β mirrors the ZCode 3.9.1
* `ClientRequestSigningV4Signer`. When enabled, the proxy probes the same
* feature gate the client uses (`GET {origin}/api/v1/agent/configs` β
* `data.codingPlanSignature.enable`) and, only if the server turns the feature
* on, signs coding-plan upstream requests (handshake + Ed25519 + proof-of-work,
* with the client's fail-open retry ladder). Start-plan and off-peak paths are
* permanently exempt.
*/
export interface ClientSigningConfig {
/** Enable gate probing + signing. Default `true`. */
enabled: boolean;
/** Base origin of the feature-gate endpoint. Default `"https://zcode.z.ai"`. */
origin: string;
}
/** Top-level proxy configuration. */
export interface ProxyConfig {
server: {
port: number;
host: string;
};
auth: AuthConfig;
/** Active upstream provider. */
provider: "zai" | "bigmodel";
/** Which plan tier to use. "coding-plan" (default) uses direct upstream endpoints; "start-plan" routes through zcode.z.ai with JWT auth. */
plan: "coding-plan" | "start-plan";
/** Per-provider endpoint overrides. */
providers: {
zai: ProviderEndpoints;
bigmodel: ProviderEndpoints;
};
/** Default model id used when client request omits `model`. */
defaultModel: string;
/** Whitelist of allowed model ids. */
models: string[];
/**
* Identity headers injected upstream. Always present after `loadConfig`;
* defaults mirror the production ZCode desktop client.
*/
identity: ProxyIdentity;
/** Local client session inference for cache-affinity experiments. */
clientIdentity: ClientIdentityConfig;
/** Responses-API (`/v1/responses`) configuration. */
responses: ResponsesConfig;
/** Server-controlled upstream URL remapping (ultra endpoints). */
endpointRouting: EndpointRoutingConfig;
/** Client request signing V4 (Ed25519 + PoW, gate-driven). */
clientSigning: ClientSigningConfig;
/** GLM MCP hosted-tool configuration. */
mcp: McpConfig;
/** Async (off-peak / idle-plan) bridge configuration. */
async: AsyncConfig;
/** Manual claim ("weekend plan") configuration. */
claim: ClaimConfig;
logging: {
level: "debug" | "info" | "warn" | "error";
};
}
|