File size: 7,872 Bytes
67d18ac | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 | server:
port: 8080
host: "0.0.0.0"
auth:
# Key that clients must provide to use the proxy.
# Set to null/omit to disable client auth.
proxyApiKey: "your-proxy-secret"
# Upstream credentials come from the OAuth login flow β run this first:
# bun run src/index.ts auth login <zai|bigmodel>
# Parsed but currently not honored β the credential store path is fixed at
# ~/.zcode-proxy/credentials.json:
# oauthCredentialsPath: "~/.zcode-proxy/credentials.json"
# Which upstream provider to use: "zai" or "bigmodel"
provider: zai
# Which plan tier to use:
# "coding-plan" (default) β direct upstream endpoints, permanent API key
# "start-plan" β routes through zcode.z.ai with JWT auth (requires `auth login`)
plan: coding-plan
providers:
zai:
anthropicBase: "https://api.z.ai/api/anthropic"
openaiBase: "https://api.z.ai/api/coding/paas/v4"
bigmodel:
anthropicBase: "https://open.bigmodel.cn/api/anthropic"
openaiBase: "https://open.bigmodel.cn/api/coding/paas/v4"
defaultModel: glm-4.6
models:
- glm-4.5-air
- glm-4.6
- glm-4.6v
- glm-4.7
- glm-5
- glm-5-turbo
- glm-5v-turbo
- glm-5.1
- glm-5.2
- glm-5.3
- glm-5.3-flash
# Configurable identity headers injected on every upstream request to mimic the
# ZCode desktop client (User-Agent, X-ZCode-App-Version, X-Title,
# X-ZCode-Agent, HTTP-Referer). Runtime platform headers (X-Platform,
# X-Os-Category, X-Os-Version) are detected dynamically and are not configured
# here. All fields below are optional; env vars override YAML, which overrides
# defaults.
identity:
# Mirrors process.env.ZCODE_APP_VERSION in the ZCode bundle.
# Must be printable ASCII; non-conforming values fall back to the default.
# Default: "3.11.2" (current ZCode release). Override to match your real client.
appVersion: "3.11.2"
# X-Title suffix β "Z Code@{sourceTitle}". Default "cli".
sourceTitle: "cli"
# HTTP-Referer URL. Default "https://zcode.z.ai".
refererOrigin: "https://zcode.z.ai"
# Device identity (X-Device-Mid) β random UUIDv4, generated ONCE and reused
# forever (mirrors ZCode's telemetry deviceMid; no hardware values involved).
# Auto-generated into this file at first `auth login` or config creation.
# Leave empty on Android β the app injects ZCODE_IDENTITY_DEVICE_MID instead.
deviceMid: ""
# Local client-session inference for cache-affinity experiments.
# "observe" (default) logs inferred sessions in debug mode but does not change
# upstream x-session-id. "enforce" reuses a stable x-session-id for inferred
# coding-plan sessions. "off" disables inference entirely.
clientIdentity:
mode: observe
ttlSeconds: 900
maxSessions: 1024
# Responses API (/v1/responses) β Codex CLI / OpenAI Agents SDK endpoint.
# Translates Responses requests through Chat Completions to the Anthropic-format
# upstream (both plans post Anthropic upstream).
# State (previous_response_id) is held in-memory (cleared on restart, 24h TTL).
responses:
enabled: true
store:
maxEntries: 1000
ttlMs: 86400000
# GLM MCP hosted-tool integration β NOT currently wired into production.
# These keys are parsed but have no effect: hosted tools (`mcp`,
# `web_search_preview`, ...) are silently stripped by the Responses
# translator, and the MCP client subsystem (src/mcp) is retained for a
# future function-injection design. Endpoints are derived from `provider`:
# zai β https://api.z.ai/api/mcp/{tool}/mcp
# bigmodel β https://open.bigmodel.cn/api/mcp/{tool}/mcp
# Auth reuses the logged-in upstream credential β no extra key needed.
mcp:
enabled: true
# (planned) Intercept `web_search`/`web_search_preview` hosted tools via
# GLM's `web_search_prime` MCP.
webSearch: true
# (planned) Inject `webReader` as a function tool the model can call directly.
webReader: false
# (planned) Inject the three `zread` tools (search_doc/get_repo_structure/read_file).
zread: false
# Async (off-peak / "idle plan") bridge.
# When enabled, exposes POST /async/v1/messages and /async/v1/chat/completions
# that route through ZCode's off-peak ticket-queue backend β free compute when
# the GLM cluster has idle capacity. The proxy holds the connection open with
# SSE keepalives during queue wait, forwards the LLM stream once a ticket is
# ready, and auto-retries on ticket-expired (up to maxRetries).
#
# IMPORTANT: off-peak needs the JWT captured by `auth login`. A credential
# without a JWT makes the route return 400 `async_credentials_unavailable`.
# Off-peak is a coding-plan feature: when `plan: start-plan`, the /async/*
# routes return 400 `async_plan_unsupported` even when enabled: true.
#
# Env overrides: ZCODE_ASYNC_ENABLED, ZCODE_ASYNC_ORIGIN,
# ZCODE_ASYNC_MAX_RETRIES, ZCODE_ASYNC_MAX_WAIT_MS.
async:
enabled: false
origin: "https://zcode.z.ai"
pollIntervalMs: 5000
keepAliveIntervalMs: 3000
# Maximum total wait for ticket to become ready, in ms. 0 = unlimited.
maxWaitMs: 0
maxRetries: 3
settleTimeoutMs: 8000
controlTimeoutMs: 15000
# Optional model override; empty uses the request's `model`.
defaultModel: ""
# Manual claim ("weekend plan") β mirrors the ZCode 3.10 desktop client's
# manualClaimPlan feature. When enabled, the proxy periodically lists claimable
# trial plans ({origin}/api/v1/zcode-plan/billing/preview) and auto-claims the
# highest-priority one ({origin}/api/v1/zcode-plan/billing/claim) using the
# OAuth JWT + an Aliyun captcha token (same solver as start-plan). Weekend
# campaigns are quota-limited first-come-first-served; claimed plans activate
# at the campaign's `starts_at` and land as Start-Plan quota.
#
# IMPORTANT: requires a logged-in credential (claim uses the JWT from
# `auth login`), and identity.appVersion must be >= the campaign's minimum
# client version (default 3.11.2) or the server rejects with `ineligible`.
#
# Env overrides: ZCODE_CLAIM_ENABLED, ZCODE_CLAIM_AUTO, ZCODE_CLAIM_ORIGIN,
# ZCODE_CLAIM_POLL_INTERVAL_MS.
claim:
# On by default: the scheduler polls the preview endpoint every 5 min and
# idles until a campaign goes live (404 preview = nothing to claim).
# `false` = CLI-only (`zcode-proxy claim [list|now]` still works).
enabled: true
# Auto-claim in the background while serving. `false` = CLI-only
# (`zcode-proxy claim` still works when `enabled: true`).
auto: true
origin: "https://zcode.z.ai"
# Preview poll cadence (5 min default).
pollIntervalMs: 300000
# Backoff after failed attempts (10 min default); quota_exhausted /
# already_claimed back off to the server-provided next window instead.
cooldownMs: 600000
# Optional: claim a specific plan_id. Empty = highest-priority preview.
planId: ""
# Server-controlled upstream URL remapping (mirrors the ZCode client's
# ProviderEndpointRoutingService). The proxy periodically fetches
# {origin}/api/v1/agent/configs and rewrites matching upstream URLs per the
# returned proxyEndpoint.mapping table (currently the coding-plan Anthropic
# endpoints -> zcode.z.ai ultra endpoints). Fail-open: any fetch/parse error
# keeps the original URLs. Env override: ZCODE_ENDPOINT_ROUTING=false.
endpointRouting:
enabled: true
origin: "https://zcode.z.ai"
# Client request signing V4 (mirrors the ZCode 3.9.1 ClientRequestSigningV4Signer).
# When enabled, the proxy probes {origin}/api/v1/agent/configs (cached 1h) and,
# only if the server sets data.codingPlanSignature.enable=true, signs coding-plan
# requests: handshake against {provider}/api/paas/c1f3a7e2/v2/client, Ed25519
# signature + proof-of-work headers on every request, fail-open retry ladder
# (two 401 VERIFY rejections -> permanent unsigned bypass). Start-plan and
# off-peak paths are never signed. Env override: ZCODE_CLIENT_SIGNING=false.
clientSigning:
enabled: true
origin: "https://zcode.z.ai"
logging:
level: info
|