File size: 7,872 Bytes
67d18ac
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
server:
  port: 8080
  host: "0.0.0.0"

auth:
  # Key that clients must provide to use the proxy.
  # Set to null/omit to disable client auth.
  proxyApiKey: "your-proxy-secret"

  # Upstream credentials come from the OAuth login flow β€” run this first:
  #   bun run src/index.ts auth login <zai|bigmodel>
  # Parsed but currently not honored β€” the credential store path is fixed at
  # ~/.zcode-proxy/credentials.json:
  # oauthCredentialsPath: "~/.zcode-proxy/credentials.json"

# Which upstream provider to use: "zai" or "bigmodel"
provider: zai

# Which plan tier to use:
#   "coding-plan" (default) β€” direct upstream endpoints, permanent API key
#   "start-plan"            β€” routes through zcode.z.ai with JWT auth (requires `auth login`)
plan: coding-plan

providers:
  zai:
    anthropicBase: "https://api.z.ai/api/anthropic"
    openaiBase: "https://api.z.ai/api/coding/paas/v4"
  bigmodel:
    anthropicBase: "https://open.bigmodel.cn/api/anthropic"
    openaiBase: "https://open.bigmodel.cn/api/coding/paas/v4"

defaultModel: glm-4.6

models:
  - glm-4.5-air
  - glm-4.6
  - glm-4.6v
  - glm-4.7
  - glm-5
  - glm-5-turbo
  - glm-5v-turbo
  - glm-5.1
  - glm-5.2
  - glm-5.3
  - glm-5.3-flash

# Configurable identity headers injected on every upstream request to mimic the
# ZCode desktop client (User-Agent, X-ZCode-App-Version, X-Title,
# X-ZCode-Agent, HTTP-Referer). Runtime platform headers (X-Platform,
# X-Os-Category, X-Os-Version) are detected dynamically and are not configured
# here. All fields below are optional; env vars override YAML, which overrides
# defaults.
identity:
  # Mirrors process.env.ZCODE_APP_VERSION in the ZCode bundle.
  # Must be printable ASCII; non-conforming values fall back to the default.
  # Default: "3.11.2" (current ZCode release). Override to match your real client.
  appVersion: "3.11.2"
  # X-Title suffix β†’ "Z Code@{sourceTitle}". Default "cli".
  sourceTitle: "cli"
  # HTTP-Referer URL. Default "https://zcode.z.ai".
  refererOrigin: "https://zcode.z.ai"
  # Device identity (X-Device-Mid) β€” random UUIDv4, generated ONCE and reused
  # forever (mirrors ZCode's telemetry deviceMid; no hardware values involved).
  # Auto-generated into this file at first `auth login` or config creation.
  # Leave empty on Android β€” the app injects ZCODE_IDENTITY_DEVICE_MID instead.
  deviceMid: ""

# Local client-session inference for cache-affinity experiments.
# "observe" (default) logs inferred sessions in debug mode but does not change
# upstream x-session-id. "enforce" reuses a stable x-session-id for inferred
# coding-plan sessions. "off" disables inference entirely.
clientIdentity:
  mode: observe
  ttlSeconds: 900
  maxSessions: 1024

# Responses API (/v1/responses) β€” Codex CLI / OpenAI Agents SDK endpoint.
# Translates Responses requests through Chat Completions to the Anthropic-format
# upstream (both plans post Anthropic upstream).
# State (previous_response_id) is held in-memory (cleared on restart, 24h TTL).
responses:
  enabled: true
  store:
    maxEntries: 1000
    ttlMs: 86400000

# GLM MCP hosted-tool integration β€” NOT currently wired into production.
# These keys are parsed but have no effect: hosted tools (`mcp`,
# `web_search_preview`, ...) are silently stripped by the Responses
# translator, and the MCP client subsystem (src/mcp) is retained for a
# future function-injection design. Endpoints are derived from `provider`:
#   zai      β†’ https://api.z.ai/api/mcp/{tool}/mcp
#   bigmodel β†’ https://open.bigmodel.cn/api/mcp/{tool}/mcp
# Auth reuses the logged-in upstream credential β€” no extra key needed.
mcp:
  enabled: true
  # (planned) Intercept `web_search`/`web_search_preview` hosted tools via
  # GLM's `web_search_prime` MCP.
  webSearch: true
  # (planned) Inject `webReader` as a function tool the model can call directly.
  webReader: false
  # (planned) Inject the three `zread` tools (search_doc/get_repo_structure/read_file).
  zread: false

# Async (off-peak / "idle plan") bridge.
# When enabled, exposes POST /async/v1/messages and /async/v1/chat/completions
# that route through ZCode's off-peak ticket-queue backend β€” free compute when
# the GLM cluster has idle capacity. The proxy holds the connection open with
# SSE keepalives during queue wait, forwards the LLM stream once a ticket is
# ready, and auto-retries on ticket-expired (up to maxRetries).
#
# IMPORTANT: off-peak needs the JWT captured by `auth login`. A credential
# without a JWT makes the route return 400 `async_credentials_unavailable`.
# Off-peak is a coding-plan feature: when `plan: start-plan`, the /async/*
# routes return 400 `async_plan_unsupported` even when enabled: true.
#
# Env overrides: ZCODE_ASYNC_ENABLED, ZCODE_ASYNC_ORIGIN,
# ZCODE_ASYNC_MAX_RETRIES, ZCODE_ASYNC_MAX_WAIT_MS.
async:
  enabled: false
  origin: "https://zcode.z.ai"
  pollIntervalMs: 5000
  keepAliveIntervalMs: 3000
  # Maximum total wait for ticket to become ready, in ms. 0 = unlimited.
  maxWaitMs: 0
  maxRetries: 3
  settleTimeoutMs: 8000
  controlTimeoutMs: 15000
  # Optional model override; empty uses the request's `model`.
  defaultModel: ""

# Manual claim ("weekend plan") β€” mirrors the ZCode 3.10 desktop client's
# manualClaimPlan feature. When enabled, the proxy periodically lists claimable
# trial plans ({origin}/api/v1/zcode-plan/billing/preview) and auto-claims the
# highest-priority one ({origin}/api/v1/zcode-plan/billing/claim) using the
# OAuth JWT + an Aliyun captcha token (same solver as start-plan). Weekend
# campaigns are quota-limited first-come-first-served; claimed plans activate
# at the campaign's `starts_at` and land as Start-Plan quota.
#
# IMPORTANT: requires a logged-in credential (claim uses the JWT from
# `auth login`), and identity.appVersion must be >= the campaign's minimum
# client version (default 3.11.2) or the server rejects with `ineligible`.
#
# Env overrides: ZCODE_CLAIM_ENABLED, ZCODE_CLAIM_AUTO, ZCODE_CLAIM_ORIGIN,
# ZCODE_CLAIM_POLL_INTERVAL_MS.
claim:
  # On by default: the scheduler polls the preview endpoint every 5 min and
  # idles until a campaign goes live (404 preview = nothing to claim).
  # `false` = CLI-only (`zcode-proxy claim [list|now]` still works).
  enabled: true
  # Auto-claim in the background while serving. `false` = CLI-only
  # (`zcode-proxy claim` still works when `enabled: true`).
  auto: true
  origin: "https://zcode.z.ai"
  # Preview poll cadence (5 min default).
  pollIntervalMs: 300000
  # Backoff after failed attempts (10 min default); quota_exhausted /
  # already_claimed back off to the server-provided next window instead.
  cooldownMs: 600000
  # Optional: claim a specific plan_id. Empty = highest-priority preview.
  planId: ""

# Server-controlled upstream URL remapping (mirrors the ZCode client's
# ProviderEndpointRoutingService). The proxy periodically fetches
# {origin}/api/v1/agent/configs and rewrites matching upstream URLs per the
# returned proxyEndpoint.mapping table (currently the coding-plan Anthropic
# endpoints -> zcode.z.ai ultra endpoints). Fail-open: any fetch/parse error
# keeps the original URLs. Env override: ZCODE_ENDPOINT_ROUTING=false.
endpointRouting:
  enabled: true
  origin: "https://zcode.z.ai"

# Client request signing V4 (mirrors the ZCode 3.9.1 ClientRequestSigningV4Signer).
# When enabled, the proxy probes {origin}/api/v1/agent/configs (cached 1h) and,
# only if the server sets data.codingPlanSignature.enable=true, signs coding-plan
# requests: handshake against {provider}/api/paas/c1f3a7e2/v2/client, Ed25519
# signature + proof-of-work headers on every request, fail-open retry ladder
# (two 401 VERIFY rejections -> permanent unsigned bypass). Start-plan and
# off-peak paths are never signed. Env override: ZCODE_CLIENT_SIGNING=false.
clientSigning:
  enabled: true
  origin: "https://zcode.z.ai"

logging:
  level: info