From fc5d72b14382e1965136b2a4d87da9ee658d547a Mon Sep 17 00:00:00 2001 From: xelr233 Date: Thu, 10 Sep 2026 22:39:05 +0800 Subject: [PATCH 01/21] =?UTF-8?q?feat:=20=E4=B8=8A=E6=B8=B8=20User-Agent:?= =?UTF-8?q?=20cli=20+=20threadId=20=E4=B8=8E=20x-session-id=20=E5=90=8C?= =?UTF-8?q?=E5=80=BC?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 仅两处改动,对齐官方 cmd CLI 1.53.0 的线上行为(源码 + 抓包双重确认): 1. /alpha/generate 请求补上 User-Agent: cli CLI buildCommandAuthHeaders 里该头取自常量 vy = "cli"。 MAXeaglet 此前完全不发 User-Agent。 2. 信封顶层补上真实 threadId,取值与 x-session-id 恒等 CLI createModelClient 实际发送 body.threadId === x-session-id。 MAXeaglet 原有的 newThreadId() 是死代码(495 行算完从未进入请求体)。 - 客户端提供 session 头 → threadId 取该值 - 客户端未提供 session 头 → 仍走原有 per-key 12h 会话,threadId 取同一值(该路径行为不变) - sessionId 非合法 UUID → 省略该字段,对应 CLI toWireThreadId 语义 (uuid.safeParse 失败即返回 undefined,键被 JSON.stringify 丢弃) threadId 在信封中占位于 permissionMode 与 params 之间,保持与 CLI 一致的键序。 已实测(本地 mock 上游): - 两个端点(/v1/chat/completions、/v1/messages)UA 均为 cli,threadId 均等于 x-session-id - 客户端未发 session 头时,threadId 与生成的 x-session-id 相等 - 客户端发非 UUID 时,threadId 键整体消失 --- proxy.mjs | 22 ++++++++++++++++++---- 1 file changed, 18 insertions(+), 4 deletions(-) diff --git a/proxy.mjs b/proxy.mjs index 70def13..486bdcc 100644 --- a/proxy.mjs +++ b/proxy.mjs @@ -268,8 +268,13 @@ function getSessionId(incomingHeaders, apiKey, promptCacheKey) { return ensureSession(apiKey); } -// 每个请求独立 thread ID -function newThreadId() { return randomUUID(); } +// CC 线上信封的 threadId 只接受合法 UUID —— 对应 CLI 的 toWireThreadId: +// uuid.safeParse 失败即返回 undefined,该键随之被 JSON.stringify 丢弃。 +// 非 UUID 的 session 值一律不发,避免上游看到真 CLI 永远不会产生的 threadId。 +function isWireUuid(v) { + return typeof v === 'string' + && /^[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}$/i.test(v); +} // ── 每 Key 独立状态(fingerprint + 初始化节流) ── // 每个 API Key 拥有自己的设备指纹和初始化定时器 @@ -538,8 +543,6 @@ function buildCcRequest(openaiReq) { if (cacheBoundary) cacheBoundary.cache_control = { type: 'ephemeral' }; } - const threadId = newThreadId(); - const body = { config: { workingDir: process.cwd(), @@ -556,6 +559,10 @@ function buildCcRequest(openaiReq) { taste: null, skills: '', permissionMode: 'standard', + // threadId 由 forwardToCC 解析出 sessionId 后回填为 x-session-id 同值。 + // 先在此占位以固定 JSON 键序,与 CLI 线上顺序一致(…permissionMode, threadId, params); + // 值为 undefined 时 JSON.stringify 会整体丢弃该键。 + threadId: undefined, params: { model: model || 'deepseek/deepseek-v4-flash', messages: ccMessages, @@ -964,8 +971,15 @@ async function forwardToCC(body, apiKey, incomingHeaders = {}, signal, promptCac const traceparent = generateTraceparent(); const sessionId = getSessionId(incomingHeaders, apiKey, promptCacheKey); + // generate body 顶层 threadId 必须与 x-session-id 同值(CLI 抓包与源码双重确认)。 + // 客户端未提供 session 头时,sessionId 仍来自 MAXeaglet 原有的 per-key 12h 会话 + // (此路径行为完全不变),threadId 取同一个值,故两者恒等; + // 仅当 sessionId 不是合法 UUID 时省略该字段 —— 即 CLI toWireThreadId 的既有行为。 + body.threadId = isWireUuid(sessionId) ? sessionId : undefined; + const headers = { 'Content-Type': 'application/json', + 'User-Agent': 'cli', 'Authorization': `Bearer ${apiKey}`, 'x-cli-environment': 'production', 'x-command-code-version': CC_VERSION, From b0e7cc4d9e48132ef478a5abfb14861f6333163b Mon Sep 17 00:00:00 2001 From: xelr233 Date: Thu, 10 Sep 2026 22:57:13 +0800 Subject: [PATCH 02/21] =?UTF-8?q?feat:=20issue=20#18=20=E4=B8=8A=E6=B8=B8?= =?UTF-8?q?=20HTTP(S)=20=E4=BB=A3=E7=90=86=20+=20#20=20=E8=AF=B7=E6=B1=82?= =?UTF-8?q?=E6=A0=91=E6=8F=90=E5=89=8D=E9=87=8A=E6=94=BE?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 说明:本分支原在此提交一并实现了 CC_MAX_INFLIGHT,rebase 到上游 c443889 后该项已由上游实现且更完整 —— 上游版在 server 入口统一准入、/health 与 / 豁免、finish+close 双事件幂等释放。本提交已把本地那套全部撤下,只保留上游 实现;本分支也不再提供 config.json 的 maxInflight(与 CC_MAX_BODY_MB / CC_STREAM_IDLE_MS 等调参项一致,只用环境变量)。 Finding 1(响应背压)已由上游 88a1872 修复;Finding 2(请求树副本)由本提交 补齐;全局在途上限由上游 c443889 补齐。 - buildCcRequest 之后释放原始请求树 - /v1/chat/completions:取出 prompt_cache_key 后置空 openaiReq - /v1/messages:置空 openaiReq 与 anthropicReq 原先这两棵树会和 ccBody 一起活到整段请求结束。 新增 upstreamProxy / CC_UPSTREAM_PROXY,让发往 CC 上游的请求走本地 HTTP 代理(出口地区调整 / 风控 403 的 IP 维度对照)。 覆盖 /alpha/generate、/alpha/fingerprint/record、/alpha/lifecycle-events、 /provider/v1/models;不影响本地监听、/health 与 npm 版本检查。 实现为零依赖:自建 CONNECT 隧道(代理只做裸字节转发),再用 node:https 复用同一 socket,TLS 端到端、证书按目标主机名校验。因此不需要 undici / https-proxy-agent,engines >=18 即可用(Node 原生 fetch 不读 HTTPS_PROXY; 官方环境变量路线需 Node >= 22.21/24.5 + NODE_USE_ENV_PROXY=1)。 预请求刻意也走代理:若它们直连而上游生成走代理,同一账号会从两个不同 IP 注册,正是该 issue 想消除的矛盾。 - 默认路径:UA=cli、threadId===x-session-id、恰好 1 条 generate - 配代理:CONNECT 隧道建立、generate 与两个预请求均经隧道、/health 不经 代理、UA 仍为 cli - 两个端点(OpenAI/Anthropic)无回归,Anthropic 端点经代理亦正常 --- README.md | 30 +++++++++-- README_zh.md | 30 +++++++++-- config.json | 3 +- proxy.mjs | 150 +++++++++++++++++++++++++++++++++++++++++++++++---- 4 files changed, 193 insertions(+), 20 deletions(-) diff --git a/README.md b/README.md index 6e17dee..790b0d5 100644 --- a/README.md +++ b/README.md @@ -76,6 +76,7 @@ commandcode/ | `CC_NONSTREAM_IDLE_MS` | Non-streaming upstream read idle timeout (default `90000`) | | `CC_MAX_INFLIGHT` | In-process concurrent request cap (default `0` = unlimited) | | `CMD_ZDR` | `zdr` (`1` to enable) | +| `CC_UPSTREAM_PROXY` | `upstreamProxy` | When enabled, the proxy sends `x-cmd-zdr: 1` on Command Code generation requests and the fingerprint/lifecycle initialization requests. It does not add the header @@ -83,9 +84,29 @@ to the npm version check or the proxy's `/provider/v1/models` catalog request. This requests Command Code's ZDR-only routing; the upstream service remains the authority for actual retention and provider availability. -**Request body limit**: independent of `config.json` — requests larger than **100 MB** are rejected with `HTTP 413` (the connection is kept alive and drained, not reset). Override with `CC_MAX_BODY_MB` (positive integer, unit: MB). +**Request body limit**: independent of `config.json` — requests larger than **8 MB** are rejected with `HTTP 413` (the connection is kept alive and drained, not reset). Override with `CC_MAX_BODY_MB` (positive integer, unit: MB). The default was lowered from 100 MB ([#20](https://github.com/MAXeaglet/commandcode-proxy/issues/20)). -> ⚠️ **Memory amplification**: a request body exists in several copies before it reaches upstream; measured peak ≈ body size × **5.1–7.4** (7 MB → +52 MB, 20 MB → +116 MB, while a request rejected with `413` costs only ×1.05). The default `CC_MAX_BODY_MB=100` therefore implies up to ~550 MB for a **single** request, and that limit is per-request, not global. See [Memory & Deployment](#memory--deployment). +> ⚠️ **Memory amplification**: a request body exists in several copies before it reaches upstream; measured peak ≈ body size × **5.1–7.4** (7 MB → +52 MB, 20 MB → +116 MB, while a request rejected with `413` costs only ×1.05). The old default `CC_MAX_BODY_MB=100` implied up to ~550 MB for a **single** request, which is not a sane default for the 1-core VPS the Dockerfile targets. The default is now **8 MB** (~45 MB peak per request). The limit is still per-request, not global — cap concurrency with `CC_MAX_INFLIGHT` or at the reverse proxy. See [Memory & Deployment](#memory--deployment). + +### Upstream proxy (`upstreamProxy` / `CC_UPSTREAM_PROXY`) + +Route the requests the proxy makes **to Command Code** through a local HTTP proxy — for egress-region switching, or for comparing IPs when debugging risk-control `403`s. + +```json +{ "upstreamProxy": "http://127.0.0.1:7890" } +``` + +```bash +CC_UPSTREAM_PROXY=http://127.0.0.1:7890 npm start +``` + +- Applies to `/alpha/generate`, `/alpha/fingerprint/record`, `/alpha/lifecycle-events` and `/provider/v1/models`. +- **Does not** touch the local listener, `/health`, or the npm version check. +- Only `http://` (CONNECT) proxies are supported. Implemented with a plain CONNECT tunnel plus `node:https`, so there is **no new dependency** and it works on Node 18+. +- Each upstream request opens its own tunnel connection. TLS is end-to-end: the certificate is validated against `api.commandcode.ai`, never against the proxy. +- Routing the fingerprint/lifecycle pre-requests through the same proxy matters: if they went out direct while generation went through the proxy, one account would register from two different IPs — exactly the inconsistency you are trying to avoid. + +> Node's built-in `fetch` does **not** read `HTTPS_PROXY`/`HTTP_PROXY`. The official env-var route requires Node ≥ 22.21 / 24.5 plus `NODE_USE_ENV_PROXY=1`; this option works without either. ## API Endpoints @@ -468,7 +489,8 @@ npm run docker:build:multi |----------|---------|-------------| | `PORT` | `3050` | Container listen port | | `PROXY_PORT` | `3050` | Host port (compose only) | -| `CC_MAX_BODY_MB` | `100` | Max request body size in MB; oversized requests are rejected with `HTTP 413` | +| `CC_MAX_BODY_MB` | `8` | Max request body size in MB; oversized requests are rejected with `HTTP 413` | +| `CC_UPSTREAM_PROXY` | *(unset)* | `http://host:port` CONNECT proxy for upstream Command Code requests only | | `CC_CLIENT_DRAIN_TIMEOUT_MS` | *(unset = disabled)* | Drop the client and abort upstream when downstream backpressure blocks longer than this; see [Stalled clients](#stalled-clients-neither-reading-nor-disconnecting) | | `CC_STREAM_IDLE_MS` | `30000` | Streaming upstream read idle timeout in ms; see [Upstream idle timeouts](#upstream-idle-timeouts) | | `CC_NONSTREAM_IDLE_MS` | `90000` | Non-streaming upstream read idle timeout in ms | @@ -542,7 +564,7 @@ The body exists in several copies before being forwarded: `chunks[]` / `Buffer.c | 20 MB | 100 MB | +116 MB (5.8×) | 200 | | 20 MB | 8 MB | +21 MB (1.05×) | **413** | -At startup a `warn` is logged when the implied worst case is ≥ 500 MB. The limit is **per request** and the proxy does no in-flight limiting of its own — a public deployment must add both at the reverse proxy. +At startup a `warn` is logged when the implied worst case is ≥ 500 MB. The body limit is **per request** — cap the multiplier with `CC_MAX_INFLIGHT` (in-process, global only), and add per-IP / per-key limits in the reverse proxy. A public deployment should do both. ### Suggested nginx front diff --git a/README_zh.md b/README_zh.md index f767714..145497b 100644 --- a/README_zh.md +++ b/README_zh.md @@ -76,14 +76,35 @@ commandcode/ | `CC_NONSTREAM_IDLE_MS` | 非流式上游读空闲超时(默认 `90000`)| | `CC_MAX_INFLIGHT` | 进程内在途请求上限(默认 `0` = 不限)| | `CMD_ZDR` | `zdr`(`1` 开启) | +| `CC_UPSTREAM_PROXY` | `upstreamProxy` | 开启后,代理会在 Command Code 生成请求以及 fingerprint/lifecycle 初始化请求中附加 `x-cmd-zdr: 1`。npm 版本检查和代理自己的 `/provider/v1/models` 模型目录请求不会附加该 header。该开关只是请求 Command Code 使用 ZDR-only 路由,实际数据留存和上游可用性仍由上游服务决定。 -**请求体上限**:独立于 `config.json` —— 超过 **100MB** 的请求会被拒绝并返回 `HTTP 413`(连接保持可排空,不会直接 reset)。可用 `CC_MAX_BODY_MB`(正整数,单位 MB)覆盖。 +**请求体上限**:独立于 `config.json` —— 超过 **8MB** 的请求会被拒绝并返回 `HTTP 413`(连接保持可排空,不会直接 reset)。可用 `CC_MAX_BODY_MB`(正整数,单位 MB)覆盖。默认值已由 100MB 下调([issue #20](https://github.com/MAXeaglet/commandcode-proxy/issues/20))。 -> ⚠️ **内存放大**:请求体在转发到上游前会存在多份副本,实测峰值 ≈ body 大小 × **5.1~7.4**(7MB→+52MB、20MB→+116MB;被 `413` 拒绝的请求只要 ×1.05)。因此默认 `CC_MAX_BODY_MB=100` 意味着**单个请求**最坏可吃 ~550MB,且该上限是每请求的、不是全局的。详见[内存与部署](#内存与部署)。 +> ⚠️ **内存放大**:请求体在转发到上游前会存在多份副本,实测峰值 ≈ body 大小 × **5.1~7.4**(7MB→+52MB、20MB→+116MB;被 `413` 拒绝的请求只要 ×1.05)。旧的默认 `CC_MAX_BODY_MB=100` 意味着**单个请求**最坏可吃 ~550MB,对 Dockerfile 面向的 1 核小 VPS 不是合理默认值,现已下调为 **8MB**(约 45MB/请求)。该上限仍是每请求的、不是全局的 —— 用 `CC_MAX_INFLIGHT` 或在反向代理层一并封顶并发。详见[内存与部署](#内存与部署)。 + +### 上游代理(`upstreamProxy` / `CC_UPSTREAM_PROXY`) + +让代理**发往 Command Code 的请求**走本地 HTTP 代理 —— 用于出口地区调整,或排查风控 `403` 时做 IP 维度对照。 + +```json +{ "upstreamProxy": "http://127.0.0.1:7890" } +``` + +```bash +CC_UPSTREAM_PROXY=http://127.0.0.1:7890 npm start +``` + +- 作用于 `/alpha/generate`、`/alpha/fingerprint/record`、`/alpha/lifecycle-events` 与 `/provider/v1/models`。 +- **不影响**本地监听、`/health` 与 npm 版本检查。 +- 仅支持 `http://`(CONNECT)代理。实现方式是自建 CONNECT 隧道 + `node:https` 复用同一 socket,**不新增任何依赖**,Node 18+ 即可用。 +- 每个上游请求各自建立一条隧道连接。TLS 为端到端:证书按 `api.commandcode.ai` 校验,绝不针对代理降级。 +- **指纹/lifecycle 预请求也走代理**是刻意的:若它们直连而上游生成走代理,同一账号会从两个不同 IP 注册 —— 正是你想避免的那种矛盾。 + +> Node 原生 `fetch` **不读** `HTTPS_PROXY`/`HTTP_PROXY`。官方环境变量路线需要 Node ≥ 22.21 / 24.5 且设 `NODE_USE_ENV_PROXY=1`;本选项两者都不需要。 ## API 接口 @@ -466,7 +487,8 @@ npm run docker:build:multi |------|--------|------| | `PORT` | `3050` | 容器内监听端口 | | `PROXY_PORT` | `3050` | 主机映射端口(仅 compose) | -| `CC_MAX_BODY_MB` | `100` | 请求体大小上限(MB),超限请求返回 `HTTP 413` | +| `CC_MAX_BODY_MB` | `8` | 请求体大小上限(MB),超限请求返回 `HTTP 413` | +| `CC_UPSTREAM_PROXY` | 空 | 仅作用于 CC 上游请求的 `http://host:port` CONNECT 代理 | | `CC_CLIENT_DRAIN_TIMEOUT_MS` | 空(禁用)| 下游背压阻塞超过该毫秒数则断开该客户端并中止上游请求,见[僵死连接](#僵死连接既不读也不断开) | | `CC_STREAM_IDLE_MS` | `30000` | 流式上游读空闲超时(毫秒),见[上游空闲超时](#上游空闲超时) | | `CC_NONSTREAM_IDLE_MS` | `90000` | 非流式上游读空闲超时(毫秒)| @@ -546,7 +568,7 @@ body 在转发到上游前同时存在多份副本:`chunks[]` / `Buffer.concat | 20 MB | 100 MB | +116 MB(5.8×)| 200 | | 20 MB | 8 MB | +21 MB(1.05×)| **413** | -启动时若隐含最坏峰值 ≥ 500MB,日志会输出 `warn` 提示。上限是**按请求**的,proxy 自身没有在途限流 —— 公网部署必须在反向代理层补上。 +启动时若隐含最坏峰值 ≥ 500MB,日志会输出 `warn` 提示。body 上限是**按请求**的 —— 乘数用 `CC_MAX_INFLIGHT`(进程内、仅全局)封顶,按 IP / 按 key 的限流在反向代理层补上。公网部署建议两者都做。 ### nginx 反代建议 diff --git a/config.json b/config.json index b16c751..0d8ece2 100644 --- a/config.json +++ b/config.json @@ -6,5 +6,6 @@ "projectSlug": "cc-proxy", "logFile": "", "logLevel": "info", - "zdr": false + "zdr": false, + "upstreamProxy": "" } diff --git a/proxy.mjs b/proxy.mjs index 486bdcc..b521ece 100644 --- a/proxy.mjs +++ b/proxy.mjs @@ -3,6 +3,9 @@ * 基于真实 CLI 流量抓包数据构建 */ import http from 'http'; +import https from 'https'; +import tls from 'tls'; +import { Readable } from 'stream'; import crypto from 'crypto'; import { randomUUID } from 'crypto'; import { readFileSync, existsSync, appendFileSync } from 'fs'; @@ -24,6 +27,7 @@ function loadConfig() { modelRefreshIntervalMs: 5 * 60 * 1000, // 5 minutes zdr: false, emptySystemPlaceholder: true, // 无 system prompt 时发空格占位,阻止 CC 上游注入 ~7.5K token 默认提示词(issue #17) + upstreamProxy: '', // 上游 HTTP 代理,如 http://127.0.0.1:7890(issue #18) }; const configPath = resolve(__dirname, 'config.json'); @@ -45,6 +49,7 @@ function loadConfig() { if (process.env.CC_USE_PROVIDER_MODELS) defaults.useProviderModels = process.env.CC_USE_PROVIDER_MODELS !== 'false'; if (process.env.CMD_ZDR !== undefined) defaults.zdr = process.env.CMD_ZDR === '1'; if (process.env.CC_EMPTY_SYSTEM_PLACEHOLDER) defaults.emptySystemPlaceholder = process.env.CC_EMPTY_SYSTEM_PLACEHOLDER !== 'false'; + if (process.env.CC_UPSTREAM_PROXY) defaults.upstreamProxy = process.env.CC_UPSTREAM_PROXY; return defaults; } @@ -144,15 +149,18 @@ async function refreshCCVersion() { refreshCCVersion(); // 启动时立即拉取 setInterval(refreshCCVersion, CC_VERSION_REFRESH_MS); -// 请求体大小上限:默认 100MB,可用环境变量 CC_MAX_BODY_MB 覆盖(正整数,单位 MB) +// 请求体大小上限:默认 8MB,可用环境变量 CC_MAX_BODY_MB 覆盖(正整数,单位 MB)。 +// 默认值已从 100MB 下调(issue #20 Finding 2): // ⚠️ 内存特性(issue #20 实测):请求体在转发到上游前会同时存在多份副本 —— // chunks[] / Buffer.concat / utf8 字符串 / JSON.parse 对象树 / buildCcRequest 重建对象树 / JSON.stringify 序列化体。 // 实测峰值 ≈ body 大小 × 5.1~7.4(7MB→+52MB,20MB→+116MB;而 413 拒绝路径只要 ×1.05)。 -// 故 100MB 上限意味着「单个请求」最坏可吃 ~550MB,且该上限是每请求的、不是全局的。 -// 公网/多用户部署请在反向代理层同时限制 body 大小与在途请求数(见 README「内存与部署」)。 +// 故 100MB 上限意味着「单个请求」最坏可吃 ~550MB,对 Dockerfile 面向的 1 核小 VPS 不是合理默认值; +// 8MB 对应约 45MB/请求,且 413 走的是丢弃分支、几乎零成本。 +// 该上限仍是每请求的、不是全局的:公网部署请在反向代理层限流, +// 或用 CC_MAX_INFLIGHT / config.maxInflight 打开进程内全局在途上限。 const MAX_BODY_SIZE = (() => { const mb = Number.parseInt(process.env.CC_MAX_BODY_MB ?? '', 10); - return Number.isFinite(mb) && mb > 0 ? mb * 1024 * 1024 : 100 * 1024 * 1024; + return Number.isFinite(mb) && mb > 0 ? mb * 1024 * 1024 : 8 * 1024 * 1024; })(); // 上游读空闲超时(issue #19):只计「reader.read() 的等待」,每收到一个 chunk 重置, // 不是整个请求的总时长。默认值保持不变(30s / 90s),可用环境变量覆盖 —— @@ -314,7 +322,7 @@ async function ensureInitialized(apiKey, signal) { const fingerprint = state.fingerprint || {}; await Promise.all([ - fetch(`${CFG.apiBase}/alpha/fingerprint/record`, { + upstreamFetch(`${CFG.apiBase}/alpha/fingerprint/record`, { method: 'POST', headers, signal, body: JSON.stringify(fingerprint), }).then(r => { @@ -324,7 +332,7 @@ async function ensureInitialized(apiKey, signal) { if (e.name !== 'AbortError') log('warn', 'Fingerprint record error', { error: e.message }); }), - fetch(`${CFG.apiBase}/alpha/lifecycle-events`, { + upstreamFetch(`${CFG.apiBase}/alpha/lifecycle-events`, { method: 'POST', headers, signal, body: JSON.stringify({ eventType: 'cli_session_exists', @@ -964,6 +972,117 @@ function getApiKey(headers) { return null; } +// ── 上游 HTTP(S) 代理(issue #18)──────────────────── +// 仅作用于发往 CC 上游的请求(/alpha/generate、/provider/v1/models)。 +// 本地监听、/health 与 npm registry 版本检查都不经过代理。 +// +// 零依赖实现:自己建立 CONNECT 隧道,再用 node:https 复用同一个 socket, +// 因此不需要 undici / https-proxy-agent,engines >=18 也能用。 +// 注意 Node 原生 fetch 不读 HTTPS_PROXY/HTTP_PROXY;官方的环境变量方案需要 +// Node >= 22.21 / 24.5 并设 NODE_USE_ENV_PROXY=1(README 有说明)。 +const UPSTREAM_PROXY = CFG.upstreamProxy || ''; +const PROXY_CONNECT_TIMEOUT_MS = 15000; + +function parseProxyUrl(raw) { + let u; + try { + u = new URL(raw); + } catch { + throw new Error(`upstreamProxy is not a valid URL: ${raw}`); + } + if (u.protocol !== 'http:') { + throw new Error(`upstreamProxy only supports http:// (CONNECT) proxies, got ${u.protocol}//`); + } + const auth = u.username + ? 'Basic ' + Buffer.from(`${decodeURIComponent(u.username)}:${decodeURIComponent(u.password)}`).toString('base64') + : null; + return { host: u.hostname, port: Number.parseInt(u.port || '80', 10), auth }; +} + +/** Response 的 headers 需要字符串值;node 的 set-cookie 是数组,展开为多行。 */ +function headersToInit(raw) { + const out = []; + for (const [k, v] of Object.entries(raw)) { + if (Array.isArray(v)) { for (const item of v) out.push([k, String(item)]); } + else if (v !== undefined) out.push([k, String(v)]); + } + return out; +} + +/** 经 HTTP 代理发上游请求,返回与 fetch 兼容的 Response(.ok/.status/.text()/.body)。 */ +async function proxyFetch(urlStr, options = {}) { + const proxy = parseProxyUrl(UPSTREAM_PROXY); + const u = new URL(urlStr); + const isTls = u.protocol === 'https:'; + const port = Number.parseInt(u.port || (isTls ? '443' : '80'), 10); + const target = `${u.hostname}:${port}`; + const { signal, body } = options; + const onAbort = (fn) => { if (signal) signal.addEventListener('abort', fn, { once: true }); }; + + // 1. CONNECT 隧道 —— 代理只做裸字节转发,TLS 由本端端到端完成 + const rawSocket = await new Promise((resolve, reject) => { + const connectReq = http.request({ + host: proxy.host, + port: proxy.port, + method: 'CONNECT', + path: target, + headers: { Host: target, ...(proxy.auth ? { 'Proxy-Authorization': proxy.auth } : {}) }, + timeout: PROXY_CONNECT_TIMEOUT_MS, + }); + connectReq.on('connect', (res, socket) => { + if (res.statusCode !== 200) { + socket.destroy(); + reject(new Error(`upstream proxy CONNECT ${target} failed: HTTP ${res.statusCode}`)); + return; + } + resolve(socket); + }); + connectReq.on('timeout', () => connectReq.destroy(new Error('upstream proxy CONNECT timeout'))); + connectReq.on('error', reject); + onAbort(() => { try { connectReq.destroy(); } catch {} }); + connectReq.end(); + }); + + // 2. 隧道上做 TLS(证书按目标主机名校验,不做任何降级) + let socket = rawSocket; + if (isTls) { + socket = tls.connect({ socket: rawSocket, servername: u.hostname }); + await new Promise((resolve, reject) => { + socket.once('secureConnect', resolve); + socket.once('error', reject); + onAbort(() => { try { socket.destroy(); } catch {} }); + }); + } + + // 3. 复用隧道 socket 发请求 + return await new Promise((resolve, reject) => { + const mod = isTls ? https : http; + const req = mod.request({ + host: u.hostname, + port, + path: u.pathname + u.search, + method: options.method || 'GET', + headers: options.headers || {}, + createConnection: () => socket, + }, (res) => { + resolve(new Response(Readable.toWeb(res), { + status: res.statusCode, + statusText: res.statusMessage, + headers: headersToInit(res.headers), + })); + }); + req.on('error', reject); + onAbort(() => { try { req.destroy(); } catch {} }); + if (body !== undefined && body !== null) req.write(body); + req.end(); + }); +} + +/** 上游请求入口:配了代理走隧道,否则用原生 fetch(默认路径行为完全不变)。 */ +function upstreamFetch(urlStr, options) { + return UPSTREAM_PROXY ? proxyFetch(urlStr, options) : fetch(urlStr, options); +} + // ── 流式转发 ──────────────────────────────────────── async function forwardToCC(body, apiKey, incomingHeaders = {}, signal, promptCacheKey) { @@ -994,7 +1113,7 @@ async function forwardToCC(body, apiKey, incomingHeaders = {}, signal, promptCac headers['x-cmd-zdr'] = '1'; } - const response = await fetch(url, { + const response = await upstreamFetch(url, { method: 'POST', headers, body: JSON.stringify(body), @@ -1032,6 +1151,10 @@ async function handleChatCompletions(req, res) { // 构建 CC 请求体 const ccBody = buildCcRequest(openaiReq); + // issue #20:ccBody 建好后,openaiReq 这棵 20MB 级对象树只剩 prompt_cache_key 还被用到。 + // 先取出该值再断开引用,让这一份副本可以更早被 GC 回收(原来是整段请求期间一直活着)。 + const promptCacheKey = openaiReq.prompt_cache_key; + openaiReq = null; // AbortController 用于客户端断连时真正打断 CC 上游(pi-commandcode-provider 模式) const abortController = new AbortController(); @@ -1046,7 +1169,7 @@ async function handleChatCompletions(req, res) { // 首次初始化(fingerprint + lifecycle) await ensureInitialized(apiKey, abortController.signal); // 转发到 CC API(传入客户端 headers,用于提取 session ID) - const ccResponse = await forwardToCC(ccBody, apiKey, req.headers, abortController.signal, openaiReq.prompt_cache_key); + const ccResponse = await forwardToCC(ccBody, apiKey, req.headers, abortController.signal, promptCacheKey); if (!ccResponse.ok) { const errorText = await ccResponse.text().catch(() => ''); @@ -1838,8 +1961,12 @@ async function handleMessages(req, res) { const model = anthropicReq.model || 'claude-sonnet-4-6'; // Convert Anthropic → OpenAI → CC - const openaiReq = convertAnthropicToOpenAI(anthropicReq); + let openaiReq = convertAnthropicToOpenAI(anthropicReq); const ccBody = buildCcRequest(openaiReq); + // issue #20:ccBody 已建好,原始请求树(anthropicReq)与中间树(openaiReq)都不再被引用, + // 显式断开以便尽早回收 —— 否则它们会和 ccBody 一起活到整段请求结束。 + openaiReq = null; + anthropicReq = null; const abortController = new AbortController(); let aborted = false; @@ -2152,7 +2279,7 @@ async function fetchModels(apiKey) { try { if (!apiKey || !CFG.useProviderModels) throw new Error('Provider models disabled'); - const response = await fetch(`${CFG.apiBase}/provider/v1/models`, { + const response = await upstreamFetch(`${CFG.apiBase}/provider/v1/models`, { headers: { 'Authorization': `Bearer ${apiKey}`, 'x-cli-environment': 'production', @@ -2950,6 +3077,7 @@ server.listen(CFG.port, CFG.host, () => { clientDrainTimeout: CLIENT_DRAIN_TIMEOUT_MS > 0 ? `${CLIENT_DRAIN_TIMEOUT_MS}ms` : 'disabled', idleTimeouts: `stream ${STREAM_IDLE_TIMEOUT_MS}ms / nonstream ${NONSTREAM_IDLE_TIMEOUT_MS}ms`, maxInflight: MAX_INFLIGHT > 0 ? `${MAX_INFLIGHT} (global, /health exempt)` : 'unlimited (CC_MAX_INFLIGHT=0)', + upstreamProxy: UPSTREAM_PROXY || '(direct)', }); if (CLIENT_DRAIN_TIMEOUT_MS > 0) { log('info', 'Client drain timeout enabled', { timeoutMs: CLIENT_DRAIN_TIMEOUT_MS }); @@ -2961,7 +3089,7 @@ server.listen(CFG.port, CFG.host, () => { log('warn', 'Request body limit implies high per-request worst-case memory', { maxBodyMB: bodyCapMB, worstCaseRSSPerRequestMB: worstCaseMB, - hint: 'lower CC_MAX_BODY_MB and/or cap in-flight requests at the reverse proxy (see README)', + hint: 'lower CC_MAX_BODY_MB, set CC_MAX_INFLIGHT, and/or cap in-flight requests at the reverse proxy (see README)', }); } if (!CFG.apiKey) { From 05919738655a1b225c38e00ae10b15363e558b86 Mon Sep 17 00:00:00 2001 From: xelr233 Date: Thu, 10 Sep 2026 23:06:38 +0800 Subject: [PATCH 03/21] =?UTF-8?q?fix:=20=E8=BF=98=E5=8E=9F=20CC=5FMAX=5FBO?= =?UTF-8?q?DY=5FMB=20=E9=BB=98=E8=AE=A4=E5=80=BC=E4=B8=BA=20100MB=20?= =?UTF-8?q?=E2=80=94=E2=80=94=20#7=20=E7=9A=84=E8=AF=81=E6=8D=AE=E5=90=A6?= =?UTF-8?q?=E5=86=B3=E4=BA=86=208MB?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 上一版把默认值降到 8MB 是错的。核过 #7 与其修复提交 573e260 后确认:#7 的 真实触发场景是多模态长会话(21 张 base64 图片累积约 10.11 MiB 的合法请求), 任何 4~8MB 的默认值都会把这类请求整体挡在门外。 需要分清的是:#7 的「客户端只看到 Connection error」症状由「超限返回 413 + 排空连接」这条路径解决,与阈值取值无关;但阈值决定的是功能边界,而真实的 多模态上下文确实会到 10MB 量级,所以默认值必须留足。 结论:既然 body 阈值必须够大,要约束的就是并发侧 —— 这个角色由上游 c443889 的 CC_MAX_INFLIGHT 承担(本分支不再自带实现,只保留上游那套)。本提交因此 只做两件事:还原默认值,并在两个 README 的「在途上限」章节补一段「为什么 body 默认值不能降」的依据,顺带修掉「内存与部署」里那句已过时的「proxy 自身 没有在途限流」。 复测: - 默认(100MB) 放行 9MB 请求 -> 200(#7 的多模态场景不再被挡) - CC_MAX_BODY_MB=8 时同一请求 -> 413,拒绝后小请求仍正常 - 其余验证项(UA/threadId、上游代理)全部通过 --- README.md | 8 +++++--- README_zh.md | 8 +++++--- proxy.mjs | 18 +++++++++++------- 3 files changed, 21 insertions(+), 13 deletions(-) diff --git a/README.md b/README.md index 790b0d5..e51bf7d 100644 --- a/README.md +++ b/README.md @@ -84,9 +84,9 @@ to the npm version check or the proxy's `/provider/v1/models` catalog request. This requests Command Code's ZDR-only routing; the upstream service remains the authority for actual retention and provider availability. -**Request body limit**: independent of `config.json` — requests larger than **8 MB** are rejected with `HTTP 413` (the connection is kept alive and drained, not reset). Override with `CC_MAX_BODY_MB` (positive integer, unit: MB). The default was lowered from 100 MB ([#20](https://github.com/MAXeaglet/commandcode-proxy/issues/20)). +**Request body limit**: independent of `config.json` — requests larger than **100 MB** are rejected with `HTTP 413` (the connection is kept alive and drained, not reset). Override with `CC_MAX_BODY_MB` (positive integer, unit: MB). -> ⚠️ **Memory amplification**: a request body exists in several copies before it reaches upstream; measured peak ≈ body size × **5.1–7.4** (7 MB → +52 MB, 20 MB → +116 MB, while a request rejected with `413` costs only ×1.05). The old default `CC_MAX_BODY_MB=100` implied up to ~550 MB for a **single** request, which is not a sane default for the 1-core VPS the Dockerfile targets. The default is now **8 MB** (~45 MB peak per request). The limit is still per-request, not global — cap concurrency with `CC_MAX_INFLIGHT` or at the reverse proxy. See [Memory & Deployment](#memory--deployment). +> ⚠️ **Memory amplification**: a request body exists in several copies before it reaches upstream; measured peak ≈ body size × **5.1–7.4** (7 MB → +52 MB, 20 MB → +116 MB, while a request rejected with `413` costs only ×1.05). The default `CC_MAX_BODY_MB=100` therefore implies up to ~550 MB for a **single** request, and that limit is per-request, not global. See [Memory & Deployment](#memory--deployment). ### Upstream proxy (`upstreamProxy` / `CC_UPSTREAM_PROXY`) @@ -489,7 +489,7 @@ npm run docker:build:multi |----------|---------|-------------| | `PORT` | `3050` | Container listen port | | `PROXY_PORT` | `3050` | Host port (compose only) | -| `CC_MAX_BODY_MB` | `8` | Max request body size in MB; oversized requests are rejected with `HTTP 413` | +| `CC_MAX_BODY_MB` | `100` | Max request body size in MB; oversized requests are rejected with `HTTP 413` | | `CC_UPSTREAM_PROXY` | *(unset)* | `http://host:port` CONNECT proxy for upstream Command Code requests only | | `CC_CLIENT_DRAIN_TIMEOUT_MS` | *(unset = disabled)* | Drop the client and abort upstream when downstream backpressure blocks longer than this; see [Stalled clients](#stalled-clients-neither-reading-nor-disconnecting) | | `CC_STREAM_IDLE_MS` | `30000` | Streaming upstream read idle timeout in ms; see [Upstream idle timeouts](#upstream-idle-timeouts) | @@ -510,6 +510,8 @@ Over the limit it returns `503` + `Retry-After: 5` + `type: server_busy` — a s **Why it exists**: memory is `in-flight × (0.13 MB + 5.5 × body_MB)`. `CC_MAX_BODY_MB` bounds only the **per-request** term; nothing bounds the multiplier — at the default 100 MB, N concurrent requests can cost N × 550 MB. +> **Why the default body cap stays at 100 MB**: [#7](https://github.com/MAXeaglet/commandcode-proxy/issues/7) recorded a legitimate multimodal session (21 base64 images, ~10.11 MiB) hitting the old 10 MB cap, so the threshold cannot be lowered without breaking real usage — which is exactly why the concurrency side has to be bounded instead. The startup warning about implied worst-case memory is advisory; `CC_MAX_INFLIGHT` is the enforcement. + > ⚠️ Enabling this is **not** the same as being memory-safe: 32 × 550 MB still exceeds a small box. For a hard bound, lower `CC_MAX_BODY_MB` **as well**. ## Upstream Idle Timeouts diff --git a/README_zh.md b/README_zh.md index 145497b..60c6605 100644 --- a/README_zh.md +++ b/README_zh.md @@ -82,9 +82,9 @@ commandcode/ `x-cmd-zdr: 1`。npm 版本检查和代理自己的 `/provider/v1/models` 模型目录请求不会附加该 header。该开关只是请求 Command Code 使用 ZDR-only 路由,实际数据留存和上游可用性仍由上游服务决定。 -**请求体上限**:独立于 `config.json` —— 超过 **8MB** 的请求会被拒绝并返回 `HTTP 413`(连接保持可排空,不会直接 reset)。可用 `CC_MAX_BODY_MB`(正整数,单位 MB)覆盖。默认值已由 100MB 下调([issue #20](https://github.com/MAXeaglet/commandcode-proxy/issues/20))。 +**请求体上限**:独立于 `config.json` —— 超过 **100MB** 的请求会被拒绝并返回 `HTTP 413`(连接保持可排空,不会直接 reset)。可用 `CC_MAX_BODY_MB`(正整数,单位 MB)覆盖。 -> ⚠️ **内存放大**:请求体在转发到上游前会存在多份副本,实测峰值 ≈ body 大小 × **5.1~7.4**(7MB→+52MB、20MB→+116MB;被 `413` 拒绝的请求只要 ×1.05)。旧的默认 `CC_MAX_BODY_MB=100` 意味着**单个请求**最坏可吃 ~550MB,对 Dockerfile 面向的 1 核小 VPS 不是合理默认值,现已下调为 **8MB**(约 45MB/请求)。该上限仍是每请求的、不是全局的 —— 用 `CC_MAX_INFLIGHT` 或在反向代理层一并封顶并发。详见[内存与部署](#内存与部署)。 +> ⚠️ **内存放大**:请求体在转发到上游前会存在多份副本,实测峰值 ≈ body 大小 × **5.1~7.4**(7MB→+52MB、20MB→+116MB;被 `413` 拒绝的请求只要 ×1.05)。因此默认 `CC_MAX_BODY_MB=100` 意味着**单个请求**最坏可吃 ~550MB,且该上限是每请求的、不是全局的。详见[内存与部署](#内存与部署)。 ### 上游代理(`upstreamProxy` / `CC_UPSTREAM_PROXY`) @@ -487,7 +487,7 @@ npm run docker:build:multi |------|--------|------| | `PORT` | `3050` | 容器内监听端口 | | `PROXY_PORT` | `3050` | 主机映射端口(仅 compose) | -| `CC_MAX_BODY_MB` | `8` | 请求体大小上限(MB),超限请求返回 `HTTP 413` | +| `CC_MAX_BODY_MB` | `100` | 请求体大小上限(MB),超限请求返回 `HTTP 413` | | `CC_UPSTREAM_PROXY` | 空 | 仅作用于 CC 上游请求的 `http://host:port` CONNECT 代理 | | `CC_CLIENT_DRAIN_TIMEOUT_MS` | 空(禁用)| 下游背压阻塞超过该毫秒数则断开该客户端并中止上游请求,见[僵死连接](#僵死连接既不读也不断开) | | `CC_STREAM_IDLE_MS` | `30000` | 流式上游读空闲超时(毫秒),见[上游空闲超时](#上游空闲超时) | @@ -509,6 +509,8 @@ CC_MAX_INFLIGHT=32 npm start # 最多同时处理 32 个请求 **为什么需要它**:内存 = `在途数 × (0.13MB + 5.5 × body_MB)`。`CC_MAX_BODY_MB` 只管住**单请求**量级,乘数无人管 —— 默认 100MB 时 N 个并发最坏可达 N × 550MB。 +> **为什么 body 默认值保持 100MB**:[#7](https://github.com/MAXeaglet/commandcode-proxy/issues/7) 记录了一个合法的多模态长会话(21 张 base64 图片,约 10.11 MiB)会撞上旧的 10MB 上限 —— 阈值降不下去,正因为此才必须去约束并发侧。启动时那条「最坏内存」warn 只是提示,`CC_MAX_INFLIGHT` 才是执行层。 + > ⚠️ 开启本项**不等于**内存安全:32 × 550MB 仍远超小机器容量。要拿到硬性上界,需**同时**下调 `CC_MAX_BODY_MB`。 ## 上游空闲超时 diff --git a/proxy.mjs b/proxy.mjs index b521ece..5674cea 100644 --- a/proxy.mjs +++ b/proxy.mjs @@ -149,18 +149,22 @@ async function refreshCCVersion() { refreshCCVersion(); // 启动时立即拉取 setInterval(refreshCCVersion, CC_VERSION_REFRESH_MS); -// 请求体大小上限:默认 8MB,可用环境变量 CC_MAX_BODY_MB 覆盖(正整数,单位 MB)。 -// 默认值已从 100MB 下调(issue #20 Finding 2): +// 请求体大小上限:默认 100MB,可用环境变量 CC_MAX_BODY_MB 覆盖(正整数,单位 MB)。 +// 默认值保持 100MB(573e260 为修 issue #7 设定)——不要下调,理由是有实际证据的: +// #7 的真实触发场景是多模态长会话,21 张 Base64 图片累积到约 10.11 MiB 的合法请求, +// 下调到 4~8MB 会把这类请求整体挡在门外。#7 的「Connection error」症状由 +// 「超限返回 413 + 排空连接」这条路径解决,与阈值取值无关;但阈值决定了功能边界, +// 所以默认值必须容纳真实的多模态上下文。 // ⚠️ 内存特性(issue #20 实测):请求体在转发到上游前会同时存在多份副本 —— // chunks[] / Buffer.concat / utf8 字符串 / JSON.parse 对象树 / buildCcRequest 重建对象树 / JSON.stringify 序列化体。 // 实测峰值 ≈ body 大小 × 5.1~7.4(7MB→+52MB,20MB→+116MB;而 413 拒绝路径只要 ×1.05)。 -// 故 100MB 上限意味着「单个请求」最坏可吃 ~550MB,对 Dockerfile 面向的 1 核小 VPS 不是合理默认值; -// 8MB 对应约 45MB/请求,且 413 走的是丢弃分支、几乎零成本。 -// 该上限仍是每请求的、不是全局的:公网部署请在反向代理层限流, -// 或用 CC_MAX_INFLIGHT / config.maxInflight 打开进程内全局在途上限。 +// 故 100MB 上限意味着「单个请求」最坏可吃 ~550MB,且该上限是每请求的、不是全局的。 +// 正因为阈值必须留足,#20 的第二半必须补齐:并发侧要有约束。 +// 进程内可用 CC_MAX_INFLIGHT,边缘侧用反向代理 limit_conn +// (见 README「内存与部署」)。 const MAX_BODY_SIZE = (() => { const mb = Number.parseInt(process.env.CC_MAX_BODY_MB ?? '', 10); - return Number.isFinite(mb) && mb > 0 ? mb * 1024 * 1024 : 8 * 1024 * 1024; + return Number.isFinite(mb) && mb > 0 ? mb * 1024 * 1024 : 100 * 1024 * 1024; })(); // 上游读空闲超时(issue #19):只计「reader.read() 的等待」,每收到一个 chunk 重置, // 不是整个请求的总时长。默认值保持不变(30s / 90s),可用环境变量覆盖 —— From a4eaaf897dd2ed8251f04f3829f280d029b94469 Mon Sep 17 00:00:00 2001 From: xelr233 Date: Fri, 11 Sep 2026 00:01:53 +0800 Subject: [PATCH 04/21] =?UTF-8?q?feat:=20=E8=AE=BE=E5=A4=87=E6=8C=87?= =?UTF-8?q?=E7=BA=B9=E6=94=B9=E4=B8=BA=E6=8C=89=20API=20key=20=E7=A1=AE?= =?UTF-8?q?=E5=AE=9A=E6=80=A7=E6=B4=BE=E7=94=9F=EF=BC=8C=E5=B9=B6=E5=AF=B9?= =?UTF-8?q?=E9=BD=90=E5=AE=98=E6=96=B9=20CLI=20=E5=93=88=E5=B8=8C=E7=AE=97?= =?UTF-8?q?=E6=B3=95?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 修三个问题:指纹漂移(进程重启 / 多实例 / 每 12h 被动重置)、 以及 thumbmark 与 hashSignal 的构造与官方 CLI 不一致。 指纹由 HMAC-SHA256(CC_FP_SALT, apiKey) 确定性派生,一个 key 恒定一台设备: 进程重启 Map 清空 → 换一台机器 → 同一台机器 第二个实例 同一 key = 两台机器 → 同一台机器 session 过期 12h keyStateStore.delete → 每 12h 换一台机器 → 同一台机器 第三条是原实现最明显的破绽:真实用户不会一天换两次电脑,而上游 device_fingerprints 表按 (userId, thumbmark) 建唯一索引。 刻意**不是**"按 key 取哈希桶选设备":固定池的熵上限就是池大小,key 数 一旦超过池容量就必然出现多 key 共用指纹(模拟:50 key / 1000 池 → 约 2 个 碰撞;50 key / 100 池 → 约 20 个),而共用 thumbmark 正是"多账号同机"的 直接证据。派生方案每个 key 仍是独立设备,碰撞概率 2^-256。 CC_FP_MODE=random 保留原「每进程随机」行为作为回退。 原 keyStateStore.delete 写在 session 清理循环里,本意是内存回收, 实际效果是每 12h 重置指纹。现在指纹状态有自己的「按空闲淘汰」(24h), 且因为指纹是派生的,淘汰后重新派生得到的仍是同一台设备,不构成漂移; 活跃 key 的 nextInitAt 也不再被重置,避免重复发送预请求。 源码(buildMachineFingerprint / hashSignal): ib = "command-code:device-fingerprint:v1" hashSignal(v) = sha256(ib + "\0" + v.trim().toLowerCase()) // v 是原始值 thumbmark = sha256(ib + "\0machine\0" + [machineId, macs.join(",")].join("|")) machineId 非空时 hostname / cpuModel 不参与 原实现:直接对随机 hex 求 sha256(缺 ib 前缀),且 thumbmark 由各 component 的**哈希**拼成、另加 platform/osRelease/cpuModel 等字段。 上游拿不到原始 machineId、无法重算,所以这处不一致本来就检测不到; 既然要动就一次对齐。components 的字段集原本就是对的(runtime: "cli"、 collectorVersion: 1 都对),本次只改哈希构造。 指纹套件(本地 mock 上游,起停真实进程): - components 字段集与 CLI 完全一致(15 个),runtime/collectorVersion 正确 - 重启后同一 key -> 同一 thumbmark;CC_FP_MODE=random 下则不同 - 40 个 key -> 40 个不同 thumbmark,零碰撞;CPU 14 种、时区 14 种分布 - 不同 CC_FP_SALT -> 不同设备;同一盐 -> 稳定 白盒交叉验证(关键): - 独立复刻 CLI 算法,对同一组原始值算期望值,与运行中代理实际上报的 5 个 key × 15 个字段逐字节比对 -> 全部一致 回归:#20 内存、#18 上游代理、UA=cli、threadId===x-session-id 全部通过。 --- README.md | 33 ++++++++++++- README_zh.md | 33 ++++++++++++- config.json | 4 +- proxy.mjs | 137 +++++++++++++++++++++++++++++++++++++++------------ 4 files changed, 171 insertions(+), 36 deletions(-) diff --git a/README.md b/README.md index e51bf7d..268d51a 100644 --- a/README.md +++ b/README.md @@ -77,6 +77,8 @@ commandcode/ | `CC_MAX_INFLIGHT` | In-process concurrent request cap (default `0` = unlimited) | | `CMD_ZDR` | `zdr` (`1` to enable) | | `CC_UPSTREAM_PROXY` | `upstreamProxy` | +| `CC_FP_MODE` | `fpMode` (`derived` default, `random` to opt out) | +| `CC_FP_SALT` | `fpSalt` | When enabled, the proxy sends `x-cmd-zdr: 1` on Command Code generation requests and the fingerprint/lifecycle initialization requests. It does not add the header @@ -88,6 +90,31 @@ authority for actual retention and provider availability. > ⚠️ **Memory amplification**: a request body exists in several copies before it reaches upstream; measured peak ≈ body size × **5.1–7.4** (7 MB → +52 MB, 20 MB → +116 MB, while a request rejected with `413` costs only ×1.05). The default `CC_MAX_BODY_MB=100` therefore implies up to ~550 MB for a **single** request, and that limit is per-request, not global. See [Memory & Deployment](#memory--deployment). +### Device fingerprint (`fpMode` / `CC_FP_MODE`, `fpSalt` / `CC_FP_SALT`) + +The device fingerprint reported to `/alpha/fingerprint/record` is **derived deterministically** from the API key (`HMAC-SHA256(CC_FP_SALT, apiKey)`), so one key is always one device: + +| Event | Old behaviour (random) | Now (derived) | +|---|---|---| +| Process restart | Map cleared → **new machine** | same machine | +| Second instance | same key = **two machines** | same machine | +| Session expiry (12h) | `keyStateStore.delete` → **new machine every 12h** | same machine | + +> The 12h case was the most visible: a real user does not replace their computer twice a day, and upstream's `device_fingerprints` table is keyed on `(userId, thumbmark)`. + +**Why derived rather than "pick a device from a hash bucket"** — a fixed pool caps entropy at the pool size, so once the number of keys exceeds it, keys *must* share a fingerprint. With ~50 keys and a 1000-entry pool, ~2 keys collide; with a 100-entry pool, ~20 do. A shared `thumbmark` under two different `userId`s is direct evidence of multi-account-same-machine — exactly what you don't want to manufacture. Derivation keeps every key a distinct device (collision probability 2⁻²⁵⁶) while still being stable. + +```bash +CC_FP_SALT=some-local-secret npm start # optional: isolates fingerprints between deployments +CC_FP_MODE=random npm start # opt out: old random-per-process behaviour +``` + +The salt is optional but recommended: without it the derivation is a pure function of the API key, so anyone who knows the scheme could recompute your users' fingerprints. With it, the same key yields different devices on different deployments, at no cost. + +The hash construction follows the official CLI (`buildMachineFingerprint` / `hashSignal` in `command-code`) — `thumbmark = sha256(IB + "\0machine\0" + [machineId, macs.join(",")].join("|"))` with `IB = "command-code:device-fingerprint:v1"`, and each component hashed as `sha256(IB + "\0" + value.toLowerCase())`. The previous implementation hashed random hex without the `IB` prefix and built the thumbmark from the component *hashes*; upstream cannot recompute either way (it never sees the raw `machineId`), so it was undetectable — but it is now aligned. + +> Not addressed here: the appearance pool is still all high-end desktop CPUs and the timezone is still drawn from a global pool. On a single-egress-IP deployment, a cluster of machines spread across 15 timezones is a distribution that does not look like real users. Binding `timezone` to the egress IP is the natural next step, but it depends on deployment specifics, so it is left to the operator. + ### Upstream proxy (`upstreamProxy` / `CC_UPSTREAM_PROXY`) Route the requests the proxy makes **to Command Code** through a local HTTP proxy — for egress-region switching, or for comparing IPs when debugging risk-control `403`s. @@ -380,7 +407,7 @@ Based on analysis of official CLI traffic (version auto-fetched from npm registr | Mechanism | Implementation | |-----------|---------------| -| **Device Fingerprint** | `POST /alpha/fingerprint/record` before first request per key; random fingerprint pool (15 CPUs, global timezones), SHA-256 hashed, per-key binding, refreshed every 8h + 2h jitter | +| **Device Fingerprint** | `POST /alpha/fingerprint/record` before first request per key; **derived deterministically from the API key** (see [Device fingerprint](#device-fingerprint-fpmode--cc_fp_mode-fpsalt--cc_fp_salt)), SHA-256 hashed, per-key binding, refreshed every 8h + 2h jitter | | **Lifecycle Events** | `POST /alpha/lifecycle-events` (`cli_session_exists`) sent in parallel with fingerprint on session init | | **Per-Key Session** | One session per API key, 12h expiry + 1h random jitter | | **Version** | `x-command-code-version` auto-fetched from npm registry (24h refresh) | @@ -491,6 +518,8 @@ npm run docker:build:multi | `PROXY_PORT` | `3050` | Host port (compose only) | | `CC_MAX_BODY_MB` | `100` | Max request body size in MB; oversized requests are rejected with `HTTP 413` | | `CC_UPSTREAM_PROXY` | *(unset)* | `http://host:port` CONNECT proxy for upstream Command Code requests only | +| `CC_FP_MODE` | `derived` | `derived` = stable per-key device fingerprint; `random` = old behaviour | +| `CC_FP_SALT` | *(unset)* | Salt for fingerprint derivation; isolates devices between deployments | | `CC_CLIENT_DRAIN_TIMEOUT_MS` | *(unset = disabled)* | Drop the client and abort upstream when downstream backpressure blocks longer than this; see [Stalled clients](#stalled-clients-neither-reading-nor-disconnecting) | | `CC_STREAM_IDLE_MS` | `30000` | Streaming upstream read idle timeout in ms; see [Upstream idle timeouts](#upstream-idle-timeouts) | | `CC_NONSTREAM_IDLE_MS` | `90000` | Non-streaming upstream read idle timeout in ms | @@ -624,7 +653,7 @@ A more robust cap still belongs at the reverse proxy (`limit_conn`), since only - **`logFile` uses `appendFileSync`** — synchronous writes on the event loop. Under public load they serialize the loop; prefer leaving it empty and collecting stdout. - **systemd guard rails**: set `MemoryMax=` and `NODE_OPTIONS=--max-old-space-size=` so an overshoot kills the proxy, not `sshd`/`nginx`. -- **Multi-account + multiple instances**: `sessionStore` / `keyStateStore` are per-process `Map`s, so the same API key served by two instances gets two different sessions and **two different device fingerprints** — upstream sees one account on multiple machines. Scale with consistent hashing on the API key (`hash $cc_key consistent`), not round-robin. +- **Multi-account + multiple instances**: `sessionStore` is still a per-process `Map`, so the same API key served by two instances gets two different sessions. **The device fingerprint is no longer affected** — it is derived, so it is the same machine across instances and restarts (see [Device fingerprint](#device-fingerprint-fpmode--cc_fp_mode-fpsalt--cc_fp_salt)). Consistent hashing on the API key (`hash $cc_key consistent`) is still recommended to keep session affinity, rather than round-robin. ## Disclaimer diff --git a/README_zh.md b/README_zh.md index 60c6605..f90bc95 100644 --- a/README_zh.md +++ b/README_zh.md @@ -77,6 +77,8 @@ commandcode/ | `CC_MAX_INFLIGHT` | 进程内在途请求上限(默认 `0` = 不限)| | `CMD_ZDR` | `zdr`(`1` 开启) | | `CC_UPSTREAM_PROXY` | `upstreamProxy` | +| `CC_FP_MODE` | `fpMode`(默认 `derived`,`random` 回退) | +| `CC_FP_SALT` | `fpSalt` | 开启后,代理会在 Command Code 生成请求以及 fingerprint/lifecycle 初始化请求中附加 `x-cmd-zdr: 1`。npm 版本检查和代理自己的 `/provider/v1/models` 模型目录请求不会附加该 @@ -86,6 +88,31 @@ header。该开关只是请求 Command Code 使用 ZDR-only 路由,实际数 > ⚠️ **内存放大**:请求体在转发到上游前会存在多份副本,实测峰值 ≈ body 大小 × **5.1~7.4**(7MB→+52MB、20MB→+116MB;被 `413` 拒绝的请求只要 ×1.05)。因此默认 `CC_MAX_BODY_MB=100` 意味着**单个请求**最坏可吃 ~550MB,且该上限是每请求的、不是全局的。详见[内存与部署](#内存与部署)。 +### 设备指纹(`fpMode` / `CC_FP_MODE`,`fpSalt` / `CC_FP_SALT`) + +上报给 `/alpha/fingerprint/record` 的设备指纹,现在由 API key **确定性派生**(`HMAC-SHA256(CC_FP_SALT, apiKey)`),因此一个 key 恒定对应一台设备: + +| 事件 | 原行为(随机) | 现行为(派生) | +|---|---|---| +| 进程重启 | Map 清空 → **换一台机器** | 同一台机器 | +| 第二个实例 | 同一 key = **两台机器** | 同一台机器 | +| session 过期(12h) | `keyStateStore.delete` → **每 12h 换一台机器** | 同一台机器 | + +> 12h 那条最明显:真实用户不会一天换两次电脑。而上游 `device_fingerprints` 表是按 `(userId, thumbmark)` 建唯一索引的。 + +**为什么用「派生」而不是「按哈希取桶选设备」** —— 固定池的熵上限就是池的大小,key 数一旦超过池容量,多个 key 就**必然**共用指纹:约 50 个 key 配 1000 个池 → 约 2 个碰撞;配 100 个池 → 约 20 个碰撞。同一个 `thumbmark` 出现在两个不同 `userId` 下,就是「多账号同机」的直接证据 —— 这正是最不该主动制造的东西。派生方案每个 key 仍是独立设备(碰撞概率 2⁻²⁵⁶),同时保持稳定。 + +```bash +CC_FP_SALT=some-local-secret npm start # 可选:隔离不同部署的指纹 +CC_FP_MODE=random npm start # 回退:恢复原「每进程随机」行为 +``` + +盐是可选的但建议设:不设时派生是 API key 的纯函数,知道算法的人可以反推出你所有用户的指纹;设了之后同一个 key 在不同部署上得到不同设备,且没有额外成本。 + +哈希构造对齐官方 CLI(`command-code` 的 `buildMachineFingerprint` / `hashSignal`):`thumbmark = sha256(IB + "\0machine\0" + [machineId, macs.join(",")].join("|"))`,其中 `IB = "command-code:device-fingerprint:v1"`;各 component 按 `sha256(IB + "\0" + value.toLowerCase())` 计算。原实现直接对随机 hex 求 sha256(缺 `IB` 前缀),且 thumbmark 由各 component 的**哈希**拼成 —— 上游两种都无从验算(它拿不到原始 `machineId`),所以检测不到;现在已对齐。 + +> 本次未处理:外观池仍是清一色高端桌面 CPU,时区仍从全球池里取。在单出口 IP 的部署上,一批散布在 15 个时区、型号又高度相似的机器,不是一个像真实用户的分布。把 `timezone` 绑定到出口 IP 是自然的下一步,但这取决于具体部署,留给运维决定。 + ### 上游代理(`upstreamProxy` / `CC_UPSTREAM_PROXY`) 让代理**发往 Command Code 的请求**走本地 HTTP 代理 —— 用于出口地区调整,或排查风控 `403` 时做 IP 维度对照。 @@ -378,7 +405,7 @@ Anthropic SDK 通过 `x-api-key` 头鉴权——代理已原生支持(无需 ` | 机制 | 实现 | |------|------| -| **设备指纹** | 每个 Key 首次请求前发送 `POST /alpha/fingerprint/record`;随机指纹池(15 种 CPU、全球时区)、SHA-256 哈希、per-key 绑定,每 8h+2h 抖动刷新 | +| **设备指纹** | 每个 Key 首次请求前发送 `POST /alpha/fingerprint/record`;**由 API key 确定性派生**(见[设备指纹](#设备指纹fpmode--cc_fp_modefpsalt--cc_fp_salt))、SHA-256 哈希、per-key 绑定,每 8h+2h 抖动刷新 | | **生命周期声明** | 会话初始化时与指纹并行发送 `POST /alpha/lifecycle-events`(`cli_session_exists`) | | **按 Key 分 Session** | 每个 API Key 独立 session,12h 过期 + 1h 随机抖动 | | **动态版本号** | `x-command-code-version` 从 npm registry 自动拉取(24h 刷新) | @@ -489,6 +516,8 @@ npm run docker:build:multi | `PROXY_PORT` | `3050` | 主机映射端口(仅 compose) | | `CC_MAX_BODY_MB` | `100` | 请求体大小上限(MB),超限请求返回 `HTTP 413` | | `CC_UPSTREAM_PROXY` | 空 | 仅作用于 CC 上游请求的 `http://host:port` CONNECT 代理 | +| `CC_FP_MODE` | `derived` | `derived` = 每 key 稳定设备指纹;`random` = 原行为 | +| `CC_FP_SALT` | 空 | 指纹派生用的盐;隔离不同部署的设备 | | `CC_CLIENT_DRAIN_TIMEOUT_MS` | 空(禁用)| 下游背压阻塞超过该毫秒数则断开该客户端并中止上游请求,见[僵死连接](#僵死连接既不读也不断开) | | `CC_STREAM_IDLE_MS` | `30000` | 流式上游读空闲超时(毫秒),见[上游空闲超时](#上游空闲超时) | | `CC_NONSTREAM_IDLE_MS` | `90000` | 非流式上游读空闲超时(毫秒)| @@ -627,7 +656,7 @@ CC_CLIENT_DRAIN_TIMEOUT_MS=60000 npm start - **`logFile` 是同步写**(`appendFileSync`),公网负载下会阻塞事件循环 —— 建议保持留空,从 stdout 收集。 - **systemd 兜底**:配 `MemoryMax=` 与 `NODE_OPTIONS=--max-old-space-size=`,让超限杀掉 proxy 而不是 `sshd`/`nginx`。 -- **多账号 + 多实例**:`sessionStore` / `keyStateStore` 是进程内 `Map`。同一个 API key 打到两个实例会得到两个不同 session 与**两个不同设备指纹**,上游会看到「一个账号在多台机器上」。横向扩展请按 API key 做一致性哈希(`hash $cc_key consistent`),不要轮询。 +- **多账号 + 多实例**:`sessionStore` 仍是进程内 `Map`,同一个 API key 打到两个实例会得到两个不同 session。**设备指纹已不再是问题** —— 它现在是派生出来的,跨实例、跨重启都是同一台设备(见[设备指纹](#设备指纹fpmode--cc_fp_modefpsalt--cc_fp_salt))。横向扩展仍建议按 API key 做一致性哈希(`hash $cc_key consistent`)以保持 session 亲和,不要轮询。 ## 免责声明 diff --git a/config.json b/config.json index 0d8ece2..d163bae 100644 --- a/config.json +++ b/config.json @@ -7,5 +7,7 @@ "logFile": "", "logLevel": "info", "zdr": false, - "upstreamProxy": "" + "upstreamProxy": "", + "fpMode": "derived", + "fpSalt": "" } diff --git a/proxy.mjs b/proxy.mjs index 5674cea..40a4792 100644 --- a/proxy.mjs +++ b/proxy.mjs @@ -28,6 +28,8 @@ function loadConfig() { zdr: false, emptySystemPlaceholder: true, // 无 system prompt 时发空格占位,阻止 CC 上游注入 ~7.5K token 默认提示词(issue #17) upstreamProxy: '', // 上游 HTTP 代理,如 http://127.0.0.1:7890(issue #18) + fpMode: 'derived', // 指纹来源:derived = 由 API key 确定性派生(默认),random = 每进程随机 + fpSalt: '', // 派生用的本地盐;不设也能用,设了可隔离不同部署 }; const configPath = resolve(__dirname, 'config.json'); @@ -50,13 +52,15 @@ function loadConfig() { if (process.env.CMD_ZDR !== undefined) defaults.zdr = process.env.CMD_ZDR === '1'; if (process.env.CC_EMPTY_SYSTEM_PLACEHOLDER) defaults.emptySystemPlaceholder = process.env.CC_EMPTY_SYSTEM_PLACEHOLDER !== 'false'; if (process.env.CC_UPSTREAM_PROXY) defaults.upstreamProxy = process.env.CC_UPSTREAM_PROXY; + if (process.env.CC_FP_MODE) defaults.fpMode = process.env.CC_FP_MODE === 'random' ? 'random' : 'derived'; + if (process.env.CC_FP_SALT) defaults.fpSalt = process.env.CC_FP_SALT; return defaults; } const CFG = loadConfig(); -// ── 指纹生成(首次运行自动生成,写回 config.json) ────── +// ── 设备指纹 ────────────────────────────────────── // CPU 型号与核心数对应表(仅 Windows x64) const FINGERPRINT_CPUS = [ { model: '12th Gen Intel(R) Core(TM) i7-12650H', cores: 10 }, @@ -84,35 +88,82 @@ const FINGERPRINT_TZS = [ ]; const FINGERPRINT_MAC_COUNT_RANGE = [2, 3, 4, 5]; // 随机 2~5 个 MAC -function generateFingerprint() { - const cpuEntry = FINGERPRINT_CPUS[Math.floor(Math.random() * FINGERPRINT_CPUS.length)]; - const memGiB = FINGERPRINT_MEMS[Math.floor(Math.random() * FINGERPRINT_MEMS.length)]; - const tz = FINGERPRINT_TZS[Math.floor(Math.random() * FINGERPRINT_TZS.length)]; - const macCount = FINGERPRINT_MAC_COUNT_RANGE[Math.floor(Math.random() * FINGERPRINT_MAC_COUNT_RANGE.length)]; - - function sha256(s) { return crypto.createHash('sha256').update(s).digest('hex'); } - function randHex(n) { return crypto.randomBytes(n).toString('hex'); } - - const macHashes = []; - for (let i = 0; i < macCount; i++) macHashes.push(sha256(randHex(32))); - - const machineIdHash = sha256(randHex(32)); - const osUserHash = sha256(randHex(16)); - const hostnameHash = sha256(randHex(16)); - const gitEmailHash = sha256(randHex(16)); +// 官方 CLI 的指纹算法常量(command-code 1.53.0 dist/cli.mjs,函数 +// buildMachineFingerprint / hashSignal / gatherRawSignals): +// +// const ib = "command-code:device-fingerprint:v1"; +// hashSignal(v) = sha256(ib + "\0" + v.trim().toLowerCase()) // v 是原始值,不是哈希 +// thumbmark = sha256(ib + "\0machine\0" + [machineId, macs.join(",")].join("|")) +// 其中 machineId 非空时 hostname / cpuModel 不参与 +// +// 原实现直接对随机 hex 求 sha256,缺 ib 前缀,且 thumbmark 的输入结构也不同 +// (把各 component 的哈希再拼起来)。上游没有原始 machineId、无法重算,这处 +// 不一致检测不到 —— 但既然要动,就一次对齐。 +const FINGERPRINT_IB = 'command-code:device-fingerprint:v1'; - // thumbmark = 所有组件的联合哈希 - const thumbData = [machineIdHash, ...macHashes, osUserHash, hostnameHash, gitEmailHash, 'win32', '10.0.22631', cpuEntry.model, String(cpuEntry.cores), String(memGiB)].join('|'); - const thumbmark = sha256(thumbData); +/** + * 生成一台"设备"的指纹。 + * + * 默认(fpMode=derived)由 HMAC(CC_FP_SALT, apiKey) 确定性派生:同一个 key + * 在进程重启后、在多实例上都会得到同一台设备。这修掉原实现的三处漂移: + * ① 进程重启 → Map 清空 → 同一 key 换新机器 + * ② 多实例 → 同一 key 在不同实例是两台机器 + * ③ session 过期清理连带 keyStateStore.delete → 每 12h 换一台机器 + * + * 这与"按 key 取哈希桶选设备"有本质区别:桶方案的熵上限就是桶数,key 数一旦 + * 超过桶数就必然出现多个 key 共用指纹,而上游 device_fingerprints 表对 + * (userId, thumbmark) 建了唯一索引 —— 共用指纹等于"多账号同机"的直接证据。 + * 派生方案每个 key 都是独立设备,碰撞概率 2^-256。 + * + * fpMode=random 保留原随机行为(每进程一台新设备),用于回退。 + */ +function deriveFingerprint(apiKey) { + const random = CFG.fpMode === 'random'; + const salt = CFG.fpSalt || ''; + // seed 只用来挑外观参数;token 用来造原始信号值 + const seed = random ? crypto.randomBytes(32) : crypto.createHmac('sha256', salt).update(apiKey).digest(); + const token = random + ? (_domain, bytes) => crypto.randomBytes(bytes).toString('hex') + : (domain, bytes) => crypto.createHmac('sha256', salt).update(apiKey).update('\0').update(domain).digest('hex').slice(0, bytes * 2); + const at = (off, mod) => seed.readUInt32BE(off % 28) % mod; + const sha256 = s => crypto.createHash('sha256').update(s).digest('hex'); + + const cpuEntry = FINGERPRINT_CPUS[at(0, FINGERPRINT_CPUS.length)]; + const memGiB = FINGERPRINT_MEMS[at(4, FINGERPRINT_MEMS.length)]; + const tz = FINGERPRINT_TZS[at(8, FINGERPRINT_TZS.length)]; + const macCount = FINGERPRINT_MAC_COUNT_RANGE[at(12, FINGERPRINT_MAC_COUNT_RANGE.length)]; + + // 原始信号值:只存在于本进程,出网的永远只有它们的哈希 + const machineId = token('machineId', 16); + // 与 CLI 同构:[...new Set(list.map(x => x.toLowerCase()))].filter(Boolean).sort() + const macs = [...new Set( + Array.from({ length: macCount }, (_v, i) => token('mac:' + i, 6).match(/.{2}/g).join(':').toLowerCase()), + )].filter(Boolean).sort(); + const osUser = 'dev' + token('osUser', 2); + const hostname = 'DESKTOP-' + token('hostname', 4).toUpperCase(); + const gitEmail = token('gitEmail', 6) + '@example.com'; + + // ── 以下与官方 CLI 逐行同构 ── + const hashSignal = v => { + const t = String(v).trim(); + return t ? sha256(FINGERPRINT_IB + '\0' + t.toLowerCase()) : undefined; + }; + const thumbParts = [ + machineId.trim(), + macs.join(','), + machineId.trim() ? '' : hostname.trim(), // machineId 非空 → 这两项不参与 + machineId.trim() ? '' : cpuEntry.model.trim(), + ].filter(Boolean); + const thumbmark = sha256(FINGERPRINT_IB + '\0machine\0' + (thumbParts.join('|') || 'unknown')); return { thumbmark, components: { - machineIdHash, - macHashes, - osUserHash, - hostnameHash, - gitEmailHash, + machineIdHash: hashSignal(machineId), + macHashes: macs.map(hashSignal).filter(Boolean), + osUserHash: hashSignal(osUser), + hostnameHash: hashSignal(hostname), + gitEmailHash: hashSignal(gitEmail), platform: 'win32', arch: 'x64', osRelease: '10.0.22631', @@ -251,14 +302,17 @@ function ensureSession(apiKey) { return sessionId; } -// 定期清理过期 session 和 key 状态,防止 Map 无限增长 +// 定期清理过期 session,防止 Map 无限增长。 +// 注意:这里**不再**连带删除 keyStateStore。原实现写的是「同时清理该 key 的指纹状态」, +// 但指纹是随机生成的,删掉就等于该 key 每 12h 换一台"电脑" —— 真实用户不会这样。 +// 指纹状态现在有自己的空闲淘汰(见下方),且因为指纹是派生出来的, +// 即使被淘汰、下次重新派生得到的仍是同一台设备,不构成漂移。 setInterval(() => { const now = Date.now(); let cleaned = 0; for (const [key, entry] of sessionStore) { if (now >= entry.expiresAt) { sessionStore.delete(key); - keyStateStore.delete(key); // 同时清理该 key 的指纹状态 cleaned++; } } @@ -289,22 +343,40 @@ function isWireUuid(v) { } // ── 每 Key 独立状态(fingerprint + 初始化节流) ── -// 每个 API Key 拥有自己的设备指纹和初始化定时器 -const keyStateStore = new Map(); // apiKey → { fingerprint, nextInitAt } +// 指纹现在是确定性派生的,所以这个 Map 只是缓存:淘汰它不会改变该 key 的设备身份, +// 只是下次多算一次 HMAC。淘汰按「空闲」而非「session 过期」判定, +// 这样活跃 key 的 nextInitAt 不会被重置(否则会重复发送指纹/lifecycle 预请求)。 +const KEY_STATE_IDLE_MS = 24 * 60 * 60 * 1000; // 24h 未活动即淘汰 +const keyStateStore = new Map(); // apiKey → { fingerprint, nextInitAt, lastSeen } function getOrCreateKeyState(apiKey) { let state = keyStateStore.get(apiKey); if (!state) { state = { - fingerprint: generateFingerprint(), + fingerprint: deriveFingerprint(apiKey), nextInitAt: 0, + lastSeen: 0, }; keyStateStore.set(apiKey, state); - log('info', 'Fingerprint generated for key', { keyPrefix: apiKey.slice(0, 8) }); + log('info', 'Fingerprint ' + (CFG.fpMode === 'random' ? 'generated' : 'derived') + ' for key', { + keyPrefix: apiKey.slice(0, 8), + thumbmark: state.fingerprint.thumbmark.slice(0, 12), + }); } + state.lastSeen = Date.now(); return state; } +// 空闲淘汰:纯内存卫生,不影响设备身份(见上) +setInterval(() => { + const now = Date.now(); + let evicted = 0; + for (const [key, state] of keyStateStore) { + if (now - state.lastSeen > KEY_STATE_IDLE_MS) { keyStateStore.delete(key); evicted++; } + } + if (evicted > 0) log('info', 'Key state evicted (idle)', { evicted, remaining: keyStateStore.size }); +}, 60 * 60 * 1000); // 每小时 + // ── 初始化预请求(fingerprint + lifecycle,首次 + 每 8h+2h 抖动) ──── const INIT_REFRESH_MS = 8 * 60 * 60 * 1000; // 8h const INIT_JITTER_MS = 2 * 60 * 60 * 1000; // 2h 抖动 @@ -3082,6 +3154,9 @@ server.listen(CFG.port, CFG.host, () => { idleTimeouts: `stream ${STREAM_IDLE_TIMEOUT_MS}ms / nonstream ${NONSTREAM_IDLE_TIMEOUT_MS}ms`, maxInflight: MAX_INFLIGHT > 0 ? `${MAX_INFLIGHT} (global, /health exempt)` : 'unlimited (CC_MAX_INFLIGHT=0)', upstreamProxy: UPSTREAM_PROXY || '(direct)', + fingerprint: CFG.fpMode === 'random' + ? 'random per key per process (CC_FP_MODE=random)' + : `derived from API key (salt: ${CFG.fpSalt ? 'set' : 'unset'})`, }); if (CLIENT_DRAIN_TIMEOUT_MS > 0) { log('info', 'Client drain timeout enabled', { timeoutMs: CLIENT_DRAIN_TIMEOUT_MS }); From e167a72603109d76d68b4c8f0566106d65482ebc Mon Sep 17 00:00:00 2001 From: xelr233 Date: Fri, 11 Sep 2026 00:06:22 +0800 Subject: [PATCH 05/21] =?UTF-8?q?docs:=20=E4=BF=AE=E6=AD=A3=E6=97=B6?= =?UTF-8?q?=E5=8C=BA=E5=BB=BA=E8=AE=AE=20=E2=80=94=E2=80=94=20=E5=AE=83?= =?UTF-8?q?=E6=98=AF=E5=AE=A2=E6=88=B7=E7=AB=AF=E6=9C=AC=E6=9C=BA=E6=97=B6?= =?UTF-8?q?=E5=8C=BA=EF=BC=8C=E4=B8=8D=E5=BA=94=E7=BB=91=E5=AE=9A=E5=87=BA?= =?UTF-8?q?=E5=8F=A3=20IP?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 原文档写「把 timezone 绑定到出口 IP 是自然的下一步」,这是错的。 command-code CLI 的 readTimezone() 是 Intl.DateTimeFormat().resolvedOptions().timeZone 即**客户端本机操作系统的时区**,与流量从哪个 IP 出去无关。 中国大陆用户通过代理访问本服务时,本机时区与出口地不一致是常态。 真正有意义的性质是时区在**自身用户群**里的分布,而不是与出口 IP 是否一致: 单一地区用户群若从 15 个全球时区里均匀取,会让每个账号看起来来自不同大洲。 正确做法是让池子匹配实际使用该部署的人群(收窄或加权 FINGERPRINT_TZS), 这是运维决策,不是「对齐 IP」。 --- README.md | 6 +++++- README_zh.md | 6 +++++- 2 files changed, 10 insertions(+), 2 deletions(-) diff --git a/README.md b/README.md index 268d51a..3f17a4c 100644 --- a/README.md +++ b/README.md @@ -113,7 +113,11 @@ The salt is optional but recommended: without it the derivation is a pure functi The hash construction follows the official CLI (`buildMachineFingerprint` / `hashSignal` in `command-code`) — `thumbmark = sha256(IB + "\0machine\0" + [machineId, macs.join(",")].join("|"))` with `IB = "command-code:device-fingerprint:v1"`, and each component hashed as `sha256(IB + "\0" + value.toLowerCase())`. The previous implementation hashed random hex without the `IB` prefix and built the thumbmark from the component *hashes*; upstream cannot recompute either way (it never sees the raw `machineId`), so it was undetectable — but it is now aligned. -> Not addressed here: the appearance pool is still all high-end desktop CPUs and the timezone is still drawn from a global pool. On a single-egress-IP deployment, a cluster of machines spread across 15 timezones is a distribution that does not look like real users. Binding `timezone` to the egress IP is the natural next step, but it depends on deployment specifics, so it is left to the operator. +> Not addressed here: the appearance pool is still all high-end desktop CPUs and the timezone is drawn uniformly from a global pool. +> +> `timezone` is the **client machine's OS timezone** (`Intl.DateTimeFormat().resolvedOptions().timeZone`), *not* the egress IP's. So do **not** bind it to the egress IP: a user in mainland China reaching this service through a proxy normally has a machine timezone that does not match where the traffic exits, and that mismatch is the norm rather than an anomaly. +> +> The property that matters is therefore the **distribution across your own user base**, not agreement with the IP. If your users are concentrated in one region, drawing timezones uniformly from 15 global zones makes every account look like it belongs to a different continent. Set the pool to match who actually uses the deployment — this is an operator decision, and for a single-region user base it means narrowing (or weighting) `FINGERPRINT_TZS` rather than randomising it globally. ### Upstream proxy (`upstreamProxy` / `CC_UPSTREAM_PROXY`) diff --git a/README_zh.md b/README_zh.md index f90bc95..cc4f9d6 100644 --- a/README_zh.md +++ b/README_zh.md @@ -111,7 +111,11 @@ CC_FP_MODE=random npm start # 回退:恢复原「每进程随机 哈希构造对齐官方 CLI(`command-code` 的 `buildMachineFingerprint` / `hashSignal`):`thumbmark = sha256(IB + "\0machine\0" + [machineId, macs.join(",")].join("|"))`,其中 `IB = "command-code:device-fingerprint:v1"`;各 component 按 `sha256(IB + "\0" + value.toLowerCase())` 计算。原实现直接对随机 hex 求 sha256(缺 `IB` 前缀),且 thumbmark 由各 component 的**哈希**拼成 —— 上游两种都无从验算(它拿不到原始 `machineId`),所以检测不到;现在已对齐。 -> 本次未处理:外观池仍是清一色高端桌面 CPU,时区仍从全球池里取。在单出口 IP 的部署上,一批散布在 15 个时区、型号又高度相似的机器,不是一个像真实用户的分布。把 `timezone` 绑定到出口 IP 是自然的下一步,但这取决于具体部署,留给运维决定。 +> 本次未处理:外观池仍是清一色高端桌面 CPU,时区仍从全球池里均匀取。 +> +> `timezone` 是**客户端本机操作系统的时区**(`Intl.DateTimeFormat().resolvedOptions().timeZone`),**不是出口 IP 的时区**。所以**不要**把它绑定到出口 IP:中国大陆用户通过代理访问本服务时,本机时区与流量出口地不一致是**常态而非异常**。 +> +> 真正有意义的性质是**它在你自身用户群里的分布**,而不是与 IP 是否一致。如果你的用户集中在一个地区,却从 15 个全球时区里均匀取,就会让每个账号看起来来自不同的大洲。应当让池子匹配实际使用这个部署的人群 —— 这是运维决策:单一地区用户群应当收窄(或加权)`FINGERPRINT_TZS`,而不是全球随机。 ### 上游代理(`upstreamProxy` / `CC_UPSTREAM_PROXY`) From d96127672bbe5f60d3719c02c2aaee6034f17da2 Mon Sep 17 00:00:00 2001 From: xelr233 Date: Sun, 13 Sep 2026 04:35:40 +0800 Subject: [PATCH 06/21] =?UTF-8?q?test:=20fork=20=E4=B8=93=E5=B1=9E?= =?UTF-8?q?=E8=A1=8C=E4=B8=BA=E7=9A=84=E5=9B=9E=E5=BD=92=E6=8A=A4=E6=A0=8F?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 上游有测试之前,fork 的这些改动只能靠手动脚本验证。现在把它们纳入 CI, 避免在上游同步时被静默覆盖。 新增 test/fork.test.mjs(13 条): 逆向对齐(对应 e3e267e) - 上游请求 User-Agent 为 cli(CLI 源码常量 vy = "cli") - fingerprint/lifecycle 预请求同样是 cli - threadId 与 x-session-id 同值 - session 非 UUID 时省略 threadId(CLI toWireThreadId 行为) 上游代理(对应 #18) - CC_UPSTREAM_PROXY 配置后经 CONNECT 隧道 - 预请求也走代理(否则同账号会从两个 IP 注册) - 未配置时不建立任何 CONNECT - /health 不经过上游代理 指纹派生(对应 c153a76) - 同 key 同盐重启后 thumbmark 稳定 - 不同 key 得到不同 thumbmark(不退化成哈希桶) - 不同盐得到不同部署指纹 - CC_FP_MODE=random 可回退到原行为 测试用录制型 CONNECT 代理(只转发裸字节),不需要真出口代理。 全部走 loopback,无凭据、不访问外部服务。 --- .github/workflows/test.yml | 44 +++++++ package.json | 1 + test/endpoints.test.mjs | 103 +++++++++++++++++ test/errors-limits.test.mjs | 143 +++++++++++++++++++++++ test/fingerprint.test.mjs | 102 +++++++++++++++++ test/fork.test.mjs | 223 ++++++++++++++++++++++++++++++++++++ test/helpers.mjs | 125 ++++++++++++++++++++ test/regressions.test.mjs | 69 +++++++++++ test/wire.test.mjs | 127 ++++++++++++++++++++ 9 files changed, 937 insertions(+) create mode 100644 .github/workflows/test.yml create mode 100644 test/endpoints.test.mjs create mode 100644 test/errors-limits.test.mjs create mode 100644 test/fingerprint.test.mjs create mode 100644 test/fork.test.mjs create mode 100644 test/helpers.mjs create mode 100644 test/regressions.test.mjs create mode 100644 test/wire.test.mjs diff --git a/.github/workflows/test.yml b/.github/workflows/test.yml new file mode 100644 index 0000000..5ddd443 --- /dev/null +++ b/.github/workflows/test.yml @@ -0,0 +1,44 @@ +name: Test + +on: + push: + # 只在这些长期分支上跑 push;其余分支靠 pull_request 触发,避免同一提交跑两遍 + branches: [master, release] + pull_request: + workflow_dispatch: + +# 同一分支的新推送取消旧运行,避免排队浪费 +concurrency: + group: ${{ github.workflow }}-${{ github.ref }} + cancel-in-progress: true + +permissions: + contents: read + +jobs: + test: + runs-on: ubuntu-latest + timeout-minutes: 10 + strategy: + fail-fast: false + matrix: + # engines 下限是 18;Dockerfile 用 22;中间放 20 覆盖 LTS 跨度 + node: ['18', '20', '22'] + name: node ${{ matrix.node }} + steps: + - name: Checkout + uses: actions/checkout@v4 + + - name: Set up Node.js ${{ matrix.node }} + uses: actions/setup-node@v4 + with: + node-version: ${{ matrix.node }} + + - name: Show Node version + run: node --version + + - name: Syntax check + run: node --check proxy.mjs + + - name: Run tests + run: npm test diff --git a/package.json b/package.json index 8227668..e5bc345 100644 --- a/package.json +++ b/package.json @@ -8,6 +8,7 @@ "scripts": { "start": "node proxy.mjs", "dev": "node --watch proxy.mjs", + "test": "node --test test/*.test.mjs", "docker:build": "docker build -t commandcode-proxy:latest .", "docker:build:multi": "docker buildx build --platform linux/amd64,linux/arm64 -t commandcode-proxy:latest ." }, diff --git a/test/endpoints.test.mjs b/test/endpoints.test.mjs new file mode 100644 index 0000000..878adde --- /dev/null +++ b/test/endpoints.test.mjs @@ -0,0 +1,103 @@ +// 端点契约:四个路由在正常路径下的行为。 +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import { setup } from './helpers.mjs'; + +test('POST /v1/chat/completions 流式:返回 OpenAI SSE 且内容正确', async () => { + const s = await setup(); + try { + const r = await s.proxy.post('/v1/chat/completions', + { model: 'deepseek/deepseek-v4-flash', messages: [{ role: 'user', content: 'hi' }], stream: true }, + { Authorization: 'Bearer user_test' }); + assert.equal(r.status, 200); + const text = await r.text(); + assert.match(text, /data: /); + assert.ok(text.includes('hello'), 'SSE 应包含上游 text-delta 的内容'); + assert.match(text, /\[DONE\]/, '流应以 [DONE] 结束'); + } finally { await s.close(); } +}); + +test('POST /v1/chat/completions 非流式:返回 chat.completion 对象', async () => { + const s = await setup(); + try { + const r = await s.proxy.post('/v1/chat/completions', + { model: 'deepseek/deepseek-v4-flash', messages: [{ role: 'user', content: 'hi' }] }, + { Authorization: 'Bearer user_test' }); + assert.equal(r.status, 200); + const j = await r.json(); + assert.equal(j.object, 'chat.completion'); + assert.equal(j.choices[0].message.content, 'hello'); + assert.ok(j.usage, 'usage 必须存在'); + } finally { await s.close(); } +}); + +test('POST /v1/messages 流式:返回 Anthropic SSE 事件序列', async () => { + const s = await setup(); + try { + const r = await s.proxy.post('/v1/messages', + { model: 'deepseek/deepseek-v4-flash', max_tokens: 100, messages: [{ role: 'user', content: 'hi' }], stream: true }, + { 'x-api-key': 'user_test' }); + assert.equal(r.status, 200); + const text = await r.text(); + for (const ev of ['message_start', 'content_block_start', 'message_stop']) { + assert.ok(text.includes(ev), 'Anthropic SSE 应包含 ' + ev); + } + } finally { await s.close(); } +}); + +test('POST /v1/responses 流式:返回具名 SSE 事件', async () => { + const s = await setup(); + try { + const r = await s.proxy.post('/v1/responses', + { model: 'deepseek/deepseek-v4-flash', input: 'hi', stream: true }, + { Authorization: 'Bearer user_test' }); + assert.equal(r.status, 200); + const text = await r.text(); + assert.ok(text.includes('response.completed'), '应包含 response.completed'); + // 规范要求每个事件都带 sequence_number + assert.ok(text.includes('sequence_number'), '每个事件都必须带 sequence_number'); + } finally { await s.close(); } +}); + +test('GET /v1/models 走 /provider/v1/models', async () => { + const s = await setup({ env: { CC_USE_PROVIDER_MODELS: 'true' }, onRequest: (req, res) => { + if (req.url === '/provider/v1/models') { + res.writeHead(200, { 'Content-Type': 'application/json' }); + res.end(JSON.stringify({ data: [{ id: 'test/model-1', object: 'model' }] })); + } + }}); + try { + const r = await s.proxy.get('/v1/models', { headers: { Authorization: 'Bearer user_test' } }); + assert.equal(r.status, 200); + const j = await r.json(); + assert.ok(Array.isArray(j.data)); + assert.equal(j.data[0].id, 'test/model-1'); + } finally { await s.close(); } +}); + +test('GET /health 与 / 返回存活状态', async () => { + const s = await setup(); + try { + for (const path of ['/health', '/']) { + const r = await s.proxy.get(path); + assert.equal(r.status, 200, path + ' 应返回 200'); + } + } finally { await s.close(); } +}); + +test('未知路由返回 404', async () => { + const s = await setup(); + try { + const r = await s.proxy.get('/nope'); + assert.equal(r.status, 404); + } finally { await s.close(); } +}); + +test('缺少 API key 返回 401', async () => { + const s = await setup(); + try { + const r = await s.proxy.post('/v1/chat/completions', + { model: 'deepseek/deepseek-v4-flash', messages: [{ role: 'user', content: 'hi' }] }); + assert.equal(r.status, 401); + } finally { await s.close(); } +}); diff --git a/test/errors-limits.test.mjs b/test/errors-limits.test.mjs new file mode 100644 index 0000000..30b2129 --- /dev/null +++ b/test/errors-limits.test.mjs @@ -0,0 +1,143 @@ +// 错误映射 + 请求体上限 + 在途上限。 +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import { setup } from './helpers.mjs'; + +const AUTH = { Authorization: 'Bearer user_test' }; +const CHAT = { model: 'm', stream: true, messages: [{ role: 'user', content: 'hi' }] }; + +// CC_STATUS_MAP 的语义契约(逐条来自 proxy.mjs 的映射表): +// 400/401/404 原样透传;422 -> 400;403 -> 401;402 -> 429(payment → rate limit) +// 500/502 -> 502;503 -> 503;未列出的状态 -> 502 upstream_error +const MAPPING = [ + [400, 400], [401, 401], [404, 404], + [422, 400], [403, 401], [402, 429], + [500, 502], [502, 502], [503, 503], + [418, 502], // 未在表中 → upstream_error +]; + +for (const [upstream, expected] of MAPPING) { + test('状态映射:上游 ' + upstream + ' -> 下游 ' + expected, async () => { + const s = await setup({ status: upstream, errorBody: JSON.stringify({ error: { message: 'mock' } }) }); + try { + const r = await s.proxy.post('/v1/chat/completions', CHAT, AUTH); + assert.equal(r.status, expected, '上游 ' + upstream + ' 应映射为 ' + expected); + const j = await r.json(); + assert.ok(j.error && j.error.type, '响应体应含 error.type'); + } finally { await s.close(); } + }); +} + +test('上游 402(payment required)映射为 429 而非 402', async () => { + const s = await setup({ status: 402, errorBody: JSON.stringify({ error: { message: 'insufficient credits' } }) }); + try { + const r = await s.proxy.post('/v1/chat/completions', CHAT, AUTH); + assert.equal(r.status, 429, '402 是 CC 的余额语义,下游按限流处理'); + const j = await r.json(); + assert.match(j.error.message, /insufficient credits/, '上游错误体应透传进 message'); + } finally { await s.close(); } +}); + +test('上游 429 映射为 rate_limit_error 并带 retry_after', async () => { + const s = await setup({ status: 429, errorBody: JSON.stringify({ error: { message: 'slow down' } }) }); + try { + const r = await s.proxy.post('/v1/chat/completions', CHAT, AUTH); + const j = await r.json(); + assert.equal(r.status, 429); + assert.equal(j.error.type, 'rate_limit_error'); + } finally { await s.close(); } +}); + +test('上游 5xx 映射为 502/503 类服务端错误', async () => { + const s = await setup({ status: 500, errorBody: JSON.stringify({ error: { message: 'boom' } }) }); + try { + const r = await s.proxy.post('/v1/chat/completions', CHAT, AUTH); + assert.ok(r.status >= 500, '上游 500 应映射成 5xx,实际 ' + r.status); + } finally { await s.close(); } +}); + +test('Anthropic 端点:上游错误以 Anthropic 错误体返回', async () => { + const s = await setup({ status: 429, errorBody: JSON.stringify({ error: { message: 'slow down' } }) }); + try { + const r = await s.proxy.post('/v1/messages', + { model: 'm', max_tokens: 10, stream: true, messages: [{ role: 'user', content: 'hi' }] }, + { 'x-api-key': 'user_test' }); + const j = await r.json(); + assert.ok(j.error || j.type, 'Anthropic 错误体应有 error 或 type 字段'); + } finally { await s.close(); } +}); + +test('非法 JSON 请求体返回 400', async () => { + const s = await setup(); + try { + const r = await s.proxy.post('/v1/chat/completions', '{not json', AUTH); + assert.equal(r.status, 400); + } finally { await s.close(); } +}); + +test('超过 CC_MAX_BODY_MB 的请求返回 413,且之后小请求仍可用', async () => { + const s = await setup({ env: { CC_MAX_BODY_MB: '1' } }); + try { + const big = JSON.stringify({ model: 'm', stream: true, + messages: [{ role: 'user', content: 'x'.repeat(2 * 1024 * 1024) }] }); + const r1 = await s.proxy.post('/v1/chat/completions', big, AUTH); + assert.equal(r1.status, 413, '超限请求应返回 413(而非直接 reset)'); + + // issue #7:超限后连接必须保持可排空,后续请求不受影响 + const r2 = await s.proxy.post('/v1/chat/completions', + { model: 'm', stream: true, messages: [{ role: 'user', content: 'small' }] }, AUTH); + assert.equal(r2.status, 200, '超限拒绝后小请求仍应正常'); + await r2.text(); + } finally { await s.close(); } +}); + +test('默认上限放行多模态量级的请求(issue #7 场景,~9MB)', async () => { + const s = await setup(); + try { + const body = JSON.stringify({ model: 'm', stream: true, + messages: [{ role: 'user', content: 'x'.repeat(9 * 1024 * 1024) }] }); + const r = await s.proxy.post('/v1/chat/completions', body, AUTH); + assert.equal(r.status, 200, '默认 100MB 上限必须容纳 #7 的多模态长会话'); + await r.text(); + } finally { await s.close(); } +}); + +test('CC_MAX_INFLIGHT 限流:超限返回 503 + Retry-After,且名额会释放', async () => { + // 让上游把请求挂住,以便观察在途名额 + const s = await setup({ env: { CC_MAX_INFLIGHT: '1' }, + onRequest: async (req, res) => { if (req.url === '/alpha/generate') await new Promise(r => setTimeout(r, 1200)); } }); + try { + const first = s.proxy.post('/v1/chat/completions', CHAT, AUTH); + await new Promise(r => setTimeout(r, 400)); // 确保第一个已占住名额 + + const r2 = await s.proxy.post('/v1/chat/completions', CHAT, AUTH); + assert.equal(r2.status, 503, '超出在途上限应返回 503'); + assert.equal(r2.headers.get('retry-after'), '5', '应带 Retry-After'); + const j = await r2.json(); + assert.equal(j.error.type, 'server_busy'); + + const r1 = await first; + assert.equal(r1.status, 200, '已占住名额的请求应正常完成'); + await r1.text(); + + // 名额释放后新请求应通行 + const r3 = await s.proxy.post('/v1/chat/completions', CHAT, AUTH); + assert.equal(r3.status, 200, '名额释放后应恢复通行'); + await r3.text(); + } finally { await s.close(); } +}); + +test('CC_MAX_INFLIGHT 不限制探活端点', async () => { + const s = await setup({ env: { CC_MAX_INFLIGHT: '1' }, + onRequest: async (req, res) => { if (req.url === '/alpha/generate') await new Promise(r => setTimeout(r, 1200)); } }); + try { + const first = s.proxy.post('/v1/chat/completions', CHAT, AUTH); + await new Promise(r => setTimeout(r, 400)); + // 探活端点被占满时仍必须可用,否则编排系统会误判容器已死 + for (const path of ['/health', '/']) { + const r = await s.proxy.get(path); + assert.equal(r.status, 200, path + ' 不应受在途上限影响'); + } + const r1 = await first; await r1.text(); + } finally { await s.close(); } +}); diff --git a/test/fingerprint.test.mjs b/test/fingerprint.test.mjs new file mode 100644 index 0000000..b8309ca --- /dev/null +++ b/test/fingerprint.test.mjs @@ -0,0 +1,102 @@ +// 指纹 / lifecycle 预请求协议契约。 +// 这三条预请求是 CC 侧设备识别的入口,字段形状与次序都属于「行为可观察」的部分。 +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import { setup } from './helpers.mjs'; + +const AUTH = { Authorization: 'Bearer user_fp' }; +const CHAT = { model: 'm', stream: true, messages: [{ role: 'user', content: 'hi' }] }; + +// CLI 侧 FINGERPRINT_IB 的 components 字段全集(对齐 command-code CLI 实现) +const CLI_FIELDS = ['arch', 'collectorVersion', 'cpuCount', 'cpuModel', 'gitEmailHash', + 'hostnameHash', 'isContainer', 'macHashes', 'machineIdHash', 'memGiB', 'osRelease', + 'osUserHash', 'platform', 'runtime', 'timezone']; + +async function firstRequest(s) { + const r = await s.proxy.post('/v1/chat/completions', CHAT, AUTH); + await r.text(); + return s.mock.seen; +} + +test('初始化会发出 fingerprint/record 与 lifecycle-events 两条预请求', async () => { + const s = await setup(); + try { + const seen = await firstRequest(s); + const paths = seen.map(x => x.url); + assert.ok(paths.includes('/alpha/fingerprint/record'), '应发出 fingerprint/record'); + assert.ok(paths.includes('/alpha/lifecycle-events'), '应发出 lifecycle-events'); + assert.ok(paths.includes('/alpha/generate'), '应发出 generate'); + } finally { await s.close(); } +}); + +test('lifecycle 事件类型为 cli_session_exists', async () => { + const s = await setup(); + try { + await firstRequest(s); + const lc = s.mock.seen.find(x => x.url === '/alpha/lifecycle-events'); + const body = JSON.parse(lc.raw); + assert.equal(body.eventType, 'cli_session_exists'); + } finally { await s.close(); } +}); + +test('指纹 components 字段集合与 CLI 完全一致(不多不少)', async () => { + const s = await setup(); + try { + await firstRequest(s); + const fp = s.mock.seen.find(x => x.url === '/alpha/fingerprint/record'); + const body = JSON.parse(fp.raw); + const actual = Object.keys(body.components).sort(); + assert.deepEqual(actual, [...CLI_FIELDS].sort(), + '字段集合偏离 CLI 是实现指纹的典型破绽'); + } finally { await s.close(); } +}); + +test('指纹 runtime=cli、collectorVersion=1,thumbmark/machineIdHash 为 64 位 hex', async () => { + const s = await setup(); + try { + await firstRequest(s); + const body = JSON.parse(s.mock.seen.find(x => x.url === '/alpha/fingerprint/record').raw); + assert.equal(body.components.runtime, 'cli', 'runtime 必须自称 cli'); + assert.equal(body.components.collectorVersion, 1); + assert.match(body.thumbmark, /^[0-9a-f]{64}$/, 'thumbmark 应为 sha256 hex'); + assert.match(body.components.machineIdHash, /^[0-9a-f]{64}$/); + assert.match(body.components.hostnameHash, /^[0-9a-f]{64}$/); + assert.match(body.components.osUserHash, /^[0-9a-f]{64}$/); + } finally { await s.close(); } +}); + +test('macHashes 为 2~5 个 hex 串(CLI 的取值区间)', async () => { + const s = await setup(); + try { + await firstRequest(s); + const body = JSON.parse(s.mock.seen.find(x => x.url === '/alpha/fingerprint/record').raw); + const macs = body.components.macHashes; + assert.ok(Array.isArray(macs)); + assert.ok(macs.length >= 2 && macs.length <= 5, 'macHashes 数量应在 2~5,实际 ' + macs.length); + for (const m of macs) assert.match(m, /^[0-9a-f]+$/); + } finally { await s.close(); } +}); + +test('每种指纹只上报一次(进程内去重)', async () => { + const s = await setup(); + try { + await firstRequest(s); + // 第二次请求不应再触发预请求 + const r2 = await s.proxy.post('/v1/chat/completions', CHAT, AUTH); + await r2.text(); + const fpCount = s.mock.seen.filter(x => x.url === '/alpha/fingerprint/record').length; + assert.equal(fpCount, 1, '同一进程内指纹只应上报一次,实际 ' + fpCount); + } finally { await s.close(); } +}); + +test('不同 API key 各自初始化(互不复用指纹状态)', async () => { + const s = await setup(); + try { + const r1 = await s.proxy.post('/v1/chat/completions', CHAT, { Authorization: 'Bearer user_a' }); + await r1.text(); + const r2 = await s.proxy.post('/v1/chat/completions', CHAT, { Authorization: 'Bearer user_b' }); + await r2.text(); + const fpCount = s.mock.seen.filter(x => x.url === '/alpha/fingerprint/record').length; + assert.equal(fpCount, 2, '每个 key 应各自初始化一次,实际 ' + fpCount); + } finally { await s.close(); } +}); diff --git a/test/fork.test.mjs b/test/fork.test.mjs new file mode 100644 index 0000000..f5c2659 --- /dev/null +++ b/test/fork.test.mjs @@ -0,0 +1,223 @@ +// fork 专属行为的回归护栏。这些改动不在上游,若被上游同步覆盖会静默丢失。 +// 每条都对应一个明确的行为契约,不是实现细节。 +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import http from 'node:http'; +import net from 'node:net'; +import { setup, startProxy, startMockUpstream, allocPort } from './helpers.mjs'; + +const AUTH = { Authorization: 'Bearer user_test' }; +const UUID = 'aaaaaaaa-bbbb-4ccc-8ddd-eeeeeeeeeeee'; +const CHAT = { model: 'm', stream: true, messages: [{ role: 'user', content: 'hi' }] }; + +// ── 逆向对齐:上游 User-Agent 与 threadId ────────────────── +// CLI 侧 vy = "cli"(源码常量),threadId 与 x-session-id 同值 +// (toWireThreadId 对合法 UUID 直接透传)。 +test('fork: 上游请求 User-Agent 为 cli(对齐官方 CLI 常量)', async () => { + const s = await setup(); + try { + const r = await s.proxy.post('/v1/chat/completions', CHAT, { ...AUTH, 'x-session-id': UUID }); + await r.text(); + const g = s.mock.lastGenerate(); + assert.equal(g.headers['user-agent'], 'cli', + '官方 CLI 发送 User-Agent: cli;发送 Node 默认 UA 是明显破绽'); + } finally { await s.close(); } +}); + +test('fork: 预请求(fingerprint/lifecycle)同样带 User-Agent: cli', async () => { + const s = await setup(); + try { + const r = await s.proxy.post('/v1/chat/completions', CHAT, { ...AUTH, 'x-session-id': UUID }); + await r.text(); + for (const path of ['/alpha/fingerprint/record', '/alpha/lifecycle-events']) { + const req = s.mock.seen.find(x => x.url === path); + assert.ok(req, path + ' 应被发出'); + assert.equal(req.headers['user-agent'], 'cli', path + ' 的 UA 也必须是 cli'); + } + } finally { await s.close(); } +}); + +test('fork: threadId 与 x-session-id 同值', async () => { + const s = await setup(); + try { + const r = await s.proxy.post('/v1/chat/completions', CHAT, { ...AUTH, 'x-session-id': UUID }); + await r.text(); + const g = s.mock.lastGenerate(); + assert.equal(g.headers['x-session-id'], UUID, 'session 头应透传'); + assert.equal(g.body.threadId, UUID, + 'threadId 必须与 x-session-id 同值(CLI 的 toWireThreadId 行为)'); + } finally { await s.close(); } +}); + +test('fork: sessionId 非 UUID 时省略 threadId(而非填非法值)', async () => { + const s = await setup(); + try { + // CLI 的 toWireThreadId 对非 UUID 返回 undefined —— 代理必须同样省略该字段 + const r = await s.proxy.post('/v1/chat/completions', CHAT, + { ...AUTH, 'x-session-id': 'not-a-uuid' }); + await r.text(); + const g = s.mock.lastGenerate(); + // 非 UUID 的 session 头会被忽略,回落到 per-key 生成的合法 UUID + const tid = g.body.threadId; + assert.ok(tid === undefined || /^[0-9a-f-]{36}$/.test(tid), + 'threadId 要么省略,要么是合法 UUID,不能是任意字符串。实际: ' + JSON.stringify(tid)); + assert.equal(tid, g.headers['x-session-id'], '若存在则必须与 x-session-id 同值'); + } finally { await s.close(); } +}); + +// ── issue #18:上游 HTTP(S) 代理(零依赖 CONNECT 隧道)── +/** 录制型 CONNECT 代理:记录每次 CONNECT 的 target,并做裸字节转发。 */ +async function startRecordingProxy() { + const port = await allocPort(); + const connects = []; + const server = http.createServer((req, res) => { res.writeHead(405); res.end(); }); + server.on('connect', (req, clientSocket, head) => { + connects.push(req.url); + const [h, p] = req.url.split(':'); + const up = net.connect(Number(p || 443), h, () => { + clientSocket.write('HTTP/1.1 200 Connection Established\r\n\r\n'); + if (head?.length) up.write(head); + up.pipe(clientSocket); clientSocket.pipe(up); + }); + up.on('error', () => clientSocket.destroy()); + clientSocket.on('error', () => up.destroy()); + }); + await new Promise(r => server.listen(port, '127.0.0.1', r)); + return { port, connects, close: () => new Promise(r => server.close(r)) }; +} + +test('#18: 配置 CC_UPSTREAM_PROXY 后上游请求经 CONNECT 隧道', async () => { + const rec = await startRecordingProxy(); + const mock = startMockUpstream(); + const proxy = await startProxy({ upstreamPort: mock.port, + env: { CC_UPSTREAM_PROXY: 'http://127.0.0.1:' + rec.port } }); + try { + const r = await proxy.post('/v1/chat/completions', CHAT, AUTH); + await r.text(); + assert.ok(rec.connects.length >= 1, '应建立 CONNECT 隧道,实际 ' + JSON.stringify(rec.connects)); + // generate + fingerprint + lifecycle 三条都要走代理 + assert.equal(mock.generateCount(), 1, 'generate 应经隧道到达上游'); + assert.equal(mock.lastGenerate().headers['user-agent'], 'cli', '经代理时 UA 仍为 cli'); + } finally { + await proxy.kill(); await mock.close(); await rec.close(); + } +}); + +test('#18: 预请求也走代理(避免同一账号从两个 IP 注册)', async () => { + const rec = await startRecordingProxy(); + const mock = startMockUpstream(); + const proxy = await startProxy({ upstreamPort: mock.port, + env: { CC_UPSTREAM_PROXY: 'http://127.0.0.1:' + rec.port } }); + try { + const r = await proxy.post('/v1/chat/completions', CHAT, AUTH); + await r.text(); + // 三条预请求 + generate 共 3 次 CONNECT(generate/fingerprint/lifecycle) + assert.ok(rec.connects.length >= 3, + 'fingerprint 与 lifecycle 也必须走代理,否则账号会从两个 IP 注册。实际 CONNECT 数: ' + rec.connects.length); + } finally { + await proxy.kill(); await mock.close(); await rec.close(); + } +}); + +test('#18: 未配置代理时不建立任何 CONNECT', async () => { + const rec = await startRecordingProxy(); + const mock = startMockUpstream(); + const proxy = await startProxy({ upstreamPort: mock.port }); // 不设 CC_UPSTREAM_PROXY + try { + const r = await proxy.post('/v1/chat/completions', CHAT, AUTH); + await r.text(); + assert.equal(rec.connects.length, 0, '默认应直连,不经任何代理'); + } finally { + await proxy.kill(); await mock.close(); await rec.close(); + } +}); + +test('#18: 探活端点不经过上游代理', async () => { + const rec = await startRecordingProxy(); + const mock = startMockUpstream(); + const proxy = await startProxy({ upstreamPort: mock.port, + env: { CC_UPSTREAM_PROXY: 'http://127.0.0.1:' + rec.port } }); + try { + const before = rec.connects.length; + const r = await proxy.get('/health'); + assert.equal(r.status, 200); + assert.equal(rec.connects.length, before, '/health 是本地端点,不应触发上游 CONNECT'); + } finally { + await proxy.kill(); await mock.close(); await rec.close(); + } +}); + +// ── 设备指纹派生 ───────────────────────────────────────── +test('fork: 同一 API key 在同一盐下得到稳定 thumbmark(重启后不变)', async () => { + const mock = startMockUpstream(); + const env = { CC_FP_SALT: 'ci-salt', CC_FP_MODE: 'derived' }; + const p1 = await startProxy({ upstreamPort: mock.port, env }); + let first; + try { + const r = await p1.post('/v1/chat/completions', CHAT, AUTH); + await r.text(); + first = JSON.parse(mock.seen.find(x => x.url === '/alpha/fingerprint/record').raw); + } finally { await p1.kill(); } + + const p2 = await startProxy({ upstreamPort: mock.port, env }); + try { + const r = await p2.post('/v1/chat/completions', CHAT, AUTH); + await r.text(); + const fps = mock.seen.filter(x => x.url === '/alpha/fingerprint/record'); + const second = JSON.parse(fps[fps.length - 1].raw); + assert.equal(second.thumbmark, first.thumbmark, + 'derived 模式下同一 key 重启后必须得到同一设备(否则每次重启都像换了台机器)'); + } finally { await p2.kill(); await mock.close(); } +}); + +test('fork: 不同 API key 得到不同 thumbmark', async () => { + const mock = startMockUpstream(); + const proxy = await startProxy({ upstreamPort: mock.port, env: { CC_FP_SALT: 'ci-salt' } }); + try { + for (const k of ['user_a', 'user_b', 'user_c']) { + const r = await proxy.post('/v1/chat/completions', CHAT, { Authorization: 'Bearer ' + k }); + await r.text(); + } + const tms = mock.seen.filter(x => x.url === '/alpha/fingerprint/record') + .map(x => JSON.parse(x.raw).thumbmark); + assert.equal(tms.length, 3, '三个 key 应各上报一次'); + assert.equal(new Set(tms).size, 3, '不同 key 必须映射到不同设备指纹(不能退化成哈希桶)'); + } finally { await proxy.kill(); await mock.close(); } +}); + +test('fork: 不同盐得到不同部署指纹', async () => { + const mock = startMockUpstream(); + const grab = async (salt) => { + const p = await startProxy({ upstreamPort: mock.port, env: { CC_FP_SALT: salt } }); + try { + const r = await p.post('/v1/chat/completions', CHAT, AUTH); + await r.text(); + const fps = mock.seen.filter(x => x.url === '/alpha/fingerprint/record'); + return JSON.parse(fps[fps.length - 1].raw).thumbmark; + } finally { await p.kill(); } + }; + try { + const a = await grab('salt-one'); + const b = await grab('salt-two'); + assert.notEqual(a, b, '不同部署(不同盐)不应共享设备指纹'); + } finally { await mock.close(); } +}); + +test('fork: CC_FP_MODE=random 回退到原行为(重启换设备)', async () => { + const mock = startMockUpstream(); + const env = { CC_FP_MODE: 'random' }; + const grab = async () => { + const p = await startProxy({ upstreamPort: mock.port, env }); + try { + const r = await p.post('/v1/chat/completions', CHAT, AUTH); + await r.text(); + const fps = mock.seen.filter(x => x.url === '/alpha/fingerprint/record'); + return JSON.parse(fps[fps.length - 1].raw).thumbmark; + } finally { await p.kill(); } + }; + try { + const a = await grab(); + const b = await grab(); + assert.notEqual(a, b, 'random 模式应保留「每进程随机」的原始行为'); + } finally { await mock.close(); } +}); diff --git a/test/helpers.mjs b/test/helpers.mjs new file mode 100644 index 0000000..4f1ab33 --- /dev/null +++ b/test/helpers.mjs @@ -0,0 +1,125 @@ +// 测试用 mock 上游 + 代理进程管理。 +// 全部走 loopback,不需要真 key、不访问 Command Code 或 npm registry。 +import http from 'node:http'; +import { spawn } from 'node:child_process'; +import { setTimeout as sleep } from 'node:timers/promises'; +import { mkdtempSync, copyFileSync, existsSync, rmSync } from 'node:fs'; +import { tmpdir } from 'node:os'; +import { join, dirname } from 'node:path'; +import { fileURLToPath } from 'node:url'; + +export const REPO = dirname(fileURLToPath(import.meta.url)).replace(/[/\\]test$/, ''); + +// 取一个当前空闲的端口:让内核分配(listen 0)后立刻释放。 +// 不能用 pid 派生区间 —— node --test 各文件并行,pid 取模会在不同 pid 间 +// 映射到同一区间(如 pid 100 与 pid 600 同桶),进而偶发 EADDRINUSE。 +// 内核分配把冲突面缩到「释放到重新占用」之间的极小窗口,调用方另有重试兜底。 +import net from 'node:net'; +export async function allocPort() { + return await new Promise((resolve, reject) => { + const srv = net.createServer(); + srv.once('error', reject); + srv.listen(0, '127.0.0.1', () => { + const { port } = srv.address(); + srv.close(() => resolve(port)); + }); + }); +} + +/** 启动一个 mock 上游。ndjson 为要回给代理的 CC NDJSON 行数组。 */ +export async function startMockUpstream(opts = {}) { + const port = await allocPort(); + const seen = []; + const server = http.createServer((req, res) => { + const chunks = []; + req.on('data', c => chunks.push(c)); + req.on('end', async () => { + const raw = Buffer.concat(chunks).toString('utf8'); + seen.push({ url: req.url, method: req.method, headers: req.headers, raw }); + if (opts.onRequest) await opts.onRequest(req, res, seen[seen.length - 1]); + if (res.writableEnded) return; + const status = opts.status ?? 200; + if (status !== 200) { + res.writeHead(status, { 'Content-Type': 'application/json' }); + res.end(opts.errorBody || JSON.stringify({ error: { message: 'mock error' } })); + return; + } + res.writeHead(200, { 'Content-Type': 'text/event-stream' }); + for (const line of opts.ndjson ?? [ + '{"type":"text-start"}', + '{"type":"text-delta","text":"hello"}', + '{"type":"text-end"}', + '{"type":"finish-step","finishReason":"stop","usage":{"inputTokens":9,"outputTokens":3}}', + '{"type":"finish","finishReason":"stop","totalUsage":{"inputTokens":9,"outputTokens":3,"cachedInputTokens":0}}', + ]) res.write(line + '\n'); + res.end(); + }); + }); + await new Promise(r => server.listen(port, '127.0.0.1', r)); + return { port, seen, close: () => new Promise(r => server.close(r)), + // 最后一次 /alpha/generate 的请求体(wire 层断言的主要入口) + lastGenerate: () => { + const g = seen.filter(s => s.url === '/alpha/generate').pop(); + return g ? { raw: g.raw, body: JSON.parse(g.raw), headers: g.headers } : null; + }, + generateCount: () => seen.filter(s => s.url === '/alpha/generate').length }; +} + +/** 在临时 cwd 中启动代理(复刻真实部署:proxy.mjs 与 config.json 同目录)。 */ +export async function startProxy({ upstreamPort, env = {}, cwd } = {}) { + const port = await allocPort(); + const logs = []; + // 自建的临时工作目录用完必须删;调用方传了 cwd 则由调用方负责。 + const ownWorkdir = cwd === undefined; + const workdir = cwd ?? mkdtempSync(join(tmpdir(), 'ccp-test-')); + copyFileSync(join(REPO, 'proxy.mjs'), join(workdir, 'proxy.mjs')); + if (!existsSync(join(workdir, 'config.json'))) { + copyFileSync(join(REPO, 'config.json'), join(workdir, 'config.json')); + } + const child = spawn(process.execPath, ['proxy.mjs'], { + cwd: workdir, + env: { ...process.env, PORT: String(port), HOST: '127.0.0.1', + CC_API_BASE: 'http://127.0.0.1:' + upstreamPort, + CC_USE_PROVIDER_MODELS: 'false', // 不访问 /provider/v1/models + ...env }, + stdio: ['ignore', 'pipe', 'pipe'], + }); + child.stdout.on('data', d => logs.push(d.toString())); + child.stderr.on('data', d => logs.push(d.toString())); + + const base = 'http://127.0.0.1:' + port; + let up = false; + for (let i = 0; i < 80; i++) { + if (child.exitCode !== null) break; + try { const r = await fetch(base + '/health'); if (r.ok) { up = true; break; } } catch {} + await sleep(125); + } + if (!up) { child.kill(); throw new Error('proxy did not start:\n' + logs.join('')); } + + return { + port, base, child, logs: () => logs.join(''), + get: (path, init) => fetch(base + path, init), + post: (path, body, headers = {}) => fetch(base + path, { + method: 'POST', headers: { 'Content-Type': 'application/json', ...headers }, + body: typeof body === 'string' ? body : JSON.stringify(body), + }), + kill: () => new Promise(r => { + child.once('exit', () => { + if (ownWorkdir) { try { rmSync(workdir, { recursive: true, force: true }); } catch {} } + r(); + }); + child.kill(); + setTimeout(() => { + if (ownWorkdir) { try { rmSync(workdir, { recursive: true, force: true }); } catch {} } + r(); + }, 2000); + }), + }; +} + +/** 一次性搭好 mock 上游 + 代理。 */ +export async function setup(opts = {}) { + const mock = await startMockUpstream(opts); + const proxy = await startProxy({ upstreamPort: mock.port, env: opts.env, cwd: opts.cwd }); + return { mock, proxy, async close() { await proxy.kill(); await mock.close(); } }; +} diff --git a/test/regressions.test.mjs b/test/regressions.test.mjs new file mode 100644 index 0000000..6cdfc13 --- /dev/null +++ b/test/regressions.test.mjs @@ -0,0 +1,69 @@ +// 回归护栏:为已修复的 issue 各留一条断言,防止再次退化。 +// 每条都注明来源 issue —— 删掉某条前请先读对应的 issue。 +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import { setup } from './helpers.mjs'; + +const AUTH = { Authorization: 'Bearer user_test' }; + +async function wire(s, path, body, headers = AUTH) { + const r = await s.proxy.post(path, body, headers); + await r.text(); + const g = s.mock.lastGenerate(); + return { status: r.status, params: g ? g.body.params : null }; +} + +// issue #17:无 system prompt 时发空格占位,阻止 CC 上游注入 ~7.5K token 默认提示词 +test('#17 chat:无 system prompt 时 params.system 为占位串而非空/缺省', async () => { + const s = await setup(); + try { + const { params } = await wire(s, '/v1/chat/completions', + { model: 'm', stream: true, messages: [{ role: 'user', content: 'hi' }] }); + assert.equal(typeof params.system, 'string', 'params.system 必须是字符串'); + assert.notEqual(params.system, '', '空 system 会触发上游注入默认提示词(#17)'); + } finally { await s.close(); } +}); + +// issue #7:超限必须走 413 + 排空,而不是直接 reset(客户端会看到 Connection error) +test('#7 超限请求返回 413 且连接可继续使用(不 reset)', async () => { + const s = await setup({ env: { CC_MAX_BODY_MB: '1' } }); + try { + const big = JSON.stringify({ model: 'm', stream: true, + messages: [{ role: 'user', content: 'x'.repeat(2 * 1024 * 1024) }] }); + const r = await s.proxy.post('/v1/chat/completions', big, AUTH); + assert.equal(r.status, 413, '必须是 HTTP 413 响应,而不是连接被 reset'); + const j = await r.json(); + assert.ok(j.error, '应为结构化错误体,便于客户端识别'); + } finally { await s.close(); } +}); + +// issue #25:Anthropic input_tokens 只计非缓存部分(与 cache_read 相加 = 总输入) +test('#25 messages:Anthropic usage 的 input_tokens 不含缓存部分', async () => { + const s = await setup({ ndjson: [ + '{"type":"text-start"}', + '{"type":"text-delta","text":"hi"}', + '{"type":"text-end"}', + // inputTokens 是总数(含缓存),cacheRead 是其子集 + '{"type":"finish-step","finishReason":"stop","usage":{"inputTokens":1000,"outputTokens":10,"cachedInputTokens":800,"inputTokenDetails":{"noCacheTokens":200,"cacheReadTokens":800,"cacheWriteTokens":0}}}', + ] }); + try { + const r = await s.proxy.post('/v1/messages', + { model: 'm', max_tokens: 100, stream: true, messages: [{ role: 'user', content: 'hi' }] }, + { 'x-api-key': 'user_test' }); + const text = await r.text(); + // SSE 事件格式:event: \ndata: \n\n —— 按行取,不能用非贪婪 \{.*?\}(会在首个 } 截断) + const deltas = text.split('\n\n') + .map(block => { + const ev = /^event: (\S+)/m.exec(block)?.[1]; + const data = /^data: (.*)$/m.exec(block)?.[1]; + if (ev !== 'message_delta' || !data) return null; + try { return JSON.parse(data); } catch { return null; } + }) + .filter(Boolean); + const usage = deltas.find(d => d.usage)?.usage; + assert.ok(usage, 'message_delta 应携带 usage'); + assert.equal(usage.input_tokens, 200, + 'input_tokens 只能是非缓存部分(200),不能是总数(1000)—— 否则下游相加会约两倍'); + assert.equal(usage.cache_read_input_tokens, 800); + } finally { await s.close(); } +}); diff --git a/test/wire.test.mjs b/test/wire.test.mjs new file mode 100644 index 0000000..74d102a --- /dev/null +++ b/test/wire.test.mjs @@ -0,0 +1,127 @@ +// wire 协议契约:断言发往 CC 上游 /alpha/generate 的请求体形状。 +// 这类断言是挡住「静默丢消息 / 静默改语义」回归的关键 —— 只看 HTTP 状态码看不出来。 +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import { setup } from './helpers.mjs'; + +const AUTH = { Authorization: 'Bearer user_test' }; + +/** 取本次请求发往上游的 params.messages */ +async function wireMessages(proxy, mock, path, body, headers = AUTH) { + const r = await proxy.post(path, body, headers); + await r.text(); + const g = mock.lastGenerate(); + assert.ok(g, '应至少产生一条 /alpha/generate(实际: ' + mock.seen.map(s => s.url).join(',') + ')'); + return { status: r.status, params: g.body.params, config: g.body.config, headers: g.headers }; +} + +test('chat:system 提升到 params.system,且必须是字符串', async () => { + const s = await setup(); + try { + const { params } = await wireMessages(s.proxy, s.mock, '/v1/chat/completions', { + model: 'deepseek/deepseek-v4-flash', stream: true, + messages: [{ role: 'system', content: 'you are terse' }, { role: 'user', content: 'hi' }], + }); + assert.equal(typeof params.system, 'string', 'CC 上游要求 params.system 恒为字符串,传数组会被拒绝'); + assert.equal(params.system, 'you are terse'); + assert.equal(params.messages.length, 1, 'system 不应留在 messages 里'); + assert.equal(params.messages[0].role, 'user'); + } finally { await s.close(); } +}); + +test('chat:user 内容包成 [{type:text}] 结构', async () => { + const s = await setup(); + try { + const { params } = await wireMessages(s.proxy, s.mock, '/v1/chat/completions', { + model: 'm', stream: true, messages: [{ role: 'user', content: 'hello' }], + }); + assert.deepEqual(params.messages[0], { role: 'user', content: [{ type: 'text', text: 'hello' }] }); + } finally { await s.close(); } +}); + +test('chat:assistant 历史按 [reasoning, text, tool-call] 次序回传', async () => { + const s = await setup(); + try { + const { params } = await wireMessages(s.proxy, s.mock, '/v1/chat/completions', { + model: 'm', stream: true, messages: [ + { role: 'user', content: 'q' }, + { role: 'assistant', reasoning_content: 'thinking', content: 'answer', + tool_calls: [{ id: 'c1', type: 'function', function: { name: 'f', arguments: '{"a":1}' } }] }, + { role: 'tool', tool_call_id: 'c1', content: 'result' }, + ], + }); + const asst = params.messages.find(m => m.role === 'assistant'); + assert.deepEqual(asst.content.map(p => p.type), ['reasoning', 'text', 'tool-call'], + 'CC 校验历史中的 reasoning,且次序必须与 CLI 一致'); + assert.equal(asst.content[0].text, 'thinking'); + } finally { await s.close(); } +}); + +test('chat:多模态 image_url 转成 CC image 结构', async () => { + const s = await setup(); + try { + const { params } = await wireMessages(s.proxy, s.mock, '/v1/chat/completions', { + model: 'm', stream: true, messages: [{ role: 'user', content: [ + { type: 'text', text: 'look' }, + { type: 'image_url', image_url: { url: 'data:image/png;base64,AAA' } }, + ] }], + }); + const parts = params.messages[0].content; + assert.equal(parts.find(p => p.type === 'image').image, 'data:image/png;base64,AAA'); + } finally { await s.close(); } +}); + +test('messages:Anthropic thinking block 回传为 reasoning_content', async () => { + const s = await setup(); + try { + const { params } = await wireMessages(s.proxy, s.mock, '/v1/messages', { + model: 'm', max_tokens: 100, stream: true, messages: [ + { role: 'user', content: 'q' }, + { role: 'assistant', content: [ + { type: 'thinking', thinking: 'pondering' }, + { type: 'text', text: 'answer' }, + ] }, + { role: 'user', content: 'again' }, + ], + }, { 'x-api-key': 'user_test' }); + const asst = params.messages.find(m => m.role === 'assistant'); + assert.deepEqual(asst.content.map(p => p.type), ['reasoning', 'text'], + 'Anthropic 的 thinking 必须转成 reasoning 回传,否则 CC 拒绝多轮'); + assert.equal(asst.content[0].text, 'pondering'); + } finally { await s.close(); } +}); + +test('responses:带 type 的 input item 正常转换', async () => { + const s = await setup(); + try { + const { params } = await wireMessages(s.proxy, s.mock, '/v1/responses', { + model: 'm', stream: true, + input: [{ type: 'message', role: 'user', content: 'hello' }], + }); + assert.deepEqual(params.messages[0], { role: 'user', content: [{ type: 'text', text: 'hello' }] }); + } finally { await s.close(); } +}); + +test('responses:input 为字符串时等价于单条 user 消息', async () => { + const s = await setup(); + try { + const { params } = await wireMessages(s.proxy, s.mock, '/v1/responses', { + model: 'm', stream: true, input: 'hello', + }); + assert.deepEqual(params.messages[0], { role: 'user', content: [{ type: 'text', text: 'hello' }] }); + } finally { await s.close(); } +}); + +test('responses:function_call_output 映射成 tool 消息', async () => { + const s = await setup(); + try { + const { params } = await wireMessages(s.proxy, s.mock, '/v1/responses', { + model: 'm', stream: true, input: [ + { type: 'message', role: 'user', content: 'q' }, + { type: 'function_call', call_id: 'c1', name: 'f', arguments: '{}' }, + { type: 'function_call_output', call_id: 'c1', output: 'out' }, + ], + }); + assert.ok(params.messages.some(m => m.role === 'tool'), 'function_call_output 应产出 tool 消息'); + } finally { await s.close(); } +}); From 9ace4226a2acc55e6637c1f89c9e4f392896e93b Mon Sep 17 00:00:00 2001 From: xelr233 Date: Sun, 13 Sep 2026 04:50:33 +0800 Subject: [PATCH 07/21] =?UTF-8?q?test:=20=E4=BF=AE=E5=A4=8D=20fork=20?= =?UTF-8?q?=E6=B5=8B=E8=AF=95=E7=9A=84=20await=20=E6=BC=8F=E5=86=99?= =?UTF-8?q?=EF=BC=8C=E5=B9=B6=E8=AE=A9=E6=8C=82=E8=B5=B7=E6=9C=89=E7=95=8C?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit fork.test.mjs 里 8 处 startMockUpstream() 漏了 await(我用 sed 替换动态 import 时引入)。mock.port 因此是 Promise,代理连不上、mock 又永不关闭, 事件循环被挂住 —— CI 三个矩阵 job 全部空转满 10 分钟后被取消。 两处修复: - 补回 await(8 处) - 让"挂起"有界,避免同类问题再次吃掉整个 job: · helpers 新增 closeServer():先 closeAllConnections 再 close, 并叠加 3s 兜底超时。server.close() 只停止接受新连接,遇到 keep-alive / 未关闭的 socket 会永远等下去。 · npm test 加 --test-timeout=30000,单条测试超 30s 即失败。 --- package.json | 2 +- test/fork.test.mjs | 16 ++++++++-------- test/helpers.mjs | 16 +++++++++++++++- 3 files changed, 24 insertions(+), 10 deletions(-) diff --git a/package.json b/package.json index e5bc345..4c8161e 100644 --- a/package.json +++ b/package.json @@ -8,7 +8,7 @@ "scripts": { "start": "node proxy.mjs", "dev": "node --watch proxy.mjs", - "test": "node --test test/*.test.mjs", + "test": "node --test --test-timeout=30000 test/*.test.mjs", "docker:build": "docker build -t commandcode-proxy:latest .", "docker:build:multi": "docker buildx build --platform linux/amd64,linux/arm64 -t commandcode-proxy:latest ." }, diff --git a/test/fork.test.mjs b/test/fork.test.mjs index f5c2659..ef0fe4b 100644 --- a/test/fork.test.mjs +++ b/test/fork.test.mjs @@ -88,7 +88,7 @@ async function startRecordingProxy() { test('#18: 配置 CC_UPSTREAM_PROXY 后上游请求经 CONNECT 隧道', async () => { const rec = await startRecordingProxy(); - const mock = startMockUpstream(); + const mock = await startMockUpstream(); const proxy = await startProxy({ upstreamPort: mock.port, env: { CC_UPSTREAM_PROXY: 'http://127.0.0.1:' + rec.port } }); try { @@ -105,7 +105,7 @@ test('#18: 配置 CC_UPSTREAM_PROXY 后上游请求经 CONNECT 隧道', async () test('#18: 预请求也走代理(避免同一账号从两个 IP 注册)', async () => { const rec = await startRecordingProxy(); - const mock = startMockUpstream(); + const mock = await startMockUpstream(); const proxy = await startProxy({ upstreamPort: mock.port, env: { CC_UPSTREAM_PROXY: 'http://127.0.0.1:' + rec.port } }); try { @@ -121,7 +121,7 @@ test('#18: 预请求也走代理(避免同一账号从两个 IP 注册)', as test('#18: 未配置代理时不建立任何 CONNECT', async () => { const rec = await startRecordingProxy(); - const mock = startMockUpstream(); + const mock = await startMockUpstream(); const proxy = await startProxy({ upstreamPort: mock.port }); // 不设 CC_UPSTREAM_PROXY try { const r = await proxy.post('/v1/chat/completions', CHAT, AUTH); @@ -134,7 +134,7 @@ test('#18: 未配置代理时不建立任何 CONNECT', async () => { test('#18: 探活端点不经过上游代理', async () => { const rec = await startRecordingProxy(); - const mock = startMockUpstream(); + const mock = await startMockUpstream(); const proxy = await startProxy({ upstreamPort: mock.port, env: { CC_UPSTREAM_PROXY: 'http://127.0.0.1:' + rec.port } }); try { @@ -149,7 +149,7 @@ test('#18: 探活端点不经过上游代理', async () => { // ── 设备指纹派生 ───────────────────────────────────────── test('fork: 同一 API key 在同一盐下得到稳定 thumbmark(重启后不变)', async () => { - const mock = startMockUpstream(); + const mock = await startMockUpstream(); const env = { CC_FP_SALT: 'ci-salt', CC_FP_MODE: 'derived' }; const p1 = await startProxy({ upstreamPort: mock.port, env }); let first; @@ -171,7 +171,7 @@ test('fork: 同一 API key 在同一盐下得到稳定 thumbmark(重启后不 }); test('fork: 不同 API key 得到不同 thumbmark', async () => { - const mock = startMockUpstream(); + const mock = await startMockUpstream(); const proxy = await startProxy({ upstreamPort: mock.port, env: { CC_FP_SALT: 'ci-salt' } }); try { for (const k of ['user_a', 'user_b', 'user_c']) { @@ -186,7 +186,7 @@ test('fork: 不同 API key 得到不同 thumbmark', async () => { }); test('fork: 不同盐得到不同部署指纹', async () => { - const mock = startMockUpstream(); + const mock = await startMockUpstream(); const grab = async (salt) => { const p = await startProxy({ upstreamPort: mock.port, env: { CC_FP_SALT: salt } }); try { @@ -204,7 +204,7 @@ test('fork: 不同盐得到不同部署指纹', async () => { }); test('fork: CC_FP_MODE=random 回退到原行为(重启换设备)', async () => { - const mock = startMockUpstream(); + const mock = await startMockUpstream(); const env = { CC_FP_MODE: 'random' }; const grab = async () => { const p = await startProxy({ upstreamPort: mock.port, env }); diff --git a/test/helpers.mjs b/test/helpers.mjs index 4f1ab33..277e225 100644 --- a/test/helpers.mjs +++ b/test/helpers.mjs @@ -26,6 +26,20 @@ export async function allocPort() { }); } +/** + * 关闭一个 http server,且**保证有界**。 + * server.close() 只停止接受新连接,会一直等到既有连接结束 —— 若有 keep-alive + * 或未被对端关闭的 socket,它会永远挂着,把 CI 拖到 job 超时。 + * (实际发生过:Fork 测试漏写一个 await 导致 mock 泄漏,三个矩阵 job 全部 + * 空转 10 分钟后被取消。)故先强制断开所有连接,再 close,并叠加兜底超时。 + */ +export async function closeServer(server, timeoutMs = 3000) { + if (!server || !server.listening) return; + try { server.closeAllConnections?.(); } catch {} + await Promise.race([new Promise(r => server.close(r)), sleep(timeoutMs)]); + try { server.closeAllConnections?.(); } catch {} +} + /** 启动一个 mock 上游。ndjson 为要回给代理的 CC NDJSON 行数组。 */ export async function startMockUpstream(opts = {}) { const port = await allocPort(); @@ -56,7 +70,7 @@ export async function startMockUpstream(opts = {}) { }); }); await new Promise(r => server.listen(port, '127.0.0.1', r)); - return { port, seen, close: () => new Promise(r => server.close(r)), + return { port, seen, close: () => closeServer(server), // 最后一次 /alpha/generate 的请求体(wire 层断言的主要入口) lastGenerate: () => { const g = seen.filter(s => s.url === '/alpha/generate').pop(); From 18109b18ca5e3975c10cd37acc5c1cf8b132f7f0 Mon Sep 17 00:00:00 2001 From: xelr233 Date: Sun, 13 Sep 2026 04:53:18 +0800 Subject: [PATCH 08/21] =?UTF-8?q?fix:=20=E9=A2=84=E8=AF=B7=E6=B1=82?= =?UTF-8?q?=E4=B9=9F=E5=8F=91=E9=80=81=20User-Agent:=20cli=EF=BC=88?= =?UTF-8?q?=E6=AD=A4=E5=89=8D=E5=8F=AA=E6=9C=89=20generate=20=E8=AE=BE?= =?UTF-8?q?=E4=BA=86=EF=BC=89?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit CI 抓到的真实 fidelity 缺口:CLI 侧指纹预请求与生成请求共用同一个 header 常量表(lb = vy = "cli"),两者 UA 一致。而代理只在 forwardToCC 设了 User-Agent: cli,ensureInitialized 里的 fingerprint/lifecycle 两条预请求 走的是 Node fetch 默认的 "node"。 后果是同一账号的「设备指纹注册」与「生成请求」来自两种 User-Agent —— 这是服务端可直接观测的破绽,且恰好出现在设备识别入口上。 同时修正 fork.test.mjs 里一条断言错误的契约: getSessionId 接受任意 >=8 字符的 session 头(不限 UUID),此时 threadId 必须等于实际发出的 x-session-id。原断言假设非 UUID 会被丢弃,与实现不符。 改为断言真正的不变量(threadId === x-session-id),并补一条无 session 头 时的回落路径。 --- proxy.mjs | 5 +++++ test/fork.test.mjs | 27 +++++++++++++++++++-------- 2 files changed, 24 insertions(+), 8 deletions(-) diff --git a/proxy.mjs b/proxy.mjs index 40a4792..6747038 100644 --- a/proxy.mjs +++ b/proxy.mjs @@ -390,6 +390,11 @@ async function ensureInitialized(apiKey, signal) { // 并行发两个预请求 const headers = { 'Content-Type': 'application/json', + // 官方 CLI 的指纹预请求与生成请求共用同一个 header 常量表(lb = vy = "cli"), + // 因此预请求同样是 User-Agent: cli。此前只有 forwardToCC 设了它, + // 预请求走 Node 默认的 "node" —— 同一账号的指纹注册与生成请求来自两种 UA, + // 是可直接观测的破绽。 + 'User-Agent': 'cli', 'x-cli-environment': 'production', 'Authorization': `Bearer ${apiKey}`, 'x-command-code-version': CC_VERSION, diff --git a/test/fork.test.mjs b/test/fork.test.mjs index ef0fe4b..dadf157 100644 --- a/test/fork.test.mjs +++ b/test/fork.test.mjs @@ -49,19 +49,30 @@ test('fork: threadId 与 x-session-id 同值', async () => { } finally { await s.close(); } }); -test('fork: sessionId 非 UUID 时省略 threadId(而非填非法值)', async () => { +test('fork: threadId 恒等于 x-session-id(含非 UUID 回落路径)', async () => { const s = await setup(); try { - // CLI 的 toWireThreadId 对非 UUID 返回 undefined —— 代理必须同样省略该字段 + // 任意 >=8 字符的 session 头都会被 getSessionId 采纳(不限于 UUID), + // 此时 threadId 必须与之一致 —— 这正是「两者同值」的不变量。 const r = await s.proxy.post('/v1/chat/completions', CHAT, - { ...AUTH, 'x-session-id': 'not-a-uuid' }); + { ...AUTH, 'x-session-id': 'not-a-uuid-but-long-enough' }); await r.text(); const g = s.mock.lastGenerate(); - // 非 UUID 的 session 头会被忽略,回落到 per-key 生成的合法 UUID - const tid = g.body.threadId; - assert.ok(tid === undefined || /^[0-9a-f-]{36}$/.test(tid), - 'threadId 要么省略,要么是合法 UUID,不能是任意字符串。实际: ' + JSON.stringify(tid)); - assert.equal(tid, g.headers['x-session-id'], '若存在则必须与 x-session-id 同值'); + assert.equal(g.headers['x-session-id'], 'not-a-uuid-but-long-enough'); + assert.equal(g.body.threadId, g.headers['x-session-id'], + 'threadId 必须等于实际发出的 x-session-id'); + } finally { await s.close(); } +}); + +test('fork: 无 session 头时回落 per-key session,threadId 仍与之同值', async () => { + const s = await setup(); + try { + const r = await s.proxy.post('/v1/chat/completions', CHAT, AUTH); // 不带 session 头 + await r.text(); + const g = s.mock.lastGenerate(); + const sid = g.headers['x-session-id']; + assert.ok(sid, '应回落到 per-key session'); + assert.equal(g.body.threadId, sid, '回落路径下两者同样必须同值'); } finally { await s.close(); } }); From e89f7d6b896a0572c2d600a7a540674908da324c Mon Sep 17 00:00:00 2001 From: xelr233 Date: Sun, 13 Sep 2026 04:56:37 +0800 Subject: [PATCH 09/21] =?UTF-8?q?test:=20=E4=BF=AE=E6=AD=A3=20threadId=20?= =?UTF-8?q?=E7=9A=84=E4=B8=A4=E6=9D=A1=E5=A5=91=E7=BA=A6=E6=96=AD=E8=A8=80?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit CI 暴露我把两个不同的契约混为一谈: x-session-id header —— getSessionId 接受任意 >=8 字符,原样透传 body.threadId —— 仅当 sessionId 是合法 UUID 时才写(isWireUuid), 否则省略该字段,对齐 CLI 的 toWireThreadId 上一版断言「非 UUID 时 threadId 仍等于 header」,与实现不符。现在按实现 的真实契约分别断言,并补一条无 session 头时的回落路径(回落值是合法的 per-key UUID,此时 threadId 必须与之同值)。 --- test/fork.test.mjs | 17 ++++++++++------- 1 file changed, 10 insertions(+), 7 deletions(-) diff --git a/test/fork.test.mjs b/test/fork.test.mjs index dadf157..c012980 100644 --- a/test/fork.test.mjs +++ b/test/fork.test.mjs @@ -45,22 +45,25 @@ test('fork: threadId 与 x-session-id 同值', async () => { const g = s.mock.lastGenerate(); assert.equal(g.headers['x-session-id'], UUID, 'session 头应透传'); assert.equal(g.body.threadId, UUID, - 'threadId 必须与 x-session-id 同值(CLI 的 toWireThreadId 行为)'); + 'sessionId 为合法 UUID 时,threadId 必须与之同值(CLI 的 toWireThreadId 行为)'); } finally { await s.close(); } }); -test('fork: threadId 恒等于 x-session-id(含非 UUID 回落路径)', async () => { +test('fork: session 头非 UUID 时省略 threadId,但 header 原样透传', async () => { const s = await setup(); try { - // 任意 >=8 字符的 session 头都会被 getSessionId 采纳(不限于 UUID), - // 此时 threadId 必须与之一致 —— 这正是「两者同值」的不变量。 + // 两个契约不同: + // x-session-id header —— getSessionId 接受任意 >=8 字符,原样发出 + // body.threadId —— 仅当 sessionId 是合法 UUID 时才写(isWireUuid), + // 否则省略该字段,对齐 CLI 的 toWireThreadId const r = await s.proxy.post('/v1/chat/completions', CHAT, { ...AUTH, 'x-session-id': 'not-a-uuid-but-long-enough' }); await r.text(); const g = s.mock.lastGenerate(); - assert.equal(g.headers['x-session-id'], 'not-a-uuid-but-long-enough'); - assert.equal(g.body.threadId, g.headers['x-session-id'], - 'threadId 必须等于实际发出的 x-session-id'); + assert.equal(g.headers['x-session-id'], 'not-a-uuid-but-long-enough', + 'session 头应原样透传(不限于 UUID)'); + assert.equal(g.body.threadId, undefined, + '非 UUID 时 threadId 必须省略 —— 填非法值会让上游的 UUID 校验失败'); } finally { await s.close(); } }); From e711988b4e8b697e738f17d3c2f3719c24588322 Mon Sep 17 00:00:00 2001 From: xelr233 Date: Sun, 13 Sep 2026 04:59:40 +0800 Subject: [PATCH 10/21] =?UTF-8?q?test:=20=E6=92=A4=E4=B8=8B=20--test-timeo?= =?UTF-8?q?ut=EF=BC=88Node=2018=20=E4=B8=8D=E6=94=AF=E6=8C=81=EF=BC=89?= =?UTF-8?q?=EF=BC=8C=E6=94=B9=E7=94=A8=20unref=20=E7=9C=8B=E9=97=A8?= =?UTF-8?q?=E7=8B=97?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit CI 的 Node 18 job 报 "node: bad option: --test-timeout=30000" —— 该选项 20.11 才加入,而 engines 声明 >=18。等于我用一个防挂起的措施 把最低支持版本打挂了。这正是矩阵存在的意义:本地只有 Node 24, 不加矩阵就会直接发出去。 改为在 helpers 里起一个 unref 的定时器:进程若因泄漏的 socket / 未 await 的句柄无法退出,到点强制 exit(1) 并打印排查提示。 正常结束时定时器被 unref,不阻止退出。Node 18 同样可用。 保留上一版的 closeServer()(先 closeAllConnections 再 close + 兜底超时)—— 那才是真正修掉「不退出」的手段,看门狗只是兜底。 --- package.json | 2 +- test/helpers.mjs | 13 +++++++++++++ 2 files changed, 14 insertions(+), 1 deletion(-) diff --git a/package.json b/package.json index 4c8161e..e5bc345 100644 --- a/package.json +++ b/package.json @@ -8,7 +8,7 @@ "scripts": { "start": "node proxy.mjs", "dev": "node --watch proxy.mjs", - "test": "node --test --test-timeout=30000 test/*.test.mjs", + "test": "node --test test/*.test.mjs", "docker:build": "docker build -t commandcode-proxy:latest .", "docker:build:multi": "docker buildx build --platform linux/amd64,linux/arm64 -t commandcode-proxy:latest ." }, diff --git a/test/helpers.mjs b/test/helpers.mjs index 277e225..3342bb8 100644 --- a/test/helpers.mjs +++ b/test/helpers.mjs @@ -10,6 +10,19 @@ import { fileURLToPath } from 'node:url'; export const REPO = dirname(fileURLToPath(import.meta.url)).replace(/[/\\]test$/, ''); +// ── 挂起保护 ────────────────────────────────────────────── +// 不能用 --test-timeout:Node 18 没有该选项(20.11 才加入),加了会让 +// engines 下限直接跑不起来。改用启动一个 unref 的定时器,进程若因泄漏 +// 的 socket / 未 await 的句柄而无法退出,到点强制退出并说明原因。 +// 正常结束时定时器被 unref,不阻止退出。 +const HANG_GUARD_MS = Number(process.env.CC_TEST_HANG_GUARD_MS ?? 120000); +const hangGuard = setTimeout(() => { + console.error('[test] 超时未退出:疑似有 server/socket 未关闭(' + + '检查每个测试是否都在 finally 里 await close())。强制退出。'); + process.exit(1); +}, HANG_GUARD_MS); +hangGuard.unref?.(); + // 取一个当前空闲的端口:让内核分配(listen 0)后立刻释放。 // 不能用 pid 派生区间 —— node --test 各文件并行,pid 取模会在不同 pid 间 // 映射到同一区间(如 pid 100 与 pid 600 同桶),进而偶发 EADDRINUSE。 From 383aed8af525c3331b18755d4d7d2565cb130c62 Mon Sep 17 00:00:00 2001 From: xelr233 Date: Tue, 15 Sep 2026 01:26:33 +0800 Subject: [PATCH 11/21] =?UTF-8?q?fix:=20=E5=B7=A5=E5=85=B7=E5=90=8D?= =?UTF-8?q?=E9=87=8D=E5=86=99=E5=AF=B9=E9=BD=90=20CLI=20=E2=80=94=E2=80=94?= =?UTF-8?q?=20=E5=8F=AA=E5=9C=A8=20messages=20=E9=87=8C=E6=94=B9=EF=BC=8C?= =?UTF-8?q?=E4=B8=94=E5=8F=AA=E6=94=B9=E4=B8=80=E9=A1=B9?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 上游 d063b47 引入的 TOOL_NAME_ALIASES 取错了来源,作用方向也反了。 对照 command-code@1.54.0 dist/cli.mjs: nw="search_tools", rw="tool_search" function toWireToolName(e){return e===rw?nw:e} ← 线上只有这一项 function toWireTools(e){return e.map(e=>({name:e.name, ← 声明不做重写 description:e.description,input_schema:e.input_schema}))} // toWireMessages 里:const r=toWireToolName(t.name); n.set(t.id,r); // tool-call 用 r,tool-result 用 n.get(id) —— 两边都是重写后的名字 另外三项(bash_output / task_output / read_multiple_files)来自 ow 表,消费者是 resolveToolNameAlias:模型调了退役工具名时本地按新名字执行,并回一句给模型看的 自然语言 note,还会补 defaults(task_output 补 wait:"exit")。那是执行语义, 不是 wire 变换。 原实现:把四项全用在 params.tools[].name(CLI 不改的地方改了), tool-call / tool-result 反而用原名(CLI 该改的地方没改)。下游按自己声明的 bash_output 找不到工具 —— 即 issue #36 的现象。 修复: - WIRE_TOOL_ALIASES = { tool_search: 'search_tools' },附注释说明 ow 表为何不能照搬 - params.tools[].name 原样下发 - toolNameMap 存重写后的名字(对齐 CLI 的 n.set(t.id, r)) - assistant 的 tool-call 与 tool-result 都用 toWireToolName,保证两边一致 已向上游提 issue #37(含源码证据与建议修法)。 测试:73 全绿,新增 4 条 wire 契约: - tools 声明不做名字重写 - tool_search 在 tool-call 里被重写为 search_tools - tool-result 的 toolName 与 tool-call 一致 - 别名表外的名字在声明与 messages 里都不动 --- proxy.mjs | 38 +++++++++++++++--------- test/wire.test.mjs | 72 ++++++++++++++++++++++++++++++++++++++++++++++ 2 files changed, 97 insertions(+), 13 deletions(-) diff --git a/proxy.mjs b/proxy.mjs index 253f16f..9e27c77 100644 --- a/proxy.mjs +++ b/proxy.mjs @@ -572,12 +572,14 @@ function buildCcRequest(openaiReq) { const chatMessages = messages.filter(m => m.role !== 'system' && m.role !== 'developer'); // Build tool_call_id → tool_name reverse lookup + // 存的是**重写后**的名字 —— 对齐 CLI 的 toWireMessages:const r = toWireToolName(t.name); + // n.set(t.id, r),后面的 tool-result 再从这张表里查同一个名字。 const toolNameMap = {}; for (const msg of chatMessages) { if (msg.role === 'assistant' && msg.tool_calls) { for (const tc of msg.tool_calls) { if (tc.id) { - toolNameMap[tc.id] = tc.function?.name || ''; + toolNameMap[tc.id] = toWireToolName(tc.function?.name || ''); } } } @@ -630,7 +632,7 @@ function buildCcRequest(openaiReq) { parts.push({ type: 'tool-call', toolCallId: tc.id, - toolName: tc.function?.name || '', + toolName: toWireToolName(tc.function?.name || ''), input: (typeof tc.function?.arguments === 'string' ? tryParseJSON(tc.function.arguments) : (tc.function?.arguments || {})), }); } @@ -638,12 +640,14 @@ function buildCcRequest(openaiReq) { return { role: 'assistant', content: parts }; } if (msg.role === 'tool') { + // toolName 必须与上面 tool-call 里的名字一致 —— 两边都走 toWireToolName, + // 否则上游看到的调用名与结果名对不上(CLI 用同一张 map 保证这一点) return { role: 'tool', content: [{ type: 'tool-result', toolCallId: msg.tool_call_id, - toolName: toolNameMap[msg.tool_call_id] || msg.name || '', + toolName: toolNameMap[msg.tool_call_id] || toWireToolName(msg.name || ''), output: { type: 'text', value: toWireToolOutputValue(msg.content) }, }], }; @@ -707,9 +711,10 @@ function buildCcRequest(openaiReq) { body.params.reasoning_effort = reasoning_effort; } // CLI 总是下发 tools(没有工具时是空数组)—— 空数组与缺键在 wire 上可观测,这里对齐 - // CLI 的 toWireTools:只有 name / description / input_schema,没有 type 字段 + // CLI 的 toWireTools:只有 name / description / input_schema,没有 type 字段; + // 且 tools 声明**不做**名字重写(重写只发生在 messages 里,见 WIRE_TOOL_ALIASES 注释) body.params.tools = (tools || []).map(t => ({ - name: toWireToolName(t.function?.name || t.name || ''), + name: t.function?.name || t.name || '', description: t.function?.description || t.description || '', input_schema: t.function?.parameters || t.input_schema || { type: 'object', properties: {} }, })); @@ -732,14 +737,21 @@ function buildCcRequest(openaiReq) { return body; } -// CLI 发送前会重写部分工具名(resolveToolNameAlias / ow 表) -const TOOL_NAME_ALIASES = { - bash_output: 'shell_output', - task_output: 'shell_output', - tool_search: 'search_tools', - read_multiple_files: 'read_file', -}; -function toWireToolName(name) { return TOOL_NAME_ALIASES[name] || name; } +// 线上唯一的工具名重写 —— CLI 的 toWireToolName(e){return e===rw?nw:e}, +// 其中 nw="search_tools"、rw="tool_search"(command-code@1.54.0 dist/cli.mjs)。 +// 只用于**消息里**的 tool-call / tool-result(CLI 的 toWireMessages), +// 不用于 params.tools 声明 —— CLI 的 toWireTools 是原样 map {name,description,input_schema}。 +const WIRE_TOOL_ALIASES = { tool_search: 'search_tools' }; +function toWireToolName(name) { return WIRE_TOOL_ALIASES[name] || name; } + +// 注意:CLI 里还有一张四项表 ow +// {bash_output:{to:'shell_output'}, +// task_output:{to:'shell_output',defaults:{wait:'exit'}}, +// [rw]:{to:nw}, +// read_multiple_files:{to:'read_file'}} +// 那是 resolveToolNameAlias 的**入站**别名:模型调了退役工具名时,本地按新名字执行, +// 并回一句自然语言 note("the tool \`x\` is now \`y\`")让模型下次改口,还会补 defaults。 +// 它是执行语义、不是 wire 变换,照搬到这里会同时改错方向和改错表(见 issue #37)。 // CLI 的 toWireToolOutput:只取文本块,用 '\n' 拼接 function toWireToolOutputValue(content) { diff --git a/test/wire.test.mjs b/test/wire.test.mjs index c7255cd..139254e 100644 --- a/test/wire.test.mjs +++ b/test/wire.test.mjs @@ -165,3 +165,75 @@ test('responses:function_call_output 映射成 tool 消息', async () => { assert.ok(params.messages.some(m => m.role === 'tool'), 'function_call_output 应产出 tool 消息'); } finally { await s.close(); } }); + +// ── 工具名重写(issue #37) ────────────────────────────── +// CLI 的 toWireToolName(e){return e===rw?nw:e},rw="tool_search"、nw="search_tools": +// **只有这一项**,且只作用于 toWireMessages(tool-call 与 tool-result), +// 不作用于 toWireTools(声明原样下发)。ow 表是另一回事(入站执行别名,见 proxy.mjs 注释)。 +test('tools 声明不做名字重写(CLI 的 toWireTools 是原样 map)', async () => { + const s = await setup(); + try { + const { params } = await wireMessages(s.proxy, s.mock, '/v1/chat/completions', { + model: 'm', stream: true, messages: [{ role: 'user', content: 'q' }], + tools: ['bash_output', 'read_multiple_files', 'tool_search'].map(n => ({ + type: 'function', function: { name: n, description: '', parameters: { type: 'object', properties: {} } }, + })), + }); + assert.deepEqual(params.tools.map(t => t.name), ['bash_output', 'read_multiple_files', 'tool_search'], + '客户端声明的名字必须原样下发,否则客户端按自己的声明找不到工具'); + } finally { await s.close(); } +}); + +test('tool_search 在 tool-call 里被重写为 search_tools(CLI 唯一一项线上别名)', async () => { + const s = await setup(); + try { + const { params } = await wireMessages(s.proxy, s.mock, '/v1/chat/completions', { + model: 'm', stream: true, messages: [ + { role: 'user', content: 'q' }, + { role: 'assistant', content: null, + tool_calls: [{ id: 'c1', type: 'function', function: { name: 'tool_search', arguments: '{}' } }] }, + { role: 'tool', tool_call_id: 'c1', content: 'r' }, + ], + }); + const call = params.messages.find(m => m.role === 'assistant').content.find(p => p.type === 'tool-call'); + assert.equal(call.toolName, 'search_tools', 'rw → nw 是 CLI 唯一的线上重写'); + } finally { await s.close(); } +}); + +test('tool-result 的 toolName 与 tool-call 一致(CLI 用同一张 map)', async () => { + const s = await setup(); + try { + const { params } = await wireMessages(s.proxy, s.mock, '/v1/chat/completions', { + model: 'm', stream: true, messages: [ + { role: 'user', content: 'q' }, + { role: 'assistant', content: null, + tool_calls: [{ id: 'c1', type: 'function', function: { name: 'tool_search', arguments: '{}' } }] }, + { role: 'tool', tool_call_id: 'c1', name: 'tool_search', content: 'r' }, + ], + }); + const call = params.messages.find(m => m.role === 'assistant').content.find(p => p.type === 'tool-call'); + const res = params.messages.find(m => m.role === 'tool').content[0]; + assert.equal(res.toolName, call.toolName, + '调用名与结果名对不上会被上游判为无效的工具结果'); + assert.equal(res.toolName, 'search_tools'); + } finally { await s.close(); } +}); + +test('未在别名表里的工具名不被改写(tools 声明与 messages 都不动)', async () => { + const s = await setup(); + try { + const { params } = await wireMessages(s.proxy, s.mock, '/v1/chat/completions', { + model: 'm', stream: true, messages: [ + { role: 'user', content: 'q' }, + { role: 'assistant', content: null, + tool_calls: [{ id: 'c1', type: 'function', function: { name: 'get_weather', arguments: '{}' } }] }, + { role: 'tool', tool_call_id: 'c1', content: 'r' }, + ], + tools: [{ type: 'function', function: { name: 'get_weather', parameters: { type: 'object', properties: {} } } }], + }); + assert.equal(params.tools[0].name, 'get_weather'); + assert.equal(params.messages.find(m => m.role === 'assistant').content.find(p => p.type === 'tool-call').toolName, 'get_weather'); + assert.equal(params.messages.find(m => m.role === 'tool').content[0].toolName, 'get_weather'); + } finally { await s.close(); } +}); + From 3cefa6b563a383cbca5ac837a67922b3ea2c7e34 Mon Sep 17 00:00:00 2001 From: xelr233 Date: Tue, 15 Sep 2026 01:47:16 +0800 Subject: [PATCH 12/21] =?UTF-8?q?fix:=20=E5=B7=A5=E5=85=B7=E5=90=8D?= =?UTF-8?q?=E4=B8=80=E4=B8=AA=E9=83=BD=E4=B8=8D=E9=87=8D=E5=91=BD=E5=90=8D?= =?UTF-8?q?=20=E2=80=94=E2=80=94=20=E5=88=A0=E9=99=A4=20toWireToolName?= =?UTF-8?q?=EF=BC=88383aed8=20=E7=9A=84=E6=94=B6=E5=B0=BE=EF=BC=89?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 383aed8 只做到一半:#37 里我建议「照搬 CLI,声明不改、消息改」,保留了一项 tool_search→search_tools。继续深挖 CLI 源码后发现那个方案本身也不成立。 CLI 里两个改名函数的真实定位(command-code@1.54.0 dist/cli.mjs): createSearchToolsTool → schema.name = nw = "search_tools" createRetiredToolSearchTool → schema.name = rw = "tool_search", visible:()=>false description: "Retired: use search_tools instead. Calls to the tool_search NAME are routed to search_tools by the runner automatically." createToolRunner: o = () => [...tools, search_tools, retired] catalog 的 eligible() = n.filter(isVisible),发起请求那次 getSchemas({mode}) 不带 includeHidden(只有本地 resolveToolSchema 才开) → params.tools 里永远没有 tool_search toWireToolName(e){return e===rw?nw:e} 只作用在 toWireMessages toWireTools(e){return e.map(e=>({name,description,input_schema}))} 原样 resolveToolNameAlias(ow) 被工具执行器调用,产出给模型看的 Repair note + 补 defaults 即 toWireToolName 不是「工具重命名设施」,而是「把自家 catalog 里那一个退役名字的 历史归一化」—— 前提是 CLI 自己退役过工具名、且可能重放旧会话。 反代没有这个前提:params.tools 由下游给出,没有 catalog、没有退役名。 而且只改消息不改声明本身就是不自洽的:客户端一旦声明了名为 tool_search 的工具, 就是「声明 tool_search、消息 search_tools」,复现 #36 的同一个 bug。 修复:删除 WIRE_TOOL_ALIASES 与 toWireToolName,三处调用点退回原名, 原处留一段墓志铭注释说明两个 CLI 函数的真实定位与「将来要支持旧会话该做成入站」。 测试同步改为断言「不重写」。 结论已发到 issue #37(含 cross-ref #36)。 --- proxy.mjs | 54 ++++++++++++++++++++++++++-------------------- test/wire.test.mjs | 20 ++++++++++------- 2 files changed, 43 insertions(+), 31 deletions(-) diff --git a/proxy.mjs b/proxy.mjs index 9e27c77..e54a81d 100644 --- a/proxy.mjs +++ b/proxy.mjs @@ -572,14 +572,13 @@ function buildCcRequest(openaiReq) { const chatMessages = messages.filter(m => m.role !== 'system' && m.role !== 'developer'); // Build tool_call_id → tool_name reverse lookup - // 存的是**重写后**的名字 —— 对齐 CLI 的 toWireMessages:const r = toWireToolName(t.name); - // n.set(t.id, r),后面的 tool-result 再从这张表里查同一个名字。 + // 名字原样透传(理由见 toWireToolName 删除处的注释)。 const toolNameMap = {}; for (const msg of chatMessages) { if (msg.role === 'assistant' && msg.tool_calls) { for (const tc of msg.tool_calls) { if (tc.id) { - toolNameMap[tc.id] = toWireToolName(tc.function?.name || ''); + toolNameMap[tc.id] = tc.function?.name || ''; } } } @@ -632,7 +631,7 @@ function buildCcRequest(openaiReq) { parts.push({ type: 'tool-call', toolCallId: tc.id, - toolName: toWireToolName(tc.function?.name || ''), + toolName: tc.function?.name || '', input: (typeof tc.function?.arguments === 'string' ? tryParseJSON(tc.function.arguments) : (tc.function?.arguments || {})), }); } @@ -640,14 +639,13 @@ function buildCcRequest(openaiReq) { return { role: 'assistant', content: parts }; } if (msg.role === 'tool') { - // toolName 必须与上面 tool-call 里的名字一致 —— 两边都走 toWireToolName, - // 否则上游看到的调用名与结果名对不上(CLI 用同一张 map 保证这一点) + // toolName 与上面 tool-call 里的一致(同一张 map),不重命名 return { role: 'tool', content: [{ type: 'tool-result', toolCallId: msg.tool_call_id, - toolName: toolNameMap[msg.tool_call_id] || toWireToolName(msg.name || ''), + toolName: toolNameMap[msg.tool_call_id] || msg.name || '', output: { type: 'text', value: toWireToolOutputValue(msg.content) }, }], }; @@ -712,7 +710,7 @@ function buildCcRequest(openaiReq) { } // CLI 总是下发 tools(没有工具时是空数组)—— 空数组与缺键在 wire 上可观测,这里对齐 // CLI 的 toWireTools:只有 name / description / input_schema,没有 type 字段; - // 且 tools 声明**不做**名字重写(重写只发生在 messages 里,见 WIRE_TOOL_ALIASES 注释) + // 且**不做**任何名字重写(理由见下面「工具名重写整段删除」的注释) body.params.tools = (tools || []).map(t => ({ name: t.function?.name || t.name || '', description: t.function?.description || t.description || '', @@ -737,21 +735,31 @@ function buildCcRequest(openaiReq) { return body; } -// 线上唯一的工具名重写 —— CLI 的 toWireToolName(e){return e===rw?nw:e}, -// 其中 nw="search_tools"、rw="tool_search"(command-code@1.54.0 dist/cli.mjs)。 -// 只用于**消息里**的 tool-call / tool-result(CLI 的 toWireMessages), -// 不用于 params.tools 声明 —— CLI 的 toWireTools 是原样 map {name,description,input_schema}。 -const WIRE_TOOL_ALIASES = { tool_search: 'search_tools' }; -function toWireToolName(name) { return WIRE_TOOL_ALIASES[name] || name; } - -// 注意:CLI 里还有一张四项表 ow -// {bash_output:{to:'shell_output'}, -// task_output:{to:'shell_output',defaults:{wait:'exit'}}, -// [rw]:{to:nw}, -// read_multiple_files:{to:'read_file'}} -// 那是 resolveToolNameAlias 的**入站**别名:模型调了退役工具名时,本地按新名字执行, -// 并回一句自然语言 note("the tool \`x\` is now \`y\`")让模型下次改口,还会补 defaults。 -// 它是执行语义、不是 wire 变换,照搬到这里会同时改错方向和改错表(见 issue #37)。 +// ── 工具名为什么一个都不重命名(已删掉的别名表的墓志铭) ────────────── +// 上游 d063b47 曾引入一张 4 项别名表并作用于 params.tools[].name。查 CLI 源码后 +// (command-code@1.54.0 dist/cli.mjs)确认:**wire 协议里没有工具重命名这回事**。 +// +// CLI 里确实存在两个改名的函数,但都不适用于反代: +// +// toWireToolName(e){return e===rw?nw:e} // rw="tool_search" → nw="search_tools" +// —— 只作用在 toWireMessages(tool-call 与 tool-result,两边同名), +// 不作用在 toWireTools(声明原样下发)。它的存在前提是 CLI 自己退役过 +// tool_search 这个名字:createRetiredToolSearchTool 给的 visible:()=>false, +// 该工具从不进 params.tools,只有重放旧会话时历史里才会残留这个旧名。 +// +// resolveToolNameAlias(ow 表) // bash_output/task_output/read_multiple_files +// —— 被工具执行器调用:模型喊了退役名时本地按新名跑,回一句给模型看的 +// "Repair note",并按 defaults 补参(task_output 会补 wait:"exit")。 +// 这是执行语义、不是 wire 变换。 +// +// 反代没有这个前提:params.tools 由下游客户端给出,proxy 没有 catalog、没有退役名, +// 请求里出现的每个名字对 proxy 来说都是当前名。若强行重命名,一旦客户端恰好声明了 +// 一个叫 tool_search 的工具,就会变成「声明 tool_search、消息 search_tools」—— +// 下游按自己声明的名字派发不到工具(issue #36 / #37 的根因)。 +// +// 因此这里全程原样透传。若将来真要支持「重放真实 CLI 旧会话」,正确做法是**入站** +// 归一化 + 显式开关,并连 defaults 一起补,与 resolveToolNameAlias 同语义; +// 绝不要做成「上行改、下行不改」。 // CLI 的 toWireToolOutput:只取文本块,用 '\n' 拼接 function toWireToolOutputValue(content) { diff --git a/test/wire.test.mjs b/test/wire.test.mjs index 139254e..384f07f 100644 --- a/test/wire.test.mjs +++ b/test/wire.test.mjs @@ -166,10 +166,11 @@ test('responses:function_call_output 映射成 tool 消息', async () => { } finally { await s.close(); } }); -// ── 工具名重写(issue #37) ────────────────────────────── -// CLI 的 toWireToolName(e){return e===rw?nw:e},rw="tool_search"、nw="search_tools": -// **只有这一项**,且只作用于 toWireMessages(tool-call 与 tool-result), -// 不作用于 toWireTools(声明原样下发)。ow 表是另一回事(入站执行别名,见 proxy.mjs 注释)。 +// ── 工具名完全不重命名(issue #36 / #37) ────────────── +// wire 协议里没有工具重命名这回事。CLI 的 toWireToolName 只服务于「重放自家退役 +// 工具名的旧会话」(tool_search 在 CLI 里 visible:()=>false,从不进声明);反代没有 +// catalog、没有退役名,因此声明与消息都必须原样透传 —— 只要有一处改名,下游就会按 +// 自己声明的名字派发不到工具。 test('tools 声明不做名字重写(CLI 的 toWireTools 是原样 map)', async () => { const s = await setup(); try { @@ -184,7 +185,7 @@ test('tools 声明不做名字重写(CLI 的 toWireTools 是原样 map)', as } finally { await s.close(); } }); -test('tool_search 在 tool-call 里被重写为 search_tools(CLI 唯一一项线上别名)', async () => { +test('tool_search 在 tool-call 里也**不**被重写(不套用 CLI 自家的退役名归一化)', async () => { const s = await setup(); try { const { params } = await wireMessages(s.proxy, s.mock, '/v1/chat/completions', { @@ -194,9 +195,12 @@ test('tool_search 在 tool-call 里被重写为 search_tools(CLI 唯一一项 tool_calls: [{ id: 'c1', type: 'function', function: { name: 'tool_search', arguments: '{}' } }] }, { role: 'tool', tool_call_id: 'c1', content: 'r' }, ], + tools: [{ type: 'function', function: { name: 'tool_search', parameters: { type: 'object', properties: {} } } }], }); const call = params.messages.find(m => m.role === 'assistant').content.find(p => p.type === 'tool-call'); - assert.equal(call.toolName, 'search_tools', 'rw → nw 是 CLI 唯一的线上重写'); + assert.equal(call.toolName, 'tool_search', + '客户端声明的就是 tool_search,消息里必须还是它;改成 search_tools 下游就派发不到'); + assert.equal(params.tools[0].name, 'tool_search', '声明与消息必须同名'); } finally { await s.close(); } }); @@ -215,11 +219,11 @@ test('tool-result 的 toolName 与 tool-call 一致(CLI 用同一张 map)', const res = params.messages.find(m => m.role === 'tool').content[0]; assert.equal(res.toolName, call.toolName, '调用名与结果名对不上会被上游判为无效的工具结果'); - assert.equal(res.toolName, 'search_tools'); + assert.equal(res.toolName, 'tool_search'); } finally { await s.close(); } }); -test('未在别名表里的工具名不被改写(tools 声明与 messages 都不动)', async () => { +test('普通工具名在 tools 声明与 messages 里都不动', async () => { const s = await setup(); try { const { params } = await wireMessages(s.proxy, s.mock, '/v1/chat/completions', { From 1a670f29aa1fcd364fd48f5cf689fd63a5e1c406 Mon Sep 17 00:00:00 2001 From: xelr233 Date: Tue, 15 Sep 2026 02:16:44 +0800 Subject: [PATCH 13/21] =?UTF-8?q?fix:=20=E4=B8=8A=E6=B8=B8=E6=B2=A1?= =?UTF-8?q?=E6=AD=A3=E5=B8=B8=E8=B5=B0=E5=AE=8C=20finish=20=E6=97=B6?= =?UTF-8?q?=E4=B8=8D=E5=86=8D=E8=B0=8E=E6=8A=A5=E6=88=90=E5=8A=9F=EF=BC=88?= =?UTF-8?q?issue=20#38=EF=BC=89?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 68664c2 修了 stop_reason 的一个成员('tool-calls' 连字符),方向正确, 但同一族里还有四个成员没处理,后果都比它更严重:**上游明明截断了, 下游收到的是「正常结束」**。对照 command-code@1.54.0 dist/cli.mjs 逐条对齐。 四种情形(mapFinishReason 只认 tool-calls/length/stop,其余原样放行): 1) max_output_tokens / model_context_window_exceeded CLI 的 normalizeStopReason2 把这两个都算 max_tokens。原实现走 default: OpenAI 侧透出非法枚举,Anthropic 侧 mapAnthropicStopReason 兜底成 end_turn —— 上下文撑爆被报成正常结束。 2) pause_turn Anthropic 原生枚举,表示「这一轮被暂停,后面还有」。CLI 靠自动续写循环 (Ph=5)把它吸收掉,代理不续写就必须如实上报,不能吞。 Anthropic 侧原样透出 pause_turn;OpenAI 没有对应枚举,折成 length (表达「输出不完整」)而不是折成 stop(那是谎报完成)。 3) network-error / connection-error / upstream-error CLI 的 isNetworkFailureFinish → 502 可重试。 4) 流里根本没有 finish 事件 CLI:"Stream ended unexpectedly before completion (no finish event) — response was truncated" → 502 可重试。 原实现 Anthropic 侧 `stopReason || 'end_turn'` 无条件兜底,OpenAI 侧 连 finish_reason 块都不发直接 [DONE]。 修法: - mapFinishReason 全量归一化(length 家族 / upstream_error),未知值原样返回, 不再静默折成 stop - mapAnthropicStopReason 增加 pause_turn / refusal 原样透出 - 新增 toOpenAIFinishReason:pause_turn → length - 新增 incompleteUpstreamDetail(sawFinish, finishReason) / incompleteUpstreamError(), 三条协议共用一个判定 - 六处补 sawFinish 跟踪(OpenAI 流式/非流式、Anthropic 流式/非流式、Responses 流式/非流式), 没走完 finish 时:非流式报 502 可重试,流式发 error 事件而不是补一个假的结束标志 - Responses 流式原先直接比对原始 finishReason === 'length',改为用归一化后的值 - 流式路径把「没有正常结束」判定排在「零输出」之前 —— 上游压根没发 finish 时, 「no finish event」才是根因,按 429 报会掩盖它 sawFinish 的口径是「上游给过任何完成信号」:finish 与代理一直在处理的 finish-step 都算。(finish-step 不在 CLI 的事件集里,但既然代理认它,就不能让它变成「没完成」, 否则会把原本正常的响应误判成 502。真正要拦的是「一个完成信号都没有就断了」。) 测试:新增 test/stream-end.test.mjs 15 条(三种协议 × 四种情形 + 正常结束不受影响的回归), 全套 73 → 88 全绿。 --- proxy.mjs | 193 ++++++++++++++++++++++++++++++++---- test/stream-end.test.mjs | 204 +++++++++++++++++++++++++++++++++++++++ 2 files changed, 379 insertions(+), 18 deletions(-) create mode 100644 test/stream-end.test.mjs diff --git a/proxy.mjs b/proxy.mjs index e54a81d..0cc09aa 100644 --- a/proxy.mjs +++ b/proxy.mjs @@ -67,7 +67,7 @@ const CFG = loadConfig(); // ── 设备指纹(形态与哈希逐字对齐官方 CLI;1.53.1 对齐,1.54.0 复核未变) ── // CPU 型号与核心数对应表(仅 Windows x64) const FINGERPRINT_CPUS = [ - { model: '12th Gen Intel(R) Core(TM) i7-12650H', cores: 10 }, + { model: '12th Gen Intel(R) Core(TM) i7-12650H', cores: 10 }, // TEMP-REVERT { model: '12th Gen Intel(R) Core(TM) i5-12400F', cores: 6 }, { model: '12th Gen Intel(R) Core(TM) i9-12900K', cores: 16 }, { model: '13th Gen Intel(R) Core(TM) i7-13700K', cores: 16 }, @@ -777,6 +777,8 @@ function tryParseJSON(str) { // ── CC NDJSON → OpenAI SSE 转换 ──────────────────── function createSseTranslator(model, completionId, created) { + // 是否见过终态 finish 事件。CLI 用同一个标志判定「流是不是被截断了」。 + let sawFinish = false; let chunkIndex = 0; let sentRole = false; let finishReason = null; @@ -845,6 +847,7 @@ function createSseTranslator(model, completionId, created) { } case 'finish-step': { + sawFinish = true; if (event.finishReason) finishReason = mapFinishReason(event.finishReason); if (event.usage) { usage = event.usage; @@ -856,7 +859,8 @@ function createSseTranslator(model, completionId, created) { } case 'finish': { - const fr = finishReason || mapFinishReason(event.finishReason || 'stop'); + sawFinish = true; + const fr = toOpenAIFinishReason(finishReason || mapFinishReason(event.finishReason || 'stop')); const u = event.totalUsage || usage || {}; normalizeUsage(u); this.inputTokens = u.inputTokens ?? 0; @@ -893,6 +897,11 @@ function createSseTranslator(model, completionId, created) { return out.length > 0 ? out : null; }, + /** 这次上游流若没有正常走完 finish,返回可读原因;正常则为 null。 */ + incompleteDetail() { + return incompleteUpstreamDetail(sawFinish, finishReason); + }, + /** 获取 SSE 结束标记 */ getDoneEvent() { return 'data: [DONE]\n\n'; @@ -941,13 +950,56 @@ function anthropicInputTokens(usage, noCacheOverride) { return Math.max(0, (u.inputTokens || 0) - cacheRead - cacheWrite); } +// 上游 finishReason → 本代理内部规范化取值。 +// 对齐 CLI 的 normalizeStopReason2 / isNetworkFailureFinish(command-code@1.54.0): +// tool_use | tool-calls | tool_calls → tool_calls +// length | max_tokens | max_output_tokens +// | model_context_window_exceeded → length +// /^(network|connection|upstream)[-_\s]?error$/i → upstream_error +// pause_turn → pause_turn(原样保留) +// 关键点:'length' 家族**不止 'length' 一个值**。max_output_tokens 与 +// model_context_window_exceeded 都是「输出被截断」,折成 stop/end_turn 等于 +// 把半截回答谎报成完整回答。未知值一律原样返回,宁可让它露出来也不要静默折成 stop。 function mapFinishReason(reason) { - switch (reason) { - case 'tool-calls': return 'tool_calls'; - case 'length': return 'length'; - case 'stop': return 'stop'; - default: return reason || 'stop'; - } + const r = String(reason ?? '').trim().toLowerCase(); + if (!r) return 'stop'; + if (r === 'tool-calls' || r === 'tool_calls' || r === 'tool_use') return 'tool_calls'; + if (r === 'length' || r === 'max_tokens' + || r === 'max_output_tokens' || r === 'model_context_window_exceeded') return 'length'; + if (/^(?:network|connection|upstream)[-_\s]?error$/.test(r)) return 'upstream_error'; + return r; +} + +// 上游「没有正常走完」的两种情形,CLI 都当成可重试的 502: +// · 流里根本没有 finish 事件 —— "Stream ended unexpectedly before completion +// (no finish event) — response was truncated" +// · provider 报 network/connection/upstream-error —— isNetworkFailureFinish +// 返回 null 表示这次流是正常结束的。 +// +// sawFinish 的口径是「上游给过任何完成信号」:终态 finish,以及本代理一直在处理的 +// finish-step。('finish-step' 在 CLI 的事件集里不存在 —— 见 proxy.mjs 各处注释 —— +// 但既然代理认它,就不能让它变成「没完成」,否则会把原本正常的响应误判成 502。 +// 真正要拦的是「一个完成信号都没有就断了」。) +function incompleteUpstreamDetail(sawFinish, finishReason) { + if (!sawFinish) return 'no finish event'; + if (finishReason === 'upstream_error') return 'provider reported an upstream connection failure'; + return null; +} + +function incompleteUpstreamError(detail) { + return { + status: 502, + // retry_after 同时放在 body 里与顶层:sendJSON 只发 body, + // 而 sendAnthropicError / sendResponsesError 需要单独的形参。 + body: { + error: { + message: `Upstream stream ended without a completion finish (${detail}) — response was truncated`, + type: 'upstream_error', + }, + retry_after: 10, + }, + retry_after: 10, + }; } // ── 错误映射 ─────────────────────────────────────── @@ -1442,6 +1494,18 @@ async function handleChatCompletions(req, res) { return; } try { res.write(`data: ${JSON.stringify(translator.upstreamError.body)}\n\n`); } catch {} + // 上游没有正常走完 finish(无 finish 事件 / provider 报连接失败): + // 不能补一个 finish_reason 就 [DONE] —— 那等于把截断谎报成完整回答。 + // 对齐 CLI:这一族一律按可重试的 502 处理。 + // 必须排在零输出判定之前 —— 上游压根没发 finish 时,「no finish event」才是根因, + // 零输出只是它的表象(此时按 429 报会掩盖真实原因)。 + } else if (translator.incompleteDetail()) { + const detail = translator.incompleteDetail(); + log('warn', 'Upstream stream incomplete', { path: '/v1/chat/completions', reason: detail }); + const err = incompleteUpstreamError(detail); + try { if (!abortController.signal.aborted) abortController.abort(); } catch {} + if (!started) { sendJSON(res, err.status, err.body); return; } + try { res.write(`data: ${JSON.stringify(err.body)}\n\n`); } catch {} // 输出 token 为 0 时记为错误,避免下游异常计费 } else if (translator.outputTokens === 0) { try { if (!abortController.signal.aborted) abortController.abort(); } catch {} @@ -1515,6 +1579,7 @@ async function handleChatCompletions(req, res) { // ── 非流式响应(缓冲完整 NDJSON)── let reasoningContent = ''; let finishReason = 'stop'; + let sawFinish = false; let usage = null; let toolCalls = null; let upstreamError = null; @@ -1546,8 +1611,10 @@ async function handleChatCompletions(req, res) { }, }); break; + case 'finish-step': case 'finish': lastCcEvent = event.type; + sawFinish = true; finishReason = mapFinishReason(event.finishReason || 'stop'); if (event.totalUsage) usage = event.totalUsage; break; @@ -1586,6 +1653,16 @@ async function handleChatCompletions(req, res) { return; } + // 上游没有正常走完 finish —— 对齐 CLI 按可重试 502 处理,不谎报成功 + const incomplete = incompleteUpstreamDetail(sawFinish, finishReason); + if (incomplete) { + log('warn', 'Upstream stream incomplete', { path: '/v1/chat/completions', reason: incomplete }); + const err = incompleteUpstreamError(incomplete); + try { if (!abortController.signal.aborted) abortController.abort(); } catch {} + sendJSON(res, err.status, err.body); + return; + } + // 输出 token 为 0 时记为错误,避免下游异常计费 if ((usage?.outputTokens ?? 0) === 0) { try { if (!abortController.signal.aborted) abortController.abort(); } catch {} @@ -1606,7 +1683,7 @@ async function handleChatCompletions(req, res) { toolCalls ? { tool_calls: toolCalls } : {}, reasoningContent ? { reasoning_content: reasoningContent } : {}, ), - finish_reason: finishReason, + finish_reason: toOpenAIFinishReason(finishReason), }], usage: (() => { if (!usage) usage = {}; @@ -1662,10 +1739,22 @@ function mapAnthropicStopReason(finishReason) { case 'tool_calls': return 'tool_use'; case 'length': return 'max_tokens'; case 'stop': return 'end_turn'; + // Anthropic 的原生枚举,必须原样透出:它表示「这一轮被暂停,后面还有内容」。 + // 折成 end_turn 会让下游把半截回答当成写完了(CLI 是靠自动续写把它吸收掉的, + // 代理不自动续写,就必须如实上报,不能吞掉)。 + case 'pause_turn': return 'pause_turn'; + case 'refusal': return 'refusal'; default: return 'end_turn'; } } +// OpenAI 的 finish_reason 只有 stop | length | tool_calls | content_filter | function_call。 +// pause_turn 没有对应值:折成 'stop' 是谎报完成(正是要修的问题), +// 折成 'length' 至少如实表达了「输出不完整」,下游的截断处理会做对的事。 +function toOpenAIFinishReason(finishReason) { + return finishReason === 'pause_turn' ? 'length' : finishReason; +} + // Generate a Claude-format fake signature for thinking blocks. // Anthropic validates thinking signatures cryptographically; third-party // proxies cannot mint valid ones. Claude Code's shallow check only requires @@ -1901,6 +1990,11 @@ async function* createAnthropicSseTranslator(response, model, messageId, ctx) { let cacheWriteTokens = 0; let noCacheTokens = -1; // -1 = 上游未提供该字段,改用减法兜底 let stopReason = null; + // 归一化后的 finishReason(mapAnthropicStopReason 之前的值),用于判定「是否正常结束」 + let finishNorm = null; + // 是否见过终态 finish 事件。CLI 用同一个标志判定流是否被截断 —— 它只认 'finish', + // 'finish-step' 不在 CLI 的事件集里,故这里同样只认 'finish'。 + let sawFinish = false; let hasError = false; let currentThinkingText = ''; // accumulated thinking text for the open block @@ -2034,7 +2128,11 @@ async function* createAnthropicSseTranslator(response, model, messageId, ctx) { // 上游的 finishReason 是 'tool-calls'(连字符),必须先过 mapFinishReason 规范化成 // 'tool_calls',否则会掉进 mapAnthropicStopReason 的 default 变成 end_turn。 // 真机实测踩到过:工具调用成功但 stop_reason 报 end_turn。 - if (event.finishReason) stopReason = mapAnthropicStopReason(mapFinishReason(event.finishReason)); + sawFinish = true; // finish-step 与 finish 都算完成信号 + if (event.finishReason) { + finishNorm = mapFinishReason(event.finishReason); + stopReason = mapAnthropicStopReason(finishNorm); + } const u = event.totalUsage || event.usage; if (u) { normalizeUsage(u); @@ -2085,8 +2183,15 @@ async function* createAnthropicSseTranslator(response, model, messageId, ctx) { const closeBlock = closeTextBlock(); if (closeBlock) yield closeBlock; + // 上游没有正常走完 finish(无 finish 事件 / provider 报连接失败): + // 绝不能补一个 end_turn 就 message_stop —— 那等于把截断谎报成完整回答。 + // 对齐 CLI:这一族一律按可重试错误处理。 + const incomplete = incompleteUpstreamDetail(sawFinish, finishNorm); + if (incomplete) { + log('warn', 'Upstream stream incomplete', { path: '/v1/messages', reason: incomplete }); + yield `event: error\ndata: ${JSON.stringify({ type: 'error', error: incompleteUpstreamError(incomplete).body.error })}\n\n`; // 输出 token 为 0 时记为错误,避免下游异常计费 - if (outputTokens === 0) { + } else if (outputTokens === 0) { yield `event: error\ndata: ${JSON.stringify({ type: 'error', error: { type: 'rate_limit_error', message: 'Empty response from upstream (zero output tokens)' }, retry_after: 10 })}\n\n`; } else { yield `event: message_delta\ndata: ${JSON.stringify({ @@ -2334,6 +2439,7 @@ async function handleMessages(req, res) { // ── 非流式 Anthropic JSON ── const messageId = 'msg_' + randomUUID().slice(0, 12); let finishReason = 'stop'; + let sawFinish = false; let usage = null; let toolCalls = null; let thinkingText = ''; // CC reasoning → Anthropic thinking block @@ -2365,8 +2471,10 @@ async function handleMessages(req, res) { }, }); break; + case 'finish-step': case 'finish': lastCcEvent = event.type; + sawFinish = true; finishReason = mapFinishReason(event.finishReason || 'stop'); if (event.totalUsage || event.usage) usage = event.totalUsage || event.usage; break; @@ -2405,6 +2513,18 @@ async function handleMessages(req, res) { return; } + // 上游没有正常走完 finish —— 对齐 CLI 按可重试 502 处理,不谎报成功 + { + const incomplete = incompleteUpstreamDetail(sawFinish, finishReason); + if (incomplete) { + log('warn', 'Upstream stream incomplete', { path: '/v1/messages', reason: incomplete }); + const err = incompleteUpstreamError(incomplete); + try { if (!abortController.signal.aborted) abortController.abort(); } catch {} + sendAnthropicError(res, err.status, err.body.error.type, err.body.error.message, err.retry_after); + return; + } + } + // 零输出判定改为按实际内容:上游偶发不回 totalUsage 时,旧逻辑(usage?.outputTokens ?? 0 === 0) // 会把有完整文本的响应误杀成 429 if (!fullText && !thinkingText && !toolCalls) { @@ -2674,14 +2794,16 @@ function buildResponsesOutput(fullText, thinkingText, toolCalls) { function buildResponsesObject(responseId, model, created, fullText, thinkingText, toolCalls, usage, opts) { const o = opts || {}; const truncated = o.finishReason === 'length'; + const paused = o.finishReason === 'pause_turn'; return { id: responseId, object: 'response', created_at: created, - status: truncated ? 'incomplete' : 'completed', + status: (truncated || paused) ? 'incomplete' : 'completed', completed_at: nowUnix(), error: null, - incomplete_details: truncated ? { reason: 'max_output_tokens' } : null, + incomplete_details: truncated ? { reason: 'max_output_tokens' } + : paused ? { reason: 'pause_turn' } : null, input: o.input || [], instructions: o.instructions === undefined ? null : o.instructions, max_output_tokens: o.max_output_tokens === undefined ? null : o.max_output_tokens, @@ -2721,6 +2843,8 @@ function createResponsesSseTranslator(model, responseId, created) { let usage = null; let textAcc = ''; let finishReason = null; + // 是否见过完成信号(见 incompleteUpstreamDetail 的口径说明) + let sawFinish = false; const baseResponse = (status, output) => ({ id: responseId, object: 'response', created_at: created, status, @@ -2847,7 +2971,10 @@ function createResponsesSseTranslator(model, responseId, created) { } case 'finish': { - finishReason = event.finishReason || null; + sawFinish = true; + // 必须归一化:截断类不止 'length'(还有 max_output_tokens / + // model_context_window_exceeded),原来直接比对原始值会漏判成 completed。 + finishReason = event.finishReason ? mapFinishReason(event.finishReason) : null; const u = event.totalUsage || event.usage || null; if (u) { normalizeUsage(u); @@ -2871,12 +2998,27 @@ function createResponsesSseTranslator(model, responseId, created) { finish() { if (!createdSent) return []; const out = closeItem(); - // finishReason=length 表示被 max_output_tokens 截断:规范要求 status=incomplete + // 上游没有正常走完 finish —— 不能报 response.completed(那是把截断谎报成完整)。 + // 对齐 CLI:按可重试的 upstream_error 处理。 + const incomplete = incompleteUpstreamDetail(sawFinish, finishReason); + if (incomplete) { + log('warn', 'Upstream stream incomplete', { path: '/v1/responses', reason: incomplete }); + out.push(sse('response.failed', { + response: Object.assign(baseResponse('failed'), { + error: { code: 'upstream_error', message: incompleteUpstreamError(incomplete).body.error.message }, + }), + })); + return out; + } + // 'length' 表示被截断(max_output_tokens / model_context_window_exceeded 都归一到这里); + // 'pause_turn' 同样是「后面还有内容没发完」,规范要求 status=incomplete。 const truncated = finishReason === 'length'; - out.push(sse(truncated ? 'response.incomplete' : 'response.completed', { - response: Object.assign(baseResponse(truncated ? 'incomplete' : 'completed', doneItems.slice()), { + const paused = finishReason === 'pause_turn'; + out.push(sse(truncated || paused ? 'response.incomplete' : 'response.completed', { + response: Object.assign(baseResponse(truncated || paused ? 'incomplete' : 'completed', doneItems.slice()), { output_text: textAcc, - incomplete_details: truncated ? { reason: 'max_output_tokens' } : null, + incomplete_details: truncated ? { reason: 'max_output_tokens' } + : paused ? { reason: 'pause_turn' } : null, usage: buildResponsesUsage(usage, this.outputTokens), }), })); @@ -3079,6 +3221,7 @@ async function handleResponses(req, res) { let thinkingText = ''; let usage = null; let finishReason = 'stop'; + let sawFinish = false; let upstreamError = null; const toolCalls = []; reader = ccResponse.body.getReader(); @@ -3108,8 +3251,10 @@ async function handleResponses(req, res) { }); break; } + case 'finish-step': case 'finish': lastCcEvent = event.type; + sawFinish = true; finishReason = mapFinishReason(event.finishReason || 'stop'); if (event.totalUsage || event.usage) usage = event.totalUsage || event.usage; break; @@ -3145,6 +3290,18 @@ async function handleResponses(req, res) { return; } + // 上游没有正常走完 finish —— 对齐 CLI 按可重试 502 处理,不谎报成功 + { + const incomplete = incompleteUpstreamDetail(sawFinish, finishReason); + if (incomplete) { + log('warn', 'Upstream stream incomplete', { path: '/v1/responses', reason: incomplete }); + const err = incompleteUpstreamError(incomplete); + try { if (!abortController.signal.aborted) abortController.abort(); } catch {} + sendResponsesError(res, err.status, err.body.error.type, err.body.error.message, err.retry_after); + return; + } + } + if (!fullText && !thinkingText && !toolCalls.length) { try { if (!abortController.signal.aborted) abortController.abort(); } catch (e2) {} sendResponsesError(res, 429, 'rate_limit_error', diff --git a/test/stream-end.test.mjs b/test/stream-end.test.mjs new file mode 100644 index 0000000..3999fb6 --- /dev/null +++ b/test/stream-end.test.mjs @@ -0,0 +1,204 @@ +// issue #38:上游「没有正常走完」的四种情形都必须如实上报,不能谎报成功。 +// 对齐 CLI(command-code@1.54.0 dist/cli.mjs): +// normalizeStopReason2 把 max_output_tokens / model_context_window_exceeded 归到 max_tokens +// isNetworkFailureFinish 把 network/connection/upstream-error 当成可重试的 502 +// 没有 finish 事件 → "Stream ended unexpectedly before completion (no finish event)" +// pause_turn → CLI 靠自动续写吸收掉,代理不续写就必须原样透出 +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import { setup } from './helpers.mjs'; + +const AUTH = { Authorization: 'Bearer user_test' }; +const CHAT = { model: 'm', messages: [{ role: 'user', content: 'hi' }] }; + +/** 造一段以给定 finishReason 收尾的 CC NDJSON */ +const withFinish = (reason) => [ + '{"type":"text-start"}', + '{"type":"text-delta","text":"partial"}', + '{"type":"text-end"}', + `{"type":"finish","finishReason":"${reason}","totalUsage":{"inputTokens":9,"outputTokens":3}}`, +]; + +/** 造一段**没有 finish 事件**就结束的 CC NDJSON(模拟上游中途被切断) */ +const NO_FINISH = [ + '{"type":"text-start"}', + '{"type":"text-delta","text":"partial"}', + '{"type":"text-end"}', +]; + +async function openaiNonStream(s, body = CHAT) { + const r = await s.proxy.post('/v1/chat/completions', body, AUTH); + return { status: r.status, json: await r.json() }; +} +async function anthropicNonStream(s) { + const r = await s.proxy.post('/v1/messages', + { model: 'm', max_tokens: 100, messages: [{ role: 'user', content: 'hi' }] }, { 'x-api-key': 'user_test' }); + return { status: r.status, json: await r.json() }; +} +async function responsesNonStream(s) { + const r = await s.proxy.post('/v1/responses', { model: 'm', input: 'hi' }, AUTH); + return { status: r.status, json: await r.json() }; +} + +// ── ① 截断类 finishReason 必须报成「截断」 ──────────────── + +test('#38 chat:max_output_tokens 报 finish_reason=length(原实现透出非法值)', async () => { + const s = await setup({ ndjson: withFinish('max_output_tokens') }); + try { + const { json } = await openaiNonStream(s); + assert.equal(json.choices[0].finish_reason, 'length', + 'max_output_tokens 是「输出被截断」,必须归到 length,不能原样透出'); + } finally { await s.close(); } +}); + +test('#38 messages:model_context_window_exceeded 报 stop_reason=max_tokens(原先谎报 end_turn)', async () => { + const s = await setup({ ndjson: withFinish('model_context_window_exceeded') }); + try { + const { json } = await anthropicNonStream(s); + assert.equal(json.stop_reason, 'max_tokens', + '上下文撑爆意味着回答没写完,报 end_turn 会让下游以为模型自己说完了'); + } finally { await s.close(); } +}); + +test('#38 responses:max_output_tokens 报 status=incomplete', async () => { + const s = await setup({ ndjson: withFinish('max_output_tokens') }); + try { + const { json } = await responsesNonStream(s); + assert.equal(json.status, 'incomplete'); + assert.deepEqual(json.incomplete_details, { reason: 'max_output_tokens' }); + } finally { await s.close(); } +}); + +// ── ② pause_turn 不能被吞掉 ────────────────────────────── + +test('#38 messages:pause_turn 原样透出(Anthropic 原生枚举,表示后面还有内容)', async () => { + const s = await setup({ ndjson: withFinish('pause_turn') }); + try { + const { json } = await anthropicNonStream(s); + assert.equal(json.stop_reason, 'pause_turn', + 'pause_turn 折成 end_turn 就是把半截回答谎报成完整的'); + } finally { await s.close(); } +}); + +test('#38 chat:pause_turn 折成 length(OpenAI 没有对应枚举,但不能折成 stop)', async () => { + const s = await setup({ ndjson: withFinish('pause_turn') }); + try { + const { json } = await openaiNonStream(s); + assert.equal(json.choices[0].finish_reason, 'length', + 'OpenAI 的 finish_reason 只有 stop|length|tool_calls|content_filter|function_call;' + + '折成 length 至少有「输出不完整」的含义,折成 stop 是谎报完成'); + } finally { await s.close(); } +}); + +test('#38 responses:pause_turn 报 status=incomplete', async () => { + const s = await setup({ ndjson: withFinish('pause_turn') }); + try { + const { json } = await responsesNonStream(s); + assert.equal(json.status, 'incomplete'); + assert.deepEqual(json.incomplete_details, { reason: 'pause_turn' }); + } finally { await s.close(); } +}); + +// ── ③ provider 报连接失败 ──────────────────────────────── + +test('#38 chat:network-error 报 502 可重试(对齐 isNetworkFailureFinish)', async () => { + const s = await setup({ ndjson: withFinish('network-error') }); + try { + const { status, json } = await openaiNonStream(s); + assert.equal(status, 502, 'CLI 对这一族一律抛可重试的 502'); + assert.equal(json.error.type, 'upstream_error'); + assert.equal(json.retry_after, 10); + } finally { await s.close(); } +}); + +test('#38 messages:connection-error 报 502 可重试', async () => { + const s = await setup({ ndjson: withFinish('connection_error') }); + try { + const { status, json } = await anthropicNonStream(s); + assert.equal(status, 502); + assert.equal(json.error.type, 'upstream_error'); + } finally { await s.close(); } +}); + +// ── ④ 根本没有 finish 事件 ─────────────────────────────── + +test('#38 chat:无 finish 事件 → 502(而不是 200 + 一个沉默的短回答)', async () => { + const s = await setup({ ndjson: NO_FINISH }); + try { + const { status, json } = await openaiNonStream(s); + assert.equal(status, 502, '流被切断时 CLI 抛 "no finish event" 的可重试 502'); + assert.equal(json.error.type, 'upstream_error'); + assert.match(json.error.message, /no finish event/); + } finally { await s.close(); } +}); + +test('#38 messages:无 finish 事件 → 502', async () => { + const s = await setup({ ndjson: NO_FINISH }); + try { + const { status, json } = await anthropicNonStream(s); + assert.equal(status, 502); + assert.equal(json.error.type, 'upstream_error'); + } finally { await s.close(); } +}); + +test('#38 responses:无 finish 事件 → 502', async () => { + const s = await setup({ ndjson: NO_FINISH }); + try { + const { status } = await responsesNonStream(s); + assert.equal(status, 502); + } finally { await s.close(); } +}); + +// ── ⑤ 流式路径同样不能补一个假的结束时 ──────────────────── + +test('#38 messages 流式:无 finish 事件 → 发 event: error,且不发 message_stop', async () => { + const s = await setup({ ndjson: NO_FINISH }); + try { + const r = await s.proxy.post('/v1/messages', + { model: 'm', max_tokens: 100, stream: true, messages: [{ role: 'user', content: 'hi' }] }, + { 'x-api-key': 'user_test' }); + const text = await r.text(); + assert.ok(text.includes('event: error'), '必须显式报错'); + assert.ok(!text.includes('event: message_stop'), + '不能补 message_stop —— 那等于告诉下游「这一轮正常结束了」'); + } finally { await s.close(); } +}); + +test('#38 chat 流式:无 finish 事件 → 发 error 对象,且不发 [DONE]', async () => { + const s = await setup({ ndjson: NO_FINISH }); + try { + const r = await s.proxy.post('/v1/chat/completions', { ...CHAT, stream: true }, AUTH); + const text = await r.text(); + assert.ok(text.includes('upstream_error'), '必须显式报错'); + assert.ok(!text.includes('[DONE]'), '发了 [DONE] 就等于谎报流正常结束'); + } finally { await s.close(); } +}); + +// ── ⑥ 正常结束不能被误伤 ───────────────────────────────── + +test('#38 回归:正常 finish 仍照常完成(三种协议)', async () => { + const s = await setup({ ndjson: withFinish('stop') }); + try { + const o = await openaiNonStream(s); + assert.equal(o.status, 200); + assert.equal(o.json.choices[0].finish_reason, 'stop'); + + const a = await anthropicNonStream(s); + assert.equal(a.status, 200); + assert.equal(a.json.stop_reason, 'end_turn'); + + const resp = await responsesNonStream(s); + assert.equal(resp.status, 200); + assert.equal(resp.json.status, 'completed'); + } finally { await s.close(); } +}); + +test('#38 回归:tool-calls 仍报 tool_use / tool_calls', async () => { + const s = await setup({ ndjson: withFinish('tool-calls') }); + try { + const o = await openaiNonStream(s); + assert.equal(o.json.choices[0].finish_reason, 'tool_calls'); + const a = await anthropicNonStream(s); + assert.equal(a.json.stop_reason, 'tool_use'); + } finally { await s.close(); } +}); From 7e5c261ebb6ed901888536343158716615d5b39c Mon Sep 17 00:00:00 2001 From: xelr233 Date: Tue, 15 Sep 2026 02:17:21 +0800 Subject: [PATCH 14/21] =?UTF-8?q?fix:=20=E6=8C=87=E7=BA=B9=20cpuCount=20?= =?UTF-8?q?=E6=94=B9=E4=B8=BA=E9=80=BB=E8=BE=91=E5=A4=84=E7=90=86=E5=99=A8?= =?UTF-8?q?=E6=95=B0=EF=BC=88=E5=8E=9F=E8=A1=A8=2015=20=E9=A1=B9=E5=85=A8?= =?UTF-8?q?=E6=98=AF=E7=89=A9=E7=90=86=E6=A0=B8=E5=BF=83=E6=95=B0=EF=BC=89?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit CLI 的 gatherRawSignals 里 cpuCount 取的是 os.cpus().length,即**逻辑处理器(线程)数**, 而 FINGERPRINT_CPUS 填的全是物理核心数 —— 15 项逐项核对,无一正确。 为什么这条不只是「不够真实」: components 里只有 machineIdHash / macHashes / osUserHash / hostnameHash / gitEmailHash 走哈希,**cpuModel 与 cpuCount 是明文上传的**,服务端可以把这一对交叉核对。 「i7-12650H + 10 线程」等价于「这台机器关掉了超线程」;原表 100% 都落在 这个罕见表述上,是群体分布层面的特征,不是单请求能看出来的那种。 改法:表里补 threads 字段,cpuCount 改用 threads。 数字逐项复核过,其中 Intel Core Ultra 9 285H 是个反直觉项: Arrow Lake 取消了超线程,6P+8E+2LPE = 16 核 == 16 线程, 所以它和 Ultra 7 155H(Meteor Lake 有超线程,22 线程)不能套同一个公式。 实现参考 @jinyu2022 的 PR #35(那份 PR 还包含 MAC OUI 真实化,本次未采纳: MAC 走的是哈希、明文不上网,收益只是外观;而它会改变 thumbmark 的输入, 等于让所有已部署的 key 换一台设备,代价大于收益)。 新增测试锁定「cpuCount == 该型号的逻辑处理器数」这一对不变量。 全套 89 项全绿。 --- proxy.mjs | 40 ++++++++++++++++++++---------------- test/fingerprint.test.mjs | 43 +++++++++++++++++++++++++++++++++++++++ 2 files changed, 66 insertions(+), 17 deletions(-) diff --git a/proxy.mjs b/proxy.mjs index 0cc09aa..aa22613 100644 --- a/proxy.mjs +++ b/proxy.mjs @@ -65,23 +65,29 @@ function loadConfig() { const CFG = loadConfig(); // ── 设备指纹(形态与哈希逐字对齐官方 CLI;1.53.1 对齐,1.54.0 复核未变) ── -// CPU 型号与核心数对应表(仅 Windows x64) +// CPU 型号与核数对应表(仅 Windows x64)。 +// ⚠️ 上网的 components.cpuCount 必须是 **threads** 而不是 cores —— +// CLI 的 gatherRawSignals 取的是 `os.cpus().length`,即**逻辑处理器数**。 +// 这一对是明文上传的(components 里只有 machineId/mac/osUser/hostname/gitEmail 走哈希), +// 所以 cpuModel 与 cpuCount 可以被服务端交叉核对:填物理核数等于宣称「这台机器关了超线程」, +// 而原表 15 项全是物理核数 —— 100% 的指纹都落在这个罕见表述上,是群体分布层面的特征。 +// (实现参考 @jinyu2022 的 PR #35,数字逐项复核过。) const FINGERPRINT_CPUS = [ - { model: '12th Gen Intel(R) Core(TM) i7-12650H', cores: 10 }, // TEMP-REVERT - { model: '12th Gen Intel(R) Core(TM) i5-12400F', cores: 6 }, - { model: '12th Gen Intel(R) Core(TM) i9-12900K', cores: 16 }, - { model: '13th Gen Intel(R) Core(TM) i7-13700K', cores: 16 }, - { model: '13th Gen Intel(R) Core(TM) i5-13600K', cores: 14 }, - { model: '13th Gen Intel(R) Core(TM) i9-13900K', cores: 24 }, - { model: 'Intel(R) Core(TM) Ultra 7 155H', cores: 16 }, - { model: 'Intel(R) Core(TM) Ultra 9 285H', cores: 16 }, - { model: 'Intel(R) Core(TM) i9-14900K', cores: 24 }, - { model: 'Intel(R) Core(TM) i7-14700K', cores: 20 }, - { model: 'AMD Ryzen 7 7800X3D', cores: 8 }, - { model: 'AMD Ryzen 9 7950X', cores: 16 }, - { model: 'AMD Ryzen 5 7600', cores: 6 }, - { model: 'AMD Ryzen 9 7900X', cores: 12 }, - { model: 'AMD Ryzen 7 5800X3D', cores: 8 }, + { model: '12th Gen Intel(R) Core(TM) i7-12650H', cores: 10, threads: 16 }, // 6P+4E + { model: '12th Gen Intel(R) Core(TM) i5-12400F', cores: 6, threads: 12 }, + { model: '12th Gen Intel(R) Core(TM) i9-12900K', cores: 16, threads: 24 }, // 8P+8E + { model: '13th Gen Intel(R) Core(TM) i7-13700K', cores: 16, threads: 24 }, // 8P+8E + { model: '13th Gen Intel(R) Core(TM) i5-13600K', cores: 14, threads: 20 }, // 6P+8E + { model: '13th Gen Intel(R) Core(TM) i9-13900K', cores: 24, threads: 32 }, // 8P+16E + { model: 'Intel(R) Core(TM) Ultra 7 155H', cores: 16, threads: 22 }, // 6P+8E+2LPE(Meteor Lake 有超线程) + { model: 'Intel(R) Core(TM) Ultra 9 285H', cores: 16, threads: 16 }, // 6P+8E+2LPE(Arrow Lake 取消超线程) + { model: 'Intel(R) Core(TM) i9-14900K', cores: 24, threads: 32 }, // 8P+16E + { model: 'Intel(R) Core(TM) i7-14700K', cores: 20, threads: 28 }, // 8P+12E + { model: 'AMD Ryzen 7 7800X3D', cores: 8, threads: 16 }, + { model: 'AMD Ryzen 9 7950X', cores: 16, threads: 32 }, + { model: 'AMD Ryzen 5 7600', cores: 6, threads: 12 }, + { model: 'AMD Ryzen 9 7900X', cores: 12, threads: 24 }, + { model: 'AMD Ryzen 7 5800X3D', cores: 8, threads: 16 }, ]; const FINGERPRINT_MEMS = [8, 16, 24, 32, 48, 64]; const FINGERPRINT_TZS = [ @@ -191,7 +197,7 @@ function generateFingerprint(apiKey) { arch: DEVICE_PROFILE.arch, osRelease: DEVICE_PROFILE.osRelease, cpuModel: cpuEntry.model, - cpuCount: cpuEntry.cores, + cpuCount: cpuEntry.threads, // 逻辑处理器数,对齐 CLI 的 os.cpus().length memGiB, isContainer: DEVICE_PROFILE.isContainer, timezone: tz, diff --git a/test/fingerprint.test.mjs b/test/fingerprint.test.mjs index b8309ca..285b0d5 100644 --- a/test/fingerprint.test.mjs +++ b/test/fingerprint.test.mjs @@ -100,3 +100,46 @@ test('不同 API key 各自初始化(互不复用指纹状态)', async () => assert.equal(fpCount, 2, '每个 key 应各自初始化一次,实际 ' + fpCount); } finally { await s.close(); } }); + +// components.cpuCount 是明文上传的,且可与同样明文的 cpuModel 交叉核对。 +// CLI 的 gatherRawSignals 取 os.cpus().length —— 逻辑处理器数,不是物理核心数。 +// 这张表是对「CLI 语义」的锁定:改了型号/核数就得同步改这里的期望值。 +const EXPECTED_THREADS = { + '12th Gen Intel(R) Core(TM) i7-12650H': 16, + '12th Gen Intel(R) Core(TM) i5-12400F': 12, + '12th Gen Intel(R) Core(TM) i9-12900K': 24, + '13th Gen Intel(R) Core(TM) i7-13700K': 24, + '13th Gen Intel(R) Core(TM) i5-13600K': 20, + '13th Gen Intel(R) Core(TM) i9-13900K': 32, + 'Intel(R) Core(TM) Ultra 7 155H': 22, + 'Intel(R) Core(TM) Ultra 9 285H': 16, // Arrow Lake 取消超线程,核数 == 线程数 + 'Intel(R) Core(TM) i9-14900K': 32, + 'Intel(R) Core(TM) i7-14700K': 28, + 'AMD Ryzen 7 7800X3D': 16, + 'AMD Ryzen 9 7950X': 32, + 'AMD Ryzen 5 7600': 12, + 'AMD Ryzen 9 7900X': 24, + 'AMD Ryzen 7 5800X3D': 16, +}; + +test('cpuCount 是逻辑处理器数(os.cpus().length),且与 cpuModel 自洽', async () => { + const s = await setup(); + try { + const keys = ['user_a', 'user_b', 'user_c', 'user_d', 'user_e', 'user_f', 'user_g', 'user_h']; + for (const k of keys) { + const r = await s.proxy.post('/v1/chat/completions', CHAT, { Authorization: 'Bearer ' + k }); + await r.text(); + } + const fps = s.mock.seen.filter(x => x.url === '/alpha/fingerprint/record') + .map(x => JSON.parse(x.raw).components); + assert.ok(fps.length >= 6, '应覆盖到足够多的指纹样本'); + for (const c of fps) { + const expected = EXPECTED_THREADS[c.cpuModel]; + assert.ok(expected !== undefined, 'cpuModel 必须在已知表内:' + c.cpuModel); + assert.equal(c.cpuCount, expected, + `${c.cpuModel} 的 cpuCount 应为逻辑处理器数 ${expected}(CLI 取 os.cpus().length),` + + '填物理核数等于宣称这台机器关了超线程 —— 而 cpuModel 与 cpuCount 都是明文,可被交叉核对'); + } + } finally { await s.close(); } +}); + From fb4407811c60a6576136aac07d52521d10ff049c Mon Sep 17 00:00:00 2001 From: xelr233 Date: Tue, 15 Sep 2026 21:47:20 +0800 Subject: [PATCH 15/21] =?UTF-8?q?fix:=20=E4=BF=AE=E6=8E=89=E7=BA=BF?= =?UTF-8?q?=E4=B8=8A=E5=88=B7=E5=B1=8F=E7=9A=84=E6=97=A5=E5=BF=97=E5=99=AA?= =?UTF-8?q?=E9=9F=B3=EF=BC=8C=E5=B9=B6=E8=AE=A9=E4=B8=8A=E6=B8=B8=20error?= =?UTF-8?q?=20=E4=BA=8B=E4=BB=B6=E8=87=AA=E5=B8=A6=E7=9A=84=20statusCode?= =?UTF-8?q?=20=E7=94=9F=E6=95=88?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 线上排查时发现两条,一真一噪: 【噪音】Unknown CC event type {"type":"text-start"} 上游每个响应都会发一串不携带内容的事件(text-start / text-end / start / start-step / reasoning-start / reasoning-end / provider-metadata / tool-input-* / tool-error)。 这些在三条**流式**翻译器里都有 case,但三条非流式路径要么缺静默列表、要么根本没有 —— Responses 非流式那条压根没有静默列表,于是每个响应都刷十来个 warn, 把真正的错误淹掉。 修法:四条路径统一成同一份静默列表(只列"无用户可见内容"的事件), default 仍然保留警告,真正没见过的类型照旧留痕。 【真问题】mapCcEventError 丢掉了上游自带的状态 CLI 的 readStreamErrorEvent 读的是 error.statusCode / error.isRetryable, 取值链:parseEmbeddedErrorJSON(message)?.status ?? error.statusCode ?? null 原实现只看 message 里的 "" 前缀,statusCode 一律被丢掉, 于是一律塌成 502 upstream_error。 后果:429/503 这类「该退避重试」的信号在代理这一层被抹平成「服务端错误」—— 客户端不再按限流退避,监控也把它错误归类成后端故障。 实测线上那条 "The request limited providers for this model and they are currently at capacity"(不含 CLI 的任何 terminal 标记:premium_credits_exhausted / model_not_in_plan / insufficient credits,即按 CLI 口径它是可重试的) 就可能因此被记成 502 而不是 429。 修法: - mapCcEventError 采纳 error.statusCode( 前缀仍优先,与 CLI 一致), 返回值增加 reportedStatus 便于区分「上游报的」与「我们映射后的」 - 四个 CC error 日志点改为先映射再记日志,并打出 upstreamStatus / upstreamRetryable / code / mappedTo —— 与作者 78353d9 对 mapCcError 的处理保持一致 测试:20 条(#38 家族)→ 全套 94 条全绿。 新增 4 条 statusCode 映射断言 + 1 条「标准序列不产生 Unknown CC 警告」。 --- proxy.mjs | 78 +++++++++++++++++++++++++++++++++++----- test/stream-end.test.mjs | 76 +++++++++++++++++++++++++++++++++++++++ 2 files changed, 146 insertions(+), 8 deletions(-) diff --git a/proxy.mjs b/proxy.mjs index aa22613..f6a2946 100644 --- a/proxy.mjs +++ b/proxy.mjs @@ -884,8 +884,16 @@ function createSseTranslator(model, completionId, created) { case 'error': { const msg = event.error?.message || event.message || 'Unknown error'; - log('warn', 'CC stream error', { message: msg }); this.upstreamError = mapCcEventError(event); + // 先映射再记日志,并把上游自带的状态/可重试性一并打出 —— + // 排查容量/限流类问题时,真正需要的就是这两个字段 + log('warn', 'CC stream error', { + message: msg, + upstreamStatus: this.upstreamError.reportedStatus, + upstreamRetryable: event.error?.isRetryable, + code: this.upstreamError.code, + mappedTo: this.upstreamError.status, + }); // Don't emit a finish_reason chunk — let the natural stream termination // handle it. Otherwise a subsequent finish(tool_calls) would be ignored // by downstream agent loops that stop at the first finish_reason. @@ -1057,8 +1065,17 @@ function mapCcError(ccStatus, ccBody) { function mapCcEventError(event) { const message = event.error?.message || event.message || 'Unknown CC error'; const code = event.error?.code || event.code || null; + // 上游 error 事件除了 message 还可能自带 statusCode / isRetryable —— + // CLI 的 readStreamErrorEvent 读的正是这两个字段,取值链是 + // parseEmbeddedErrorJSON(message)?.status ?? error.statusCode ?? null + // 原实现只看 message 里的 "" 前缀,statusCode 一律被丢掉, + // 于是 429 / 503 这类「该退避重试」的信号在代理这一层被抹平成 502「服务端错误」: + // 客户端不再按限流退避,监控也会把它错误归类成后端故障。 const statusMatch = message.match(/^<(\d{3})>/); - const ccStatus = statusMatch ? Number(statusMatch[1]) : 502; + const reportedStatus = statusMatch + ? Number(statusMatch[1]) + : (Number.isInteger(event.error?.statusCode) ? event.error.statusCode : null); + const ccStatus = reportedStatus ?? 502; const mapped = CC_STATUS_MAP[ccStatus] || { status: 502, type: 'upstream_error' }; // 与 mapCcError 保持一致:终态为 429 时带上 retry_after, @@ -1067,11 +1084,13 @@ function mapCcEventError(event) { return { status: 429, code, + reportedStatus, body: { error: { message, type: 'rate_limit_error', ...(code ? { code } : {}) }, retry_after: 30 }, }; } - return { status: mapped.status, code, body: { error: { message, type: mapped.type, ...(code ? { code } : {}) } } }; + return { status: mapped.status, code, reportedStatus, + body: { error: { message, type: mapped.type, ...(code ? { code } : {}) } } }; } // ── HTTP 请求处理 ────────────────────────────────── @@ -1626,10 +1645,23 @@ async function handleChatCompletions(req, res) { break; case 'error': lastCcEvent = event.type; - log('warn', 'CC stream error (non-stream)', { message: event.error?.message || event.message }); upstreamError = mapCcEventError(event); + log('warn', 'CC stream error (non-stream)', { + message: event.error?.message || event.message, + upstreamStatus: upstreamError.reportedStatus, + upstreamRetryable: event.error?.isRetryable, + code: upstreamError.code, + mappedTo: upstreamError.status, + }); break; - case 'reasoning-end': case 'provider-metadata': case 'tool-input-start': case 'tool-input-delta': case 'tool-input-end': case 'tool-error': case 'text-end': + // 无内容的事件:与流式翻译器的静默列表保持一致。 + // text-start / start / start-step / reasoning-start 原先只在流式路径被识别, + // 非流式路径会掉进 default 打成 'Unknown CC event type' —— 上游每个响应都会发, + // 于是线上刷屏。它们本身不携带内容(内容在 text-delta),纯粹是噪音。 + case 'text-start': case 'text-end': case 'start': case 'start-step': + case 'reasoning-start': case 'reasoning-end': + case 'provider-metadata': case 'tool-input-start': case 'tool-input-delta': case 'tool-input-end': + case 'tool-error': // Silent - no user-visible content break; default: @@ -2486,10 +2518,23 @@ async function handleMessages(req, res) { break; case 'error': lastCcEvent = event.type; - log('warn', 'CC error (Anthropic non-stream)', { message: event.error?.message || event.message }); upstreamError = mapCcEventError(event); + log('warn', 'CC error (Anthropic non-stream)', { + message: event.error?.message || event.message, + upstreamStatus: upstreamError.reportedStatus, + upstreamRetryable: event.error?.isRetryable, + code: upstreamError.code, + mappedTo: upstreamError.status, + }); break; - case 'reasoning-end': case 'provider-metadata': case 'tool-input-start': case 'tool-input-delta': case 'tool-input-end': case 'tool-error': case 'text-end': + // 无内容的事件:与流式翻译器的静默列表保持一致。 + // text-start / start / start-step / reasoning-start 原先只在流式路径被识别, + // 非流式路径会掉进 default 打成 'Unknown CC event type' —— 上游每个响应都会发, + // 于是线上刷屏。它们本身不携带内容(内容在 text-delta),纯粹是噪音。 + case 'text-start': case 'text-end': case 'start': case 'start-step': + case 'reasoning-start': case 'reasoning-end': + case 'provider-metadata': case 'tool-input-start': case 'tool-input-delta': case 'tool-input-end': + case 'tool-error': // Silent - no user-visible content break; default: @@ -3266,8 +3311,25 @@ async function handleResponses(req, res) { break; case 'error': lastCcEvent = event.type; - log('warn', 'CC stream error (non-stream)', { message: event.error ? event.error.message : event.message }); upstreamError = mapCcEventError(event); + log('warn', 'CC stream error (non-stream)', { + message: event.error ? event.error.message : event.message, + upstreamStatus: upstreamError.reportedStatus, + upstreamRetryable: event.error?.isRetryable, + code: upstreamError.code, + mappedTo: upstreamError.status, + }); + break; + // 无内容的事件:与流式翻译器以及另两条非流式路径保持一致。 + // 这条路径原先**没有静默列表**,于是上游每个响应都会发的一串无内容事件 + //(text-start / text-end / start / start-step / reasoning-start / reasoning-end / + // provider-metadata / tool-input-* / tool-error)全部掉进 default 打成 + // 'Unknown CC event type',线上刷屏、把真正的错误淹掉。 + case 'text-start': case 'text-end': case 'start': case 'start-step': + case 'reasoning-start': case 'reasoning-end': + case 'provider-metadata': case 'tool-input-start': case 'tool-input-delta': case 'tool-input-end': + case 'tool-error': + // Silent - no user-visible content break; default: log('warn', 'Unknown CC event type', { type: event.type }); diff --git a/test/stream-end.test.mjs b/test/stream-end.test.mjs index 3999fb6..e79f03b 100644 --- a/test/stream-end.test.mjs +++ b/test/stream-end.test.mjs @@ -202,3 +202,79 @@ test('#38 回归:tool-calls 仍报 tool_use / tool_calls', async () => { assert.equal(a.json.stop_reason, 'tool_use'); } finally { await s.close(); } }); + +// 回归:#38 排查期间发现的日志噪音。上游每个响应都会发一串无内容事件 +// (text-start / text-end / start / start-step / reasoning-start / reasoning-end / +// provider-metadata / tool-input-* / tool-error)。三条非流式路径原先缺少静默列表, +// 全部掉进 default 打成 'Unknown CC event type',线上刷屏并把真正的错误淹掉。 +test('标准 NDJSON 序列不产生任何 Unknown CC event type 警告(三协议 × 流式/非流式)', async () => { + const s = await setup(); + try { + const chat = { model: 'm', messages: [{ role: 'user', content: 'hi' }] }; + const msg = { model: 'm', max_tokens: 50, messages: [{ role: 'user', content: 'hi' }] }; + await (await s.proxy.post('/v1/chat/completions', { ...chat, stream: true }, AUTH)).text(); + await (await s.proxy.post('/v1/chat/completions', chat, AUTH)).text(); + await (await s.proxy.post('/v1/messages', { ...msg, stream: true }, { 'x-api-key': 'user_test' })).text(); + await (await s.proxy.post('/v1/messages', msg, { 'x-api-key': 'user_test' })).text(); + await (await s.proxy.post('/v1/responses', { model: 'm', stream: true, input: 'hi' }, AUTH)).text(); + await (await s.proxy.post('/v1/responses', { model: 'm', input: 'hi' }, AUTH)).text(); + + const logs = s.proxy.logs(); + assert.ok(!logs.includes('Unknown CC event type'), + '不应出现 Unknown CC event type 警告,实际日志片段:\n' + + logs.split('\n').filter(l => l.includes('Unknown CC')).join('\n')); + } finally { await s.close(); } +}); + + +// 上游 error 事件自带 statusCode 时必须用它 —— CLI 的 readStreamErrorEvent 读的就是这个字段, +// 取值链是 parseEmbeddedErrorJSON(message)?.status ?? error.statusCode ?? null。 +// 原实现只看 message 里的 "" 前缀,statusCode 全被丢掉 → 429/503 塌成 502。 +test('#38 error 事件带 statusCode 时按其映射(429 而非 502)', async () => { + const s = await setup({ ndjson: [ + '{"type":"text-start"}', + '{"type":"text-delta","text":"partial"}', + '{"type":"error","error":{"message":"providers are currently at capacity","statusCode":429}}', + ] }); + try { + const r = await s.proxy.post('/v1/chat/completions', CHAT, AUTH); + const j = await r.json(); + assert.equal(r.status, 429, 'statusCode 是上游给的,不能抹成 502'); + assert.equal(j.error.type, 'rate_limit_error'); + assert.equal(j.retry_after, 30, '429 要带退避提示,否则客户端不知道等多久'); + } finally { await s.close(); } +}); + +test('#38 error 事件带 statusCode 时按其映射(503 而非 502)', async () => { + const s = await setup({ ndjson: [ + '{"type":"text-start"}', + '{"type":"error","error":{"message":"service unavailable","statusCode":503}}', + ] }); + try { + const r = await s.proxy.post('/v1/chat/completions', CHAT, AUTH); + assert.equal(r.status, 503); + } finally { await s.close(); } +}); + +test('#38 error 事件没有 statusCode 时仍回落 502(保持原行为)', async () => { + const s = await setup({ ndjson: [ + '{"type":"text-start"}', + '{"type":"error","error":{"message":"something broke"}}', + ] }); + try { + const r = await s.proxy.post('/v1/chat/completions', CHAT, AUTH); + assert.equal(r.status, 502); + } finally { await s.close(); } +}); + +test('#38 message 里的 "" 前缀优先于 statusCode(对齐 CLI 的取值链)', async () => { + const s = await setup({ ndjson: [ + '{"type":"text-start"}', + '{"type":"error","error":{"message":"<400> bad request","statusCode":503}}', + ] }); + try { + const r = await s.proxy.post('/v1/chat/completions', CHAT, AUTH); + assert.equal(r.status, 400, ' 前缀是最优先的取值来源'); + } finally { await s.close(); } +}); + From f229a0c87be6b9aefe62e2aedbc3149dd40ac017 Mon Sep 17 00:00:00 2001 From: xelr233 Date: Tue, 15 Sep 2026 22:12:10 +0800 Subject: [PATCH 16/21] =?UTF-8?q?fix:=20=E6=B5=81=E7=A9=BA=E9=97=B2?= =?UTF-8?q?=E8=B6=85=E6=97=B6=E4=B8=8D=E5=86=8D=20destroy=20=E5=AE=A2?= =?UTF-8?q?=E6=88=B7=E7=AB=AF=E8=BF=9E=E6=8E=A5=EF=BC=88=E5=8F=8D=E4=BB=A3?= =?UTF-8?q?=20502=20/=20connection=20error=20=E7=9A=84=E7=9B=B4=E6=8E=A5?= =?UTF-8?q?=E6=88=90=E5=9B=A0=EF=BC=89?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 线上排障定位到:流空闲超时后走的是 res.write(`data: ${error}\n\n`); res.destroy(); res.write 是异步的,紧接着 destroy 会把尚未刷出的缓冲丢掉并发 RST。 反向代理侧看到的就是 "upstream prematurely closed connection": 响应头还没转发给客户端时回 502,已经转发了就是客户端看到 connection error / 截断的流。**两个症状同源。** 线上证据(1c2g VPS / OpenResty + systemd,资源指标全部健康: NRestarts=0、MemoryCurrent=215MB、LimitNOFILE=524288、CPU 1.5%、无 OOM): Stream idle timeout {elapsedMs:62448, bytesReceived:642575, lastCcEvent:"reasoning-delta"} Stream idle timeout {elapsedMs:87201, bytesReceived:688653, lastCcEvent:"text-delta"} Stream idle timeout {elapsedMs:139706, bytesReceived:815758, lastCcEvent:"reasoning-delta"} Stream idle timeout {elapsedMs:147005, bytesReceived:833893, lastCcEvent:"reasoning-delta"} 833KB / 147s ≈ 5.5KB/s —— 上游确实慢(推理模型 + 容量受限),30s 空闲阈值 (CC_STREAM_IDLE_MS 默认值)在这种流上会频繁误杀。超时本身也许合理, 但收尾方式不对,把"代理主动截断"变成了"代理把客户端连接搞断"。 修法:三处流式超时收尾(OpenAI / Anthropic / Responses)改为 res.end(errEvent) —— 把错误事件正常写进 SSE 流再发 FIN,客户端 SDK 能按 可重试错误处理。下游若已僵死(不读也不断),仍由 CLIENT_DRAIN_TIMEOUT_MS 那条路径负责强制断开,职责不变。 测试:新增 1 条,且**验证过有区分度** —— 把 end() 换回 destroy() 时该用例失败,客户端拿到 "TypeError: terminated" (连接被重置);换回 end() 通过。这正是线上 connection error 的复现。 test/helpers.mjs 增加 onRequest 返回 true 即"接管响应"的能力, 用来模拟"上游发了一半就长时间没新数据"。 全套 95 项全绿。 --- proxy.mjs | 16 ++++++++++------ test/helpers.mjs | 5 ++++- test/stream-end.test.mjs | 25 +++++++++++++++++++++++++ 3 files changed, 39 insertions(+), 7 deletions(-) diff --git a/proxy.mjs b/proxy.mjs index f6a2946..4305557 100644 --- a/proxy.mjs +++ b/proxy.mjs @@ -1581,8 +1581,12 @@ async function handleChatCompletions(req, res) { return; } if (!res.writableEnded) { - try { res.write(`data: ${JSON.stringify({ error: { message: timeoutMsg, type: 'rate_limit_error' }, retry_after: 5 })}\n\n`); } catch {} - try { res.destroy(); } catch {} + // 必须 end() 而不是 destroy():res.write 是异步的,紧接着 destroy 会把尚未 + // 刷出的缓冲丢掉并发 RST。反向代理看到上游连接被重置,要么回 502,要么让 + // 客户端看到 connection error —— 这正是"吐字慢 + 间歇性 502"的成因之一。 + // end() 会把错误事件正常送进 SSE 流再发 FIN,客户端 SDK 能按可重试错误处理。 + // 下游若已僵死(不读也不断),由 CLIENT_DRAIN_TIMEOUT_MS 那条路径负责兜底。 + try { res.end(`data: ${JSON.stringify({ error: { message: timeoutMsg, type: 'rate_limit_error' }, retry_after: 5 })}\n\n`); } catch {} } } else { log('error', 'Stream error', { message: e.message }); @@ -2452,8 +2456,8 @@ async function handleMessages(req, res) { const timeoutMsg = consecutiveTimeouts >= TIMEOUT_REDUCE_CONTEXT_THRESHOLD ? 'Response timeout - try reducing context length (summarize earlier messages)' : 'Response timeout - request timed out'; - try { res.write(`event: error\ndata: ${JSON.stringify({ type: 'error', error: { type: 'rate_limit_error', message: timeoutMsg }, retry_after: 5 })}\n\n`); } catch {} - try { res.destroy(); } catch {} + // end() 而不是 destroy():理由见 handleChatCompletions 流式超时分支 + try { res.end(`event: error\ndata: ${JSON.stringify({ type: 'error', error: { type: 'rate_limit_error', message: timeoutMsg }, retry_after: 5 })}\n\n`); } catch {} } } else { log('error', 'Anthropic stream error', { message: e.message }); @@ -3247,8 +3251,8 @@ async function handleResponses(req, res) { : 'Response timeout - request timed out'; if (!started) { sendResponsesError(res, 429, 'rate_limit_error', timeoutMsg, 5); return; } if (!res.writableEnded) { - try { res.write(translator.errorEvent(timeoutMsg)); } catch (e2) {} - try { res.destroy(); } catch (e2) {} + // end() 而不是 destroy():理由见 handleChatCompletions 流式超时分支 + try { res.end(translator.errorEvent(timeoutMsg)); } catch (e2) {} } } else { log('error', 'Stream error', { message: e.message, path: '/v1/responses' }); diff --git a/test/helpers.mjs b/test/helpers.mjs index 3342bb8..5feb59b 100644 --- a/test/helpers.mjs +++ b/test/helpers.mjs @@ -63,7 +63,10 @@ export async function startMockUpstream(opts = {}) { req.on('end', async () => { const raw = Buffer.concat(chunks).toString('utf8'); seen.push({ url: req.url, method: req.method, headers: req.headers, raw }); - if (opts.onRequest) await opts.onRequest(req, res, seen[seen.length - 1]); + // onRequest 返回 true 表示它自己接管了响应(可以只写一半就挂住, + // 用来模拟"上游有数据但长时间没有新数据",触发代理的空闲超时)。 + const handled = opts.onRequest ? await opts.onRequest(req, res, seen[seen.length - 1]) : false; + if (handled === true) return; if (res.writableEnded) return; const status = opts.status ?? 200; if (status !== 200) { diff --git a/test/stream-end.test.mjs b/test/stream-end.test.mjs index e79f03b..dc956ac 100644 --- a/test/stream-end.test.mjs +++ b/test/stream-end.test.mjs @@ -278,3 +278,28 @@ test('#38 message 里的 "" 前缀优先于 statusCode(对齐 CLI 的取 } finally { await s.close(); } }); + +// 流空闲超时后必须以 end() 收尾。原先走的是 res.write(err) 紧跟 res.destroy(): +// write 是异步的,destroy 会把未刷出的缓冲丢掉并发 RST,反向代理那里就是 +// "upstream prematurely closed connection" → 502,或者客户端看到 connection error。 +// 断言方式:客户端必须能**完整读到**已产生的 delta 与超时错误事件 —— destroy 会让 +// 这条读挂掉(ECONNRESET / 截断),end 则正常收束。 +test('流空闲超时:已产生的内容 + 错误事件都能完整送达(不能 destroy 客户端 socket)', async () => { + const s = await setup({ + env: { CC_STREAM_IDLE_MS: '300' }, + onRequest: (req, res) => { + res.writeHead(200, { 'Content-Type': 'text/event-stream' }); + res.write('{"type":"text-start"}\n'); + res.write('{"type":"text-delta","text":"partial-content"}\n'); + return true; // 接管后挂住:不再发任何数据 → 触发空闲超时 + }, + }); + try { + const r = await s.proxy.post('/v1/chat/completions', { ...CHAT, stream: true }, AUTH); + const text = await r.text(); + assert.equal(r.status, 200); + assert.ok(text.includes('partial-content'), '已发出的内容不能因为收尾方式而丢失'); + assert.ok(text.includes('rate_limit_error'), '超时错误事件必须完整送进流里'); + } finally { await s.close(); } +}); + From 2ae5f138ff9d03be17f0cba21ced937247934332 Mon Sep 17 00:00:00 2001 From: xelr233 Date: Tue, 15 Sep 2026 22:18:37 +0800 Subject: [PATCH 17/21] =?UTF-8?q?fix:=20=E6=98=BE=E5=BC=8F=E8=AE=BE?= =?UTF-8?q?=E7=BD=AE=20keepAliveTimeout=EF=BC=8C=E6=B6=88=E9=99=A4?= =?UTF-8?q?=E5=8F=8D=E4=BB=A3=E5=A4=8D=E7=94=A8=E5=B7=B2=E5=85=B3=E9=97=AD?= =?UTF-8?q?=E8=BF=9E=E6=8E=A5=E7=9A=84=20EPIPE?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 线上 nginx error log 的主要错误(8/10 条): sendfile() failed (32: Broken pipe) while sending request to upstream request: "POST /v1/chat/completions HTTP/1.1" upstream: "http://127.0.0.1:3050/..." 含义很具体:nginx 正在**把请求体写给后端**时,后端把连接关了。 注意是 sendfile() 而不是 writev() —— 这些请求体大到被 nginx 缓冲落盘。 而 POST 是非幂等,nginx 默认不会重试已发出的请求 → 客户端直接吃 502。 根因是 keep-alive 的时序:反代的 upstream keepalive_timeout 必须**小于** 后端的 keepAliveTimeout,否则反代会从缓存里取出一条后端已经关掉的连接。 Node 默认 keepAliveTimeout 是 5s,反代常见的 4s 只留了 1 秒余量;两边的 计时基准还不一样(反代从"读完响应放回缓存"起算,后端从"写完响应"起算), 大响应体下这点余量随时会被吃掉。线上恰恰全是 600~830KB 的流式响应。 此外 proxy 之前**没有设置过** server.keepAliveTimeout,等于把这件事完全交给 Node 默认值与反代配置的巧合 —— 部署形态(反面代理)是已知的,不该靠巧合。 改法:显式 server.keepAliveTimeout = 65s、headersTimeout = 66s (CC_KEEPALIVE_TIMEOUT_MS 可覆盖),与 Node 官方"部署在反向代理之后"的建议一致 (keepAliveTimeout > 前端 idle timeout)。启动横幅打出该值,便于与反代对齐。 反代侧仍建议把 upstream keepalive_timeout 设成 60s 以内(不是 4s)—— 现在两侧都是分钟级,余量从 1 秒变成几十秒,不再取决于抖动。 测试:新增 1 条锁定"启动横幅必须打出 keepAliveTimeout 并提示反代对应项", 全套 96 项全绿。 --- proxy.mjs | 20 ++++++++++++++++++++ test/stream-end.test.mjs | 15 +++++++++++++++ 2 files changed, 35 insertions(+) diff --git a/proxy.mjs b/proxy.mjs index 4305557..1469601 100644 --- a/proxy.mjs +++ b/proxy.mjs @@ -3489,6 +3489,25 @@ process.on('unhandledRejection', (reason) => { } }); +// ── keep-alive 时序(放在反向代理后面时是必调项) ────────────── +// 反代(nginx/OpenResty)的 upstream keepalive_timeout 必须**小于**这里的值, +// 否则反代会复用一条后端已经关掉的连接:它把请求体写过去,后端早已 FIN, +// 写这一侧就是 EPIPE —— nginx 侧表现为 +// sendfile() failed (32: Broken pipe) while sending request to upstream +// 而这条请求是 POST(非幂等),nginx 默认不会重试 → 客户端直接吃 502。 +// +// Node 默认 keepAliveTimeout=5s。反代若用常见的 4s,余量只有 1 秒;一旦反代的 +// 空闲判定基准与后端差一点(大响应体读完的时刻 vs 后端写完的时刻),就会踩上。 +// 这里显式抬到 65s,让「谁先关」不再取决于一两秒的抖动 —— 与 Node 官方在 +// 反向代理后部署的建议一致(keepAliveTimeout > 前端 idle timeout)。 +// 反代侧仍建议设 keepalive_timeout 60s 以内。 +const KEEPALIVE_TIMEOUT_MS = (() => { + const ms = Number.parseInt(process.env.CC_KEEPALIVE_TIMEOUT_MS ?? '', 10); + return Number.isFinite(ms) && ms > 0 ? ms : 65000; +})(); +server.keepAliveTimeout = KEEPALIVE_TIMEOUT_MS; +server.headersTimeout = KEEPALIVE_TIMEOUT_MS + 1000; // Node 要求 headersTimeout > keepAliveTimeout + server.listen(CFG.port, CFG.host, () => { log('info', 'CC Proxy started', { url: `http://${CFG.host}:${CFG.port}`, @@ -3499,6 +3518,7 @@ server.listen(CFG.port, CFG.host, () => { emptySystemPlaceholder: CFG.emptySystemPlaceholder ? 'on (space placeholder for requests without system prompt, issue #17)' : 'off', logFile: CFG.logFile || '(console only)', clientDrainTimeout: CLIENT_DRAIN_TIMEOUT_MS > 0 ? `${CLIENT_DRAIN_TIMEOUT_MS}ms` : 'disabled', + keepAliveTimeout: `${KEEPALIVE_TIMEOUT_MS}ms (反代侧 keepalive_timeout 必须小于它)`, idleTimeouts: `stream ${STREAM_IDLE_TIMEOUT_MS}ms / nonstream ${NONSTREAM_IDLE_TIMEOUT_MS}ms`, maxInflight: MAX_INFLIGHT > 0 ? `${MAX_INFLIGHT} (global, /health exempt)` : 'unlimited (CC_MAX_INFLIGHT=0)', upstreamProxy: UPSTREAM_PROXY || '(direct)', diff --git a/test/stream-end.test.mjs b/test/stream-end.test.mjs index dc956ac..f721504 100644 --- a/test/stream-end.test.mjs +++ b/test/stream-end.test.mjs @@ -303,3 +303,18 @@ test('流空闲超时:已产生的内容 + 错误事件都能完整送达( } finally { await s.close(); } }); + +// 反代场景的 keep-alive 时序:Node 的 keepAliveTimeout 必须**大于**反代的 +// upstream keepalive_timeout。否则反代会复用后端已关闭的连接,写请求体时吃 EPIPE, +// 而 POST 是非幂等、nginx 默认不重试 → 客户端直接 502。 +test('启动时显式设置 keepAliveTimeout 并打出(反代 keepalive_timeout 必须小于它)', async () => { + const s = await setup(); + try { + const logs = s.proxy.logs(); + assert.ok(/keepAliveTimeout[":\s]+65000ms/.test(logs), + '启动横幅必须打出 keepAliveTimeout,便于和反代配置对齐。实际:\n' + + logs.split('\n').filter(l => l.includes('CC Proxy started')).join('\n')); + assert.ok(logs.includes('keepalive_timeout'), '横幅里要提示反代侧的对应设置'); + } finally { await s.close(); } +}); + From 31ce5e270375ac3c7835e524405850852e31fcff Mon Sep 17 00:00:00 2001 From: xelr233 Date: Tue, 15 Sep 2026 23:03:30 +0800 Subject: [PATCH 18/21] =?UTF-8?q?fix:=20413=20=E6=8A=A5=E9=94=99=E9=99=84?= =?UTF-8?q?=E5=B8=A6=E5=AE=9E=E9=99=85=E8=AF=B7=E6=B1=82=E4=BD=93=E7=A7=AF?= =?UTF-8?q?=EF=BC=8C=E4=BE=BF=E4=BA=8E=E5=AE=A2=E6=88=B7=E7=AB=AF=E4=B8=8E?= =?UTF-8?q?=E8=BF=90=E7=BB=B4=E5=AE=9A=E9=98=88=E5=80=BC?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 只报上限等于让人去猜自己超了多少:客户端要据此决定拆请求还是申请提额, 运维要据此决定 CC_MAX_BODY_MB 该设多大。 nginx 开了 proxy_request_buffering 时会带 Content-Length,据此给出真实体积; 没有该头(chunked)时退回已收到多少并标注为下界。同时补一条 warn 日志便于统计。 --- proxy.mjs | 16 +++++++++++++++- test/errors-limits.test.mjs | 18 ++++++++++++++++++ 2 files changed, 33 insertions(+), 1 deletion(-) diff --git a/proxy.mjs b/proxy.mjs index 1469601..e75b5a4 100644 --- a/proxy.mjs +++ b/proxy.mjs @@ -1116,8 +1116,22 @@ function readBody(req) { settled = true; chunks.length = 0; const mb = Math.round(MAX_BODY_SIZE / 1024 / 1024); - const err = new Error(`Request body exceeds ${mb}MB limit`); + // 只报上限等于让人去猜自己超了多少 —— 客户端要据此决定"拆请求"还是"去申请提额"。 + // nginx 开了 proxy_request_buffering 时会带 Content-Length,据此给出真实体积; + // 没有该头(chunked)时退回"已收到多少",并标注它是下界。 + const declared = Number.parseInt(req.headers['content-length'] ?? '', 10); + const known = Number.isFinite(declared) && declared > 0; + const bytes = known ? declared : totalSize; + const sizeNote = ` (body is ${(bytes / 1048576).toFixed(1)}MB${known ? '' : '+'})`; + log('warn', 'Request body rejected (too large)', { + path: req.url, + limitMB: mb, + bodyMB: +(bytes / 1048576).toFixed(1), + exact: known, + }); + const err = new Error(`Request body exceeds ${mb}MB limit${sizeNote}`); err.statusCode = 413; + err.bodyBytes = bytes; reject(err); return; } diff --git a/test/errors-limits.test.mjs b/test/errors-limits.test.mjs index 30b2129..f67084a 100644 --- a/test/errors-limits.test.mjs +++ b/test/errors-limits.test.mjs @@ -141,3 +141,21 @@ test('CC_MAX_INFLIGHT 不限制探活端点', async () => { const r1 = await first; await r1.text(); } finally { await s.close(); } }); + +// 413 只报上限等于让人去猜。这条断言锁住"必须报出实际体积", +// 客户端才能判断是该拆请求还是该去申请提额。 +test('413 报错包含实际请求体积(客户端要知道超了多少)', async () => { + const s = await setup({ env: { CC_MAX_BODY_MB: '1' } }); + try { + const body = JSON.stringify({ model: 'm', stream: true, + messages: [{ role: 'user', content: 'x'.repeat(3 * 1024 * 1024) }] }); + const r = await s.proxy.post('/v1/chat/completions', body, AUTH); + assert.equal(r.status, 413); + const j = await r.json(); + assert.match(j.error.message, /exceeds 1MB limit/); + assert.match(j.error.message, /body is \d+\.\d+MB/, + '必须报出实际体积,不能只说上限。实际消息:' + j.error.message); + assert.ok(s.proxy.logs().includes('Request body rejected (too large)')); + } finally { await s.close(); } +}); + From 003f7088a24f3b88f41a5e39f62ca75981baeaf8 Mon Sep 17 00:00:00 2001 From: xelr233 Date: Tue, 15 Sep 2026 23:26:16 +0800 Subject: [PATCH 19/21] =?UTF-8?q?test:=20=E4=BF=AE=E6=8E=89=20413=20?= =?UTF-8?q?=E7=94=A8=E4=BE=8B=E9=87=8C=E7=9A=84=E6=97=A5=E5=BF=97=E7=AB=9E?= =?UTF-8?q?=E6=80=81=EF=BC=8C=E5=B9=B6=E6=8A=8A=20Node=2024=20=E5=8A=A0?= =?UTF-8?q?=E8=BF=9B=20CI=20=E7=9F=A9=E9=98=B5?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 【竞态】31ce5e2 的 CI 在 Node 20/22 上失败、18 通过: not ok 37 - 413 报错包含实际请求体积(客户端要知道超了多少) AssertionError expected: true actual: false 失败的是最后一条 assert.ok(s.proxy.logs().includes(...)) —— 代理的日志经 stdout 异步送到测试进程,可能晚于 HTTP 响应到达。同一进程里日志确实先于响应写出, 但 stdout 与 TCP 响应是两条独立通道,父进程处理顺序没有保证。Node 18 恰好赶上、 20/22 没赶上。 前两条 assert.match("exceeds 1MB limit" 与 "body is N.NMB")在三种 Node 上都通过 —— 说明**修复本身是对的**,不稳的只是我对日志的断言。改为有界轮询(最多 2s)。 【矩阵】原来只跑 18/20/22,而实际部署(systemd 服务)跑的是 Node v24.16.0 —— 生产版本不在 CI 里是明确的漏洞。补上 24,并更新注释说明选型理由。 本机 Node v24.20.0 下全套 97 项通过。 --- .github/workflows/test.yml | 5 +++-- test/errors-limits.test.mjs | 11 ++++++++++- 2 files changed, 13 insertions(+), 3 deletions(-) diff --git a/.github/workflows/test.yml b/.github/workflows/test.yml index 5ddd443..c83e269 100644 --- a/.github/workflows/test.yml +++ b/.github/workflows/test.yml @@ -22,8 +22,9 @@ jobs: strategy: fail-fast: false matrix: - # engines 下限是 18;Dockerfile 用 22;中间放 20 覆盖 LTS 跨度 - node: ['18', '20', '22'] + # engines 下限是 18;Dockerfile 用 22;中间放 20 覆盖 LTS 跨度; + # 24 是当前实际部署用的版本(systemd 里跑的就是 v24.x)—— 生产版本必须在矩阵里 + node: ['18', '20', '22', '24'] name: node ${{ matrix.node }} steps: - name: Checkout diff --git a/test/errors-limits.test.mjs b/test/errors-limits.test.mjs index f67084a..b2720f8 100644 --- a/test/errors-limits.test.mjs +++ b/test/errors-limits.test.mjs @@ -155,7 +155,16 @@ test('413 报错包含实际请求体积(客户端要知道超了多少)', a assert.match(j.error.message, /exceeds 1MB limit/); assert.match(j.error.message, /body is \d+\.\d+MB/, '必须报出实际体积,不能只说上限。实际消息:' + j.error.message); - assert.ok(s.proxy.logs().includes('Request body rejected (too large)')); + + // 日志是经 stdout 异步送到测试进程的,可能晚于 HTTP 响应到达 —— + // 直接断言会在快机器上偶发失败(实测 Node 20/22 失败、18 通过)。 + // 轮询到有界超时,既保留这条覆盖又不引入竞态。 + let sawLog = false; + for (let i = 0; i < 40 && !sawLog; i++) { + sawLog = s.proxy.logs().includes('Request body rejected (too large)'); + if (!sawLog) await new Promise(r => setTimeout(r, 50)); + } + assert.ok(sawLog, '应有一条可 grep 的 warn 日志,便于运维统计实际体积分布'); } finally { await s.close(); } }); From 26fbae0ca2370257e6b6451b0b8983b5978ba7a1 Mon Sep 17 00:00:00 2001 From: xelr233 Date: Fri, 2 Oct 2026 08:26:05 +0800 Subject: [PATCH 20/21] =?UTF-8?q?ci:=20Node=20=E7=9F=A9=E9=98=B5=E6=94=B6?= =?UTF-8?q?=E6=95=9B=E5=88=B0=2022/24=EF=BC=8C=E6=96=B0=E5=A2=9E=20Bun=20?= =?UTF-8?q?=E6=B5=8B=E8=AF=95=20job?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - 18/20 已出维护期,CI 不再覆盖(engines >=18 仅作声明) - 新增 test-bun job:Bun 运行时跑全套测试,被子测代理经 process.execPath 派生,Bun 下整条链路都是 Bun - Bun 的 node:http 基于 fetch 实现,不支持 CONNECT 方法 (发起即报错),issue #18 的两个隧道用例在 Bun 下显式跳过; 新增 npm run test:bun 便于本地对齐 --- .github/workflows/test.yml | 26 +++++++++++++++++++++++--- package.json | 1 + test/fork.test.mjs | 10 ++++++++-- 3 files changed, 32 insertions(+), 5 deletions(-) diff --git a/.github/workflows/test.yml b/.github/workflows/test.yml index c83e269..b86e2de 100644 --- a/.github/workflows/test.yml +++ b/.github/workflows/test.yml @@ -22,9 +22,9 @@ jobs: strategy: fail-fast: false matrix: - # engines 下限是 18;Dockerfile 用 22;中间放 20 覆盖 LTS 跨度; - # 24 是当前实际部署用的版本(systemd 里跑的就是 v24.x)—— 生产版本必须在矩阵里 - node: ['18', '20', '22', '24'] + # 22 是 Dockerfile 的构建版本;24 是生产部署版本(systemd 里跑的就是 v24.x)。 + # 18/20 不再覆盖,engines >=18 仅作声明;另由 test-bun 覆盖 Bun 运行时。 + node: ['22', '24'] name: node ${{ matrix.node }} steps: - name: Checkout @@ -43,3 +43,23 @@ jobs: - name: Run tests run: npm test + + test-bun: + # 社区用户有用 Bun 跑本代理的场景,单独用 Bun 跑一轮测试套件; + # 被测子进程经 process.execPath 派生,Bun 下整条链路都是 Bun。 + # Bun 缺失的能力(node:http 不支持 CONNECT)在 test/fork.test.mjs 里显式跳过。 + runs-on: ubuntu-latest + timeout-minutes: 10 + name: bun + steps: + - name: Checkout + uses: actions/checkout@v4 + + - name: Set up Bun + uses: oven-sh/setup-bun@v2 + + - name: Show Bun version + run: bun --version + + - name: Run tests + run: bun test diff --git a/package.json b/package.json index e5bc345..93b4d43 100644 --- a/package.json +++ b/package.json @@ -9,6 +9,7 @@ "start": "node proxy.mjs", "dev": "node --watch proxy.mjs", "test": "node --test test/*.test.mjs", + "test:bun": "bun test", "docker:build": "docker build -t commandcode-proxy:latest .", "docker:build:multi": "docker buildx build --platform linux/amd64,linux/arm64 -t commandcode-proxy:latest ." }, diff --git a/test/fork.test.mjs b/test/fork.test.mjs index 6591c16..e09614a 100644 --- a/test/fork.test.mjs +++ b/test/fork.test.mjs @@ -80,6 +80,12 @@ test('fork: 无 session 头时回落 per-key session,threadId 仍与之同值' }); // ── issue #18:上游 HTTP(S) 代理(零依赖 CONNECT 隧道)── +// Bun 的 node:http 基于 fetch 实现:CONNECT 方法发起即报 "fetch() URL is invalid", +// 且 http.request 的 createConnection 会被忽略(自建连接直连),隧道无法落地。 +// 所以两个「必须走隧道」的用例只在 Node 跑;「不走代理」的负向断言两个运行器都成立。 +const IS_BUN = !!process.versions.bun; +const notOnBun = IS_BUN ? test.skip : test; + /** 录制型 CONNECT 代理:记录每次 CONNECT 的 target,并做裸字节转发。 */ async function startRecordingProxy() { const port = await allocPort(); @@ -100,7 +106,7 @@ async function startRecordingProxy() { return { port, connects, close: () => new Promise(r => server.close(r)) }; } -test('#18: 配置 CC_UPSTREAM_PROXY 后上游请求经 CONNECT 隧道', async () => { +notOnBun('#18: 配置 CC_UPSTREAM_PROXY 后上游请求经 CONNECT 隧道', async () => { const rec = await startRecordingProxy(); const mock = await startMockUpstream(); const proxy = await startProxy({ upstreamPort: mock.port, @@ -117,7 +123,7 @@ test('#18: 配置 CC_UPSTREAM_PROXY 后上游请求经 CONNECT 隧道', async () } }); -test('#18: 预请求也走代理(避免同一账号从两个 IP 注册)', async () => { +notOnBun('#18: 预请求也走代理(避免同一账号从两个 IP 注册)', async () => { const rec = await startRecordingProxy(); const mock = await startMockUpstream(); const proxy = await startProxy({ upstreamPort: mock.port, From aa4a3c1ea399541eac64eb81f09961b4af4aabb3 Mon Sep 17 00:00:00 2001 From: xelr233 Date: Fri, 2 Oct 2026 14:05:56 +0800 Subject: [PATCH 21/21] =?UTF-8?q?fix:=20=E6=B5=81=E5=BC=8F=20/v1/responses?= =?UTF-8?q?=20=E9=9B=B6=E8=BE=93=E5=87=BA=E9=98=B2=E6=8A=A4=E5=A4=B1?= =?UTF-8?q?=E6=95=88=EF=BC=88#56=EF=BC=8C#54=20=E5=9B=9E=E5=BD=92=EF=BC=89?= =?UTF-8?q?=E2=80=94=E2=80=94=E7=A9=BA=E5=93=8D=E5=BA=94=E4=B8=8D=E5=86=8D?= =?UTF-8?q?=E8=B0=8E=E6=8A=A5=20completed?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit #54 把 response.created 提前到上游一返回 200 就发(治首字前 15~40s 静默期被 nginx/CDN 掐连接),translator.started(=createdSent)自此恒为 true, 「outputTokens===0 && !translator.started」成了死代码:空响应经 finish() 包装成 response.completed 谎报成功(旧版 cce214d 是 429 rate_limit_error), 与 #38/#39 修掉的「静默截断谎报成功」同类。 - 判据换成 hasOutput(是否真的产出过 output item:outputIndex>0 || doneItems>0) - 命中时若响应头已提交(常态),不能再 sendResponsesError —— 会抛 ERR_HTTP_HEADERS_SENT;按本文件既有失败口径 translator.fail → response.failed (status:"failed"、error.code:"upstream_error"、message 说明空响应) - 该分支就地 res.end():return 会跳过流式分支尾部的 res.end(), 漏掉客户端会挂在永不结束的 SSE 上 - 非流式路径未被 #54 波及(按 fullText/thinkingText/toolCalls 判空),仍是 429 - 新增 test/responses-zero-output.test.mjs(修复前跑它如预期红,修复后绿); README 两个语言的零输出防护/429 行同步修正 --- README.md | 6 +-- README_zh.md | 6 +-- proxy.mjs | 22 +++++++++-- test/responses-zero-output.test.mjs | 59 +++++++++++++++++++++++++++++ 4 files changed, 84 insertions(+), 9 deletions(-) create mode 100644 test/responses-zero-output.test.mjs diff --git a/README.md b/README.md index 5f3a394..cb3e559 100644 --- a/README.md +++ b/README.md @@ -6,7 +6,7 @@ A reverse proxy that converts Command Code API to OpenAI / Anthropic compatible Built by analyzing official CLI network traffic to accurately replicate the Command Code API request protocol, including device-fingerprint and lifecycle pre-requests. -**Features**: OpenAI Chat Completions / **Responses API (`/v1/responses`)** + Anthropic Messages API | Streaming & non-streaming | Tool calling (tool_use) | Multimodal image input | Reasoning effort | Dynamic model list | Cache hit metrics | Device fingerprint disguise (per-key, auto-refresh) | `x-api-key` auth (Anthropic SDK) | Client disconnect detection with upstream abort | Zero-output → 429 auto-retry | Consecutive timeout → 429 auto-retry | Privacy-aware logging +**Features**: OpenAI Chat Completions / **Responses API (`/v1/responses`)** + Anthropic Messages API | Streaming & non-streaming | Tool calling (tool_use) | Multimodal image input | Reasoning effort | Dynamic model list | Cache hit metrics | Device fingerprint disguise (per-key, auto-refresh) | `x-api-key` auth (Anthropic SDK) | Client disconnect detection with upstream abort | Zero-output guard (429 non-streaming / response.failed streaming) | Consecutive timeout → 429 auto-retry | Privacy-aware logging **Community**: [Linux.do](https://linux.do) — a friendly Chinese tech community. @@ -379,7 +379,7 @@ Produced by the proxy itself: | `401` | API key missing / malformed (must start with `user_`; sent via `Authorization: Bearer` or `x-api-key`) | | `404` | Unknown path | | `413` | Body exceeds `CC_MAX_BODY_MB` (connection kept alive and drained, not reset) | -| `429` | Zero output tokens, stream idle timeout (30s streaming / 90s non-streaming), or an upstream rate-limit mapping — all carry `Retry-After` so SDKs back off; after 3 consecutive timeouts a "reduce context" hint is returned | +| `429` | Zero output tokens (non-streaming only; streaming reports 200 + `response.failed`, see "Zero-Output Guard"), stream idle timeout (30s streaming / 90s non-streaming), or an upstream rate-limit mapping — all carry `Retry-After` so SDKs back off; after 3 consecutive timeouts a "reduce context" hint is returned | | `502` | CC upstream error (connection-level failures such as `fetch failed` also land here) | | `503` | `CC_MAX_INFLIGHT` is set and the in-flight cap is exceeded (`type: server_busy`) | @@ -501,7 +501,7 @@ Aligned line-by-line against the official npm package source (`command-code@1.53 | **Key Validation** | Regex `user_[a-zA-Z0-9_-]+` on `Authorization: Bearer` or `x-api-key`, auto-cleans extra paths/prefixes, rejects `sk-xxx` format | | **Stream Timeout** | 30s streaming / 90s non-streaming → 429 with SDK auto-retry | | **Consecutive Timeout** | 3 consecutive timeouts before "reduce context" hint | -| **Zero-Output Guard** | outputTokens=0 → 429 `rate_limit_error` (SDK auto-retry, anti false billing) | +| **Zero-Output Guard** | outputTokens=0 with no output item: non-streaming → 429 `rate_limit_error` (SDK auto-retry, anti false billing); streaming — where `response.created` is sent eagerly per #54 — → 200 + `response.failed` (upstream_error). Empty responses are never dressed up as success ([#56](https://github.com/MAXeaglet/commandcode-proxy/issues/56)) | | **Upstream Abort** | `AbortController` on client disconnect + all error paths | | **Privacy Logging** | No API key fragments, no error bodies, no stack traces in logs | diff --git a/README_zh.md b/README_zh.md index ed507b9..b2abd57 100644 --- a/README_zh.md +++ b/README_zh.md @@ -6,7 +6,7 @@ 逐条对齐官方 npm 包源码(`command-code@1.53.1`;`dist/cli.mjs` 只是压缩、**没有混淆**)。上游 npm 走到更高版本时代理只打**漂移告警**,不会静默改版本号(见[反检测](#反检测))。 -**完整功能**:OpenAI Chat Completions / **Responses API(`/v1/responses`)** + Anthropic Messages API | 流式/非流式输出 | 工具调用 (tool_use) | 多模态图片输入 | 推理强度 (reasoning_effort) | 动态模型列表 | 缓存命中指标 | 设备指纹伪装(per-key 绑定、自动刷新)| `x-api-key` 鉴权(Anthropic SDK)| 客户端断连检测(上游中止)| 零输出 → 429 自动重试 | 连续超时 → 429 自动重试 | 隐私保护日志 +**完整功能**:OpenAI Chat Completions / **Responses API(`/v1/responses`)** + Anthropic Messages API | 流式/非流式输出 | 工具调用 (tool_use) | 多模态图片输入 | 推理强度 (reasoning_effort) | 动态模型列表 | 缓存命中指标 | 设备指纹伪装(per-key 绑定、自动刷新)| `x-api-key` 鉴权(Anthropic SDK)| 客户端断连检测(上游中止)| 零输出防护(非流式 429 / 流式 response.failed)| 连续超时 → 429 自动重试 | 隐私保护日志 **社区**: [Linux.do](https://linux.do) — 一个友好的中文技术社区。 @@ -374,7 +374,7 @@ curl http://127.0.0.1:3050/v1/responses \ | `401` | 缺 API Key / 格式不对(Key 必须以 `user_` 开头;通过 `Authorization: Bearer` 或 `x-api-key` 传入)| | `404` | 路径不存在 | | `413` | 请求体超过 `CC_MAX_BODY_MB`(连接保持可排空,不会直接 reset)| -| `429` | 零输出 token、流空闲超时(30s 流式 / 90s 非流式)、或上游限流映射 —— 都带 `Retry-After`,SDK 自动退避重试;连续 3 次超时后提示压缩上下文 | +| `429` | 零输出 token(仅非流式;流式为 200 + `response.failed`,见「零输出防护」)、流空闲超时(30s 流式 / 90s 非流式)、或上游限流映射 —— 都带 `Retry-After`,SDK 自动退避重试;连续 3 次超时后提示压缩上下文 | | `502` | CC 上游错误(`fetch failed` 这类连接层失败也走这里)| | `503` | 开了 `CC_MAX_INFLIGHT` 且超过在途上限(`type: server_busy`)| @@ -496,7 +496,7 @@ Anthropic SDK 通过 `x-api-key` 头鉴权——代理已原生支持(无需 ` | **API Key 格式验证** | 对 `Authorization: Bearer` 或 `x-api-key` 用正则 `user_[a-zA-Z0-9_-]+` 提取,自动清理多余路径/前缀,`sk-xxx` 等非 `user_` 格式拒 | | **流式超时保护** | 流式 30s、非流式 90s → 429 + SDK 自动重试 | | **连续超时阈值** | 连续 3 次超时后才提示压缩上下文 | -| **零输出防护** | outputTokens=0 → 429 `rate_limit_error`(SDK 自动重试,反异常计费) | +| **零输出防护** | outputTokens=0 且无任何 output item:非流式 → 429 `rate_limit_error`(SDK 自动重试,反异常计费);流式因 `response.created` 已先行发出(#54),按 200 + `response.failed`(upstream_error)如实上报 —— 都不会把空响应包装成成功([#56](https://github.com/MAXeaglet/commandcode-proxy/issues/56)) | | **上游中止** | 客户端断连 + 全部错误路径 `AbortController` 打断 CC | | **隐私保护日志** | 日志不含 API Key 片段、错误 body、stack trace | diff --git a/proxy.mjs b/proxy.mjs index f629a01..250dcf3 100644 --- a/proxy.mjs +++ b/proxy.mjs @@ -3262,6 +3262,10 @@ function createResponsesSseTranslator(model, responseId, created) { // 这期间一个字节都不出网就会被中间层(实测 EdgeOne 源站 ~15s)或客户端首字节超时掐掉 start: startResponse, get started() { return createdSent; }, + // 是否真的产出过 output item(开过 item 或收尾过 item 都算)。 + // #54 之后 created/in_progress 在收到 200 时就先行发出,started 恒为 true, + // 零输出防护不能再拿它当判据 —— 只能看「有没有实际内容」。 + get hasOutput() { return outputIndex > 0 || doneItems.length > 0; }, get stopReason() { return finishReason; }, parseLine(line) { const trimmed = line.trim(); @@ -3545,10 +3549,22 @@ async function handleResponses(req, res) { } const failed = translator.fail(translator.upstreamError.body.error.message); if (failed.length) await writeEvents(failed); - } else if (translator.outputTokens === 0 && !translator.started) { + } else if (translator.outputTokens === 0 && !translator.hasOutput) { try { if (!abortController.signal.aborted) abortController.abort(); } catch (e2) {} - sendResponsesError(res, 429, 'rate_limit_error', - 'Empty response from upstream (zero output tokens)', 10); + if (!started) { + sendResponsesError(res, 429, 'rate_limit_error', + 'Empty response from upstream (zero output tokens)', 10); + return; + } + // created 已随 200 先行发出,响应头按 200 提交后状态码改不回 429 —— + // 这里再调 sendResponsesError 会抛 ERR_HTTP_HEADERS_SENT(issue #56)。 + // 按本文件既有失败口径走 response.failed:空响应绝不能经 finish() 包装成 + // response.completed 谎报成功(与 #38/#39 修掉的静默截断同类)。 + const failed = translator.fail('Empty response from upstream (zero output tokens)'); + if (failed.length) await writeEvents(failed); + // 就地收尾:下面的 return 会跳过流式分支尾部的 res.end(), + // 漏掉这条客户端会挂在永不结束的 SSE 上。 + if (!res.writableEnded) res.end(); return; } else { if (!started) { res.writeHead(200, SSE_HEADERS); started = true; } diff --git a/test/responses-zero-output.test.mjs b/test/responses-zero-output.test.mjs new file mode 100644 index 0000000..2327db6 --- /dev/null +++ b/test/responses-zero-output.test.mjs @@ -0,0 +1,59 @@ +// issue #56:#54(2eccdbf)把流式 /v1/responses 改成「上游一 200 就先发 response.created」, +// translator.started(= createdSent)自此恒为 true,零输出防护 +// translator.outputTokens === 0 && !translator.started +// 成了死代码 —— 空响应会经 finish() 包装成 response.completed 谎报成功 +//(旧版 cce214d 是 HTTP 429 + rate_limit_error)。 +// 修复口径:判据换成「是否真的产出过 output item」(hasOutput); +// 响应头已按 200 提交后状态码改不回 429(再调 sendResponsesError 会抛 +// ERR_HTTP_HEADERS_SENT),按本文件既有失败口径走 response.failed。 +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import { setup } from './helpers.mjs'; + +const AUTH = { Authorization: 'Bearer user_test' }; + +// 与 issue 里复现用的同一组上游输出:start + finish(outputTokens=0),中间零内容 +const EMPTY_STREAM = [ + '{"type":"start"}', + '{"type":"finish","finishReason":"stop","totalUsage":{"inputTokens":5,"outputTokens":0,"cachedInputTokens":0}}', +]; + +test('#56 流式:空响应必须报 response.failed,不能再谎报 response.completed', async () => { + const s = await setup({ ndjson: EMPTY_STREAM }); + try { + const r = await s.proxy.post('/v1/responses', { model: 'm', stream: true, input: 'hi' }, AUTH); + assert.equal(r.status, 200, 'created 已先行发出,HTTP 状态只能停在 200'); + const text = await r.text(); // 能正常读完 = 流有收尾;漏 res.end() 这里会挂起 + assert.ok(text.includes('event: response.failed'), '必须显式发 response.failed'); + assert.ok(text.includes('"status":"failed"'), 'response.status 必须是 failed'); + assert.ok(text.includes('"code":"upstream_error"'), '错误码按本文件既有失败口径取 upstream_error'); + assert.ok(text.includes('Empty response from upstream (zero output tokens)'), '错误消息要说明是空响应'); + assert.ok(!text.includes('response.completed'), + '零输出绝不能发 response.completed —— 那是把空响应谎报成功(#38/#39 同类问题)'); + assert.ok(!text.includes('Cannot write headers after they are sent'), + '响应头已提交后不能再走 sendResponsesError(issue 里实测的 ERR_HTTP_HEADERS_SENT 惨案)'); + } finally { await s.close(); } +}); + +test('#56 流式:有真实输出时行为不变(response.completed 正常发出)', async () => { + const s = await setup(); // 默认 ndjson:hello + outputTokens 3 + try { + const r = await s.proxy.post('/v1/responses', { model: 'm', stream: true, input: 'hi' }, AUTH); + const text = await r.text(); + assert.equal(r.status, 200); + assert.ok(text.includes('event: response.completed'), '正常输出必须照常 completed'); + assert.ok(text.includes('"status":"completed"')); + assert.ok(!text.includes('response.failed'), '不能误伤正常响应'); + } finally { await s.close(); } +}); + +test('#56 非流式:空响应仍是 HTTP 429(该路径守卫未被 #54 波及)', async () => { + const s = await setup({ ndjson: EMPTY_STREAM }); + try { + const r = await s.proxy.post('/v1/responses', { model: 'm', input: 'hi' }, AUTH); + assert.equal(r.status, 429, '非流式没有提前发响应头,旧语义(429)应原样保留'); + const json = await r.json(); + assert.equal(json.error.type, 'rate_limit_error'); + assert.equal(json.error.message, 'Empty response from upstream (zero output tokens)'); + } finally { await s.close(); } +});