diff --git a/.github/workflows/test.yml b/.github/workflows/test.yml new file mode 100644 index 0000000..b86e2de --- /dev/null +++ b/.github/workflows/test.yml @@ -0,0 +1,65 @@ +name: Test + +on: + push: + # 只在这些长期分支上跑 push;其余分支靠 pull_request 触发,避免同一提交跑两遍 + branches: [master, release] + pull_request: + workflow_dispatch: + +# 同一分支的新推送取消旧运行,避免排队浪费 +concurrency: + group: ${{ github.workflow }}-${{ github.ref }} + cancel-in-progress: true + +permissions: + contents: read + +jobs: + test: + runs-on: ubuntu-latest + timeout-minutes: 10 + strategy: + fail-fast: false + matrix: + # 22 是 Dockerfile 的构建版本;24 是生产部署版本(systemd 里跑的就是 v24.x)。 + # 18/20 不再覆盖,engines >=18 仅作声明;另由 test-bun 覆盖 Bun 运行时。 + node: ['22', '24'] + name: node ${{ matrix.node }} + steps: + - name: Checkout + uses: actions/checkout@v4 + + - name: Set up Node.js ${{ matrix.node }} + uses: actions/setup-node@v4 + with: + node-version: ${{ matrix.node }} + + - name: Show Node version + run: node --version + + - name: Syntax check + run: node --check proxy.mjs + + - name: Run tests + run: npm test + + test-bun: + # 社区用户有用 Bun 跑本代理的场景,单独用 Bun 跑一轮测试套件; + # 被测子进程经 process.execPath 派生,Bun 下整条链路都是 Bun。 + # Bun 缺失的能力(node:http 不支持 CONNECT)在 test/fork.test.mjs 里显式跳过。 + runs-on: ubuntu-latest + timeout-minutes: 10 + name: bun + steps: + - name: Checkout + uses: actions/checkout@v4 + + - name: Set up Bun + uses: oven-sh/setup-bun@v2 + + - name: Show Bun version + run: bun --version + + - name: Run tests + run: bun test diff --git a/README.md b/README.md index 34247b4..cb3e559 100644 --- a/README.md +++ b/README.md @@ -6,7 +6,7 @@ A reverse proxy that converts Command Code API to OpenAI / Anthropic compatible Built by analyzing official CLI network traffic to accurately replicate the Command Code API request protocol, including device-fingerprint and lifecycle pre-requests. -**Features**: OpenAI Chat Completions / **Responses API (`/v1/responses`)** + Anthropic Messages API | Streaming & non-streaming | Tool calling (tool_use) | Multimodal image input | Reasoning effort | Dynamic model list | Cache hit metrics | Device fingerprint disguise (per-key, auto-refresh) | `x-api-key` auth (Anthropic SDK) | Client disconnect detection with upstream abort | Zero-output → 429 auto-retry | Consecutive timeout → 429 auto-retry | Privacy-aware logging +**Features**: OpenAI Chat Completions / **Responses API (`/v1/responses`)** + Anthropic Messages API | Streaming & non-streaming | Tool calling (tool_use) | Multimodal image input | Reasoning effort | Dynamic model list | Cache hit metrics | Device fingerprint disguise (per-key, auto-refresh) | `x-api-key` auth (Anthropic SDK) | Client disconnect detection with upstream abort | Zero-output guard (429 non-streaming / response.failed streaming) | Consecutive timeout → 429 auto-retry | Privacy-aware logging **Community**: [Linux.do](https://linux.do) — a friendly Chinese tech community. @@ -104,6 +104,40 @@ authority for actual retention and provider availability. > ⚠️ **Memory amplification**: a request body exists in several copies before it reaches upstream; measured peak ≈ body size × **5.1–7.4** (7 MB → +52 MB, 20 MB → +116 MB, while a request rejected with `413` costs only ×1.05). The default `CC_MAX_BODY_MB=100` therefore implies up to ~550 MB for a **single** request, and that limit is per-request, not global. See [Memory & Deployment](#memory--deployment). +### Device fingerprint + +Relevant config: `fingerprintSalt` / `CC_FINGERPRINT_SALT`, `deviceProjectDir` / `CC_DEVICE_PROJECT_DIR`. + +The device fingerprint reported to `/alpha/fingerprint/record` is **derived deterministically** from the API key (`fpDigest(apiKey, field) = sha256(salt + "\\0" + apiKey + "\\0" + field)`), so one key is always one device: + +| Event | Old behaviour (random) | Now (derived) | +|---|---|---| +| Process restart | Map cleared → **new machine** | same machine | +| Second instance | same key = **two machines** | same machine | +| Session expiry (12h) | `keyStateStore.delete` → **new machine every 12h** | same machine | + +> The 12h case was the most visible: a real user does not replace their computer twice a day, and upstream's `device_fingerprints` table is keyed on `(userId, thumbmark)`. + +**Why derived rather than "pick a device from a hash bucket"** — a fixed pool caps entropy at the pool size, so once the number of keys exceeds it, keys *must* share a fingerprint. With ~50 keys and a 1000-entry pool, ~2 keys collide; with a 100-entry pool, ~20 do. A shared `thumbmark` under two different `userId`s is direct evidence of multi-account-same-machine — exactly what you don't want to manufacture. Derivation keeps every key a distinct device (collision probability 2⁻²⁵⁶) while still being stable. + +```bash +CC_FINGERPRINT_SALT=some-local-secret npm start # optional: bulk-reset every key's device identity +CC_DEVICE_PROJECT_DIR='C:\\Users\\you\\projects\\app' npm start # optional: change the fabricated project dir (slug follows) +``` + +The salt is optional but recommended: without it the derivation is a pure function of the API key, so anyone who knows the scheme could recompute your users' fingerprints. With it, the same key yields different devices on different deployments, at no cost. + +**The signal values are fabricated too.** The official CLI reads the real machine (Windows registry MachineGuid, NIC MACs, `os.userInfo`, `git config`); this proxy derives plausible-looking substitutes from the API key — MachineGuid's `8-4-4-4-12` shape, `xx:xx:xx:xx:xx:xx` MACs, a `DESKTOP-xxxxxx` hostname, a readable git email. Those raw values never leave process memory; only their hashes go on the wire. + +**Pool selection scores-and-takes-the-max rather than using modulo** — modulo would rotate *every* key's device whenever the pool grows; taking the max only affects keys where the new candidate happens to win. + +The hash construction follows the official CLI (`buildMachineFingerprint` / `hashSignal` in `command-code`) — `thumbmark = sha256(IB + "\0machine\0" + [machineId, macs.join(",")].join("|"))` with `IB = "command-code:device-fingerprint:v1"`, and each component hashed as `sha256(IB + "\0" + value.toLowerCase())`. The previous implementation hashed random hex without the `IB` prefix and built the thumbmark from the component *hashes*; upstream cannot recompute either way (it never sees the raw `machineId`), so it was undetectable — but it is now aligned. + +> Not addressed here: the appearance pool is still all high-end desktop CPUs and the timezone is drawn uniformly from a global pool. +> +> `timezone` is the **client machine's OS timezone** (`Intl.DateTimeFormat().resolvedOptions().timeZone`), *not* the egress IP's. So do **not** bind it to the egress IP: a user in mainland China reaching this service through a proxy normally has a machine timezone that does not match where the traffic exits, and that mismatch is the norm rather than an anomaly. +> +> The property that matters is therefore the **distribution across your own user base**, not agreement with the IP. If your users are concentrated in one region, drawing timezones uniformly from 15 global zones makes every account look like it belongs to a different continent. Set the pool to match who actually uses the deployment — this is an operator decision, and for a single-region user base it means narrowing (or weighting) `FINGERPRINT_TZS` rather than randomising it globally. ### Tool screenshot budget Images inside tool results are sent upstream as **separate** `image` blocks (an `input_image` inside @@ -345,7 +379,7 @@ Produced by the proxy itself: | `401` | API key missing / malformed (must start with `user_`; sent via `Authorization: Bearer` or `x-api-key`) | | `404` | Unknown path | | `413` | Body exceeds `CC_MAX_BODY_MB` (connection kept alive and drained, not reset) | -| `429` | Zero output tokens, stream idle timeout (30s streaming / 90s non-streaming), or an upstream rate-limit mapping — all carry `Retry-After` so SDKs back off; after 3 consecutive timeouts a "reduce context" hint is returned | +| `429` | Zero output tokens (non-streaming only; streaming reports 200 + `response.failed`, see "Zero-Output Guard"), stream idle timeout (30s streaming / 90s non-streaming), or an upstream rate-limit mapping — all carry `Retry-After` so SDKs back off; after 3 consecutive timeouts a "reduce context" hint is returned | | `502` | CC upstream error (connection-level failures such as `fetch failed` also land here) | | `503` | `CC_MAX_INFLIGHT` is set and the in-flight cap is exceeded (`type: server_busy`) | @@ -455,7 +489,7 @@ Aligned line-by-line against the official npm package source (`command-code@1.53 | Mechanism | Implementation | |-----------|---------------| | **Device Fingerprint** | `POST /alpha/fingerprint/record` before first request per key; signal values (Windows MachineGuid shape, real-shaped MACs, `DESKTOP-xxxxxx` hostname) are **derived deterministically from the API key** and hashed exactly like the CLI, so one key always reports the same device — across restarts, memory reclamation and multiple instances (bulk reset via `CC_FINGERPRINT_SALT`) | -| **Lifecycle Events** | `POST /alpha/lifecycle-events` (`cli_session_exists`, metadata `{sessionId, cliVersion, mode, os}`) sent in parallel with the fingerprint on key init | +| **Lifecycle Events** | `POST /alpha/lifecycle-events` (`cli_session_exists`, metadata `{sessionId, cliVersion, mode, os}`) sent in parallel with the fingerprint on key init, using the same `User-Agent: cli` as generate | | **Per-Key Session** | One session per API key, 12h expiry + 1h random jitter | | **Version** | `x-command-code-version` reports the **protocol version actually implemented** (currently `1.53.1`); newer npm releases only raise a drift **warning**, never a silent version bump | | **CLI Envelope** | 9 keys: `config / memory / taste / skills / permissionMode / threadId / mode / promptCache / params` | @@ -467,7 +501,7 @@ Aligned line-by-line against the official npm package source (`command-code@1.53 | **Key Validation** | Regex `user_[a-zA-Z0-9_-]+` on `Authorization: Bearer` or `x-api-key`, auto-cleans extra paths/prefixes, rejects `sk-xxx` format | | **Stream Timeout** | 30s streaming / 90s non-streaming → 429 with SDK auto-retry | | **Consecutive Timeout** | 3 consecutive timeouts before "reduce context" hint | -| **Zero-Output Guard** | outputTokens=0 → 429 `rate_limit_error` (SDK auto-retry, anti false billing) | +| **Zero-Output Guard** | outputTokens=0 with no output item: non-streaming → 429 `rate_limit_error` (SDK auto-retry, anti false billing); streaming — where `response.created` is sent eagerly per #54 — → 200 + `response.failed` (upstream_error). Empty responses are never dressed up as success ([#56](https://github.com/MAXeaglet/commandcode-proxy/issues/56)) | | **Upstream Abort** | `AbortController` on client disconnect + all error paths | | **Privacy Logging** | No API key fragments, no error bodies, no stack traces in logs | @@ -588,6 +622,8 @@ Over the limit it returns `503` + `Retry-After: 5` + `type: server_busy` — a s **Why it exists**: memory is `in-flight × (0.13 MB + 5.5 × body_MB)`. `CC_MAX_BODY_MB` bounds only the **per-request** term; nothing bounds the multiplier — at the default 100 MB, N concurrent requests can cost N × 550 MB. +> **Why the default body cap stays at 100 MB**: [#7](https://github.com/MAXeaglet/commandcode-proxy/issues/7) recorded a legitimate multimodal session (21 base64 images, ~10.11 MiB) hitting the old 10 MB cap, so the threshold cannot be lowered without breaking real usage — which is exactly why the concurrency side has to be bounded instead. The startup warning about implied worst-case memory is advisory; `CC_MAX_INFLIGHT` is the enforcement. + > ⚠️ Enabling this is **not** the same as being memory-safe: 32 × 550 MB still exceeds a small box. For a hard bound, lower `CC_MAX_BODY_MB` **as well**. ## Upstream Idle Timeouts @@ -686,7 +722,7 @@ The body exists in several copies before being forwarded: `chunks[]` / `Buffer.c | 20 MB | 100 MB | +116 MB (5.8×) | 200 | | 20 MB | 8 MB | +21 MB (1.05×) | **413** | -At startup a `warn` is logged when the implied worst case is ≥ 500 MB. The limit is **per request** and the proxy does no in-flight limiting of its own — a public deployment must add both at the reverse proxy. +At startup a `warn` is logged when the implied worst case is ≥ 500 MB. The body limit is **per request** — cap the multiplier with `CC_MAX_INFLIGHT` (in-process, global only), and add per-IP / per-key limits in the reverse proxy. A public deployment should do both. ### Suggested nginx front @@ -750,7 +786,7 @@ A more robust cap still belongs at the reverse proxy (`limit_conn`), since only - **`logFile` uses `appendFileSync`** — synchronous writes on the event loop. Under public load they serialize the loop; prefer leaving it empty and collecting stdout. - **systemd guard rails**: set `MemoryMax=` and `NODE_OPTIONS=--max-old-space-size=` so an overshoot kills the proxy, not `sshd`/`nginx`. -- **Multi-account + multiple instances**: `sessionStore` / `keyStateStore` are per-process `Map`s, so the same API key served by two instances gets two different sessions and **two different device fingerprints** — upstream sees one account on multiple machines. Scale with consistent hashing on the API key (`hash $cc_key consistent`), not round-robin. +- **Multi-account + multiple instances**: `sessionStore` is still a per-process `Map`, so the same API key served by two instances gets two different sessions. **The device fingerprint is no longer affected** — it is derived, so it is the same machine across instances and restarts (see [Device fingerprint](#device-fingerprint)). Consistent hashing on the API key (`hash $cc_key consistent`) is still recommended to keep session affinity, rather than round-robin. ## Disclaimer diff --git a/README_zh.md b/README_zh.md index efca161..b2abd57 100644 --- a/README_zh.md +++ b/README_zh.md @@ -6,7 +6,7 @@ 逐条对齐官方 npm 包源码(`command-code@1.53.1`;`dist/cli.mjs` 只是压缩、**没有混淆**)。上游 npm 走到更高版本时代理只打**漂移告警**,不会静默改版本号(见[反检测](#反检测))。 -**完整功能**:OpenAI Chat Completions / **Responses API(`/v1/responses`)** + Anthropic Messages API | 流式/非流式输出 | 工具调用 (tool_use) | 多模态图片输入 | 推理强度 (reasoning_effort) | 动态模型列表 | 缓存命中指标 | 设备指纹伪装(per-key 绑定、自动刷新)| `x-api-key` 鉴权(Anthropic SDK)| 客户端断连检测(上游中止)| 零输出 → 429 自动重试 | 连续超时 → 429 自动重试 | 隐私保护日志 +**完整功能**:OpenAI Chat Completions / **Responses API(`/v1/responses`)** + Anthropic Messages API | 流式/非流式输出 | 工具调用 (tool_use) | 多模态图片输入 | 推理强度 (reasoning_effort) | 动态模型列表 | 缓存命中指标 | 设备指纹伪装(per-key 绑定、自动刷新)| `x-api-key` 鉴权(Anthropic SDK)| 客户端断连检测(上游中止)| 零输出防护(非流式 429 / 流式 response.failed)| 连续超时 → 429 自动重试 | 隐私保护日志 **社区**: [Linux.do](https://linux.do) — 一个友好的中文技术社区。 @@ -102,6 +102,40 @@ header。该开关只是请求 Command Code 使用 ZDR-only 路由,实际数 > ⚠️ **内存放大**:请求体在转发到上游前会存在多份副本,实测峰值 ≈ body 大小 × **5.1~7.4**(7MB→+52MB、20MB→+116MB;被 `413` 拒绝的请求只要 ×1.05)。因此默认 `CC_MAX_BODY_MB=100` 意味着**单个请求**最坏可吃 ~550MB,且该上限是每请求的、不是全局的。详见[内存与部署](#内存与部署)。 +### 设备指纹 + +相关配置:`fingerprintSalt` / `CC_FINGERPRINT_SALT`、`deviceProjectDir` / `CC_DEVICE_PROJECT_DIR`。 + +上报给 `/alpha/fingerprint/record` 的设备指纹由 API key **确定性派生**(`fpDigest(apiKey, field) = sha256(salt + "\\0" + apiKey + "\\0" + field)`),因此一个 key 恒定对应一台设备: + +| 事件 | 原行为(随机) | 现行为(派生) | +|---|---|---| +| 进程重启 | Map 清空 → **换一台机器** | 同一台机器 | +| 第二个实例 | 同一 key = **两台机器** | 同一台机器 | +| session 过期(12h) | `keyStateStore.delete` → **每 12h 换一台机器** | 同一台机器 | + +> 12h 那条最明显:真实用户不会一天换两次电脑。而上游 `device_fingerprints` 表是按 `(userId, thumbmark)` 建唯一索引的。 + +**为什么用「派生」而不是「按哈希取桶选设备」** —— 固定池的熵上限就是池的大小,key 数一旦超过池容量,多个 key 就**必然**共用指纹:约 50 个 key 配 1000 个池 → 约 2 个碰撞;配 100 个池 → 约 20 个碰撞。同一个 `thumbmark` 出现在两个不同 `userId` 下,就是「多账号同机」的直接证据 —— 这正是最不该主动制造的东西。派生方案每个 key 仍是独立设备(碰撞概率 2⁻²⁵⁶),同时保持稳定。 + +```bash +CC_FINGERPRINT_SALT=some-local-secret npm start # 可选:成批更换所有 key 的设备身份 +CC_DEVICE_PROJECT_DIR='C:\\Users\\you\\projects\\app' npm start # 可选:改伪造的项目目录(slug 随之改变) +``` + +盐是可选的但建议设:不设时派生是 API key 的纯函数,知道算法的人可以反推出你所有用户的指纹;设了之后同一个 key 在不同部署上得到不同设备,且没有额外成本。 + +**信号值也是伪造的**:上游 CLI 读真实机器(Windows 注册表 MachineGuid、网卡 MAC、`os.userInfo`、`git config`),本代理按 API key 派生出一组**形状逼真**的替代值 —— MachineGuid 的 `8-4-4-4-12` 形状、`xx:xx:xx:xx:xx:xx` 的 MAC、`DESKTOP-xxxxxx` 主机名、可读的 git 邮箱。这些原始值只存在于进程内存,出网的只有它们的哈希。 + +**候选池选取用「打分取最大」而非取模** —— 取模在池子扩容时会让**所有** key 一起换设备;打分取最大只影响「新候选恰好胜出」的那部分 key。 + +哈希构造对齐官方 CLI(`command-code` 的 `buildMachineFingerprint` / `hashSignal`):`thumbmark = sha256(IB + "\0machine\0" + [machineId, macs.join(",")].join("|"))`,其中 `IB = "command-code:device-fingerprint:v1"`;各 component 按 `sha256(IB + "\0" + value.toLowerCase())` 计算。原实现直接对随机 hex 求 sha256(缺 `IB` 前缀),且 thumbmark 由各 component 的**哈希**拼成 —— 上游两种都无从验算(它拿不到原始 `machineId`),所以检测不到;现在已对齐。 + +> 本次未处理:外观池仍是清一色高端桌面 CPU,时区仍从全球池里均匀取。 +> +> `timezone` 是**客户端本机操作系统的时区**(`Intl.DateTimeFormat().resolvedOptions().timeZone`),**不是出口 IP 的时区**。所以**不要**把它绑定到出口 IP:中国大陆用户通过代理访问本服务时,本机时区与流量出口地不一致是**常态而非异常**。 +> +> 真正有意义的性质是**它在你自身用户群里的分布**,而不是与 IP 是否一致。如果你的用户集中在一个地区,却从 15 个全球时区里均匀取,就会让每个账号看起来来自不同的大洲。应当让池子匹配实际使用这个部署的人群 —— 这是运维决策:单一地区用户群应当收窄(或加权)`FINGERPRINT_TZS`,而不是全球随机。 ### 工具截图预算 工具结果里的图片会**单独**作为 `image` 块发给上游(`function_call_output.output` 里的 `input_image` @@ -340,7 +374,7 @@ curl http://127.0.0.1:3050/v1/responses \ | `401` | 缺 API Key / 格式不对(Key 必须以 `user_` 开头;通过 `Authorization: Bearer` 或 `x-api-key` 传入)| | `404` | 路径不存在 | | `413` | 请求体超过 `CC_MAX_BODY_MB`(连接保持可排空,不会直接 reset)| -| `429` | 零输出 token、流空闲超时(30s 流式 / 90s 非流式)、或上游限流映射 —— 都带 `Retry-After`,SDK 自动退避重试;连续 3 次超时后提示压缩上下文 | +| `429` | 零输出 token(仅非流式;流式为 200 + `response.failed`,见「零输出防护」)、流空闲超时(30s 流式 / 90s 非流式)、或上游限流映射 —— 都带 `Retry-After`,SDK 自动退避重试;连续 3 次超时后提示压缩上下文 | | `502` | CC 上游错误(`fetch failed` 这类连接层失败也走这里)| | `503` | 开了 `CC_MAX_INFLIGHT` 且超过在途上限(`type: server_busy`)| @@ -450,7 +484,7 @@ Anthropic SDK 通过 `x-api-key` 头鉴权——代理已原生支持(无需 ` | 机制 | 实现 | |------|------| | **设备指纹** | 每个 Key 首次请求前发送 `POST /alpha/fingerprint/record`;信号值(Windows MachineGuid 形状、真实形状的 MAC、`DESKTOP-xxxxxx` 主机名)由 API key **确定性派生**,并按 CLI 的算法哈希 —— 同一个 key 永远报告同一台设备:重启、内存回收、多实例都一致(用 `CC_FINGERPRINT_SALT` 成批换身份)| -| **生命周期声明** | Key 初始化时与指纹并行发送 `POST /alpha/lifecycle-events`(`cli_session_exists`,metadata `{sessionId, cliVersion, mode, os}`)| +| **生命周期声明** | Key 初始化时与指纹并行发送 `POST /alpha/lifecycle-events`(`cli_session_exists`,metadata `{sessionId, cliVersion, mode, os}`),与生成请求共用 `User-Agent: cli` | | **按 Key 分 Session** | 每个 API Key 独立 session,12h 过期 + 1h 随机抖动 | | **协议版本号** | `x-command-code-version` 报**实际实现的协议版本**(当前 `1.53.1`);npm 上有新版本只打**漂移告警**,不会静默改版本号 | | **CLI 信封格式** | 9 键:`config / memory / taste / skills / permissionMode / threadId / mode / promptCache / params` | @@ -462,7 +496,7 @@ Anthropic SDK 通过 `x-api-key` 头鉴权——代理已原生支持(无需 ` | **API Key 格式验证** | 对 `Authorization: Bearer` 或 `x-api-key` 用正则 `user_[a-zA-Z0-9_-]+` 提取,自动清理多余路径/前缀,`sk-xxx` 等非 `user_` 格式拒 | | **流式超时保护** | 流式 30s、非流式 90s → 429 + SDK 自动重试 | | **连续超时阈值** | 连续 3 次超时后才提示压缩上下文 | -| **零输出防护** | outputTokens=0 → 429 `rate_limit_error`(SDK 自动重试,反异常计费) | +| **零输出防护** | outputTokens=0 且无任何 output item:非流式 → 429 `rate_limit_error`(SDK 自动重试,反异常计费);流式因 `response.created` 已先行发出(#54),按 200 + `response.failed`(upstream_error)如实上报 —— 都不会把空响应包装成成功([#56](https://github.com/MAXeaglet/commandcode-proxy/issues/56)) | | **上游中止** | 客户端断连 + 全部错误路径 `AbortController` 打断 CC | | **隐私保护日志** | 日志不含 API Key 片段、错误 body、stack trace | @@ -584,6 +618,8 @@ CC_MAX_INFLIGHT=32 npm start # 最多同时处理 32 个请求 **为什么需要它**:内存 = `在途数 × (0.13MB + 5.5 × body_MB)`。`CC_MAX_BODY_MB` 只管住**单请求**量级,乘数无人管 —— 默认 100MB 时 N 个并发最坏可达 N × 550MB。 +> **为什么 body 默认值保持 100MB**:[#7](https://github.com/MAXeaglet/commandcode-proxy/issues/7) 记录了一个合法的多模态长会话(21 张 base64 图片,约 10.11 MiB)会撞上旧的 10MB 上限 —— 阈值降不下去,正因为此才必须去约束并发侧。启动时那条「最坏内存」warn 只是提示,`CC_MAX_INFLIGHT` 才是执行层。 + > ⚠️ 开启本项**不等于**内存安全:32 × 550MB 仍远超小机器容量。要拿到硬性上界,需**同时**下调 `CC_MAX_BODY_MB`。 ## 上游空闲超时 @@ -682,7 +718,7 @@ body 在转发到上游前同时存在多份副本:`chunks[]` / `Buffer.concat | 20 MB | 100 MB | +116 MB(5.8×)| 200 | | 20 MB | 8 MB | +21 MB(1.05×)| **413** | -启动时若隐含最坏峰值 ≥ 500MB,日志会输出 `warn` 提示。上限是**按请求**的,proxy 自身没有在途限流 —— 公网部署必须在反向代理层补上。 +启动时若隐含最坏峰值 ≥ 500MB,日志会输出 `warn` 提示。body 上限是**按请求**的 —— 乘数用 `CC_MAX_INFLIGHT`(进程内、仅全局)封顶,按 IP / 按 key 的限流在反向代理层补上。公网部署建议两者都做。 ### nginx 反代建议 @@ -744,7 +780,7 @@ CC_CLIENT_DRAIN_TIMEOUT_MS=60000 npm start - **`logFile` 是同步写**(`appendFileSync`),公网负载下会阻塞事件循环 —— 建议保持留空,从 stdout 收集。 - **systemd 兜底**:配 `MemoryMax=` 与 `NODE_OPTIONS=--max-old-space-size=`,让超限杀掉 proxy 而不是 `sshd`/`nginx`。 -- **多账号 + 多实例**:`sessionStore` / `keyStateStore` 是进程内 `Map`。同一个 API key 打到两个实例会得到两个不同 session 与**两个不同设备指纹**,上游会看到「一个账号在多台机器上」。横向扩展请按 API key 做一致性哈希(`hash $cc_key consistent`),不要轮询。 +- **多账号 + 多实例**:`sessionStore` 仍是进程内 `Map`,同一个 API key 打到两个实例会得到两个不同 session。**设备指纹已不再是问题** —— 它现在是派生出来的,跨实例、跨重启都是同一台设备(见[设备指纹](#设备指纹))。横向扩展仍建议按 API key 做一致性哈希(`hash $cc_key consistent`)以保持 session 亲和,不要轮询。 ## 免责声明 diff --git a/config.json b/config.json index 0d8ece2..00155b3 100644 --- a/config.json +++ b/config.json @@ -7,5 +7,9 @@ "logFile": "", "logLevel": "info", "zdr": false, - "upstreamProxy": "" + "upstreamProxy": "", + "fingerprintSalt": "", + "deviceProjectDir": "", + "cliMode": "agent", + "cliSessionMode": "interactive" } diff --git a/package.json b/package.json index e5bc345..93b4d43 100644 --- a/package.json +++ b/package.json @@ -9,6 +9,7 @@ "start": "node proxy.mjs", "dev": "node --watch proxy.mjs", "test": "node --test test/*.test.mjs", + "test:bun": "bun test", "docker:build": "docker build -t commandcode-proxy:latest .", "docker:build:multi": "docker buildx build --platform linux/amd64,linux/arm64 -t commandcode-proxy:latest ." }, diff --git a/proxy.mjs b/proxy.mjs index c83c784..250dcf3 100644 --- a/proxy.mjs +++ b/proxy.mjs @@ -64,24 +64,30 @@ function loadConfig() { const CFG = loadConfig(); -// ── 设备指纹(形态与哈希逐字对齐官方 CLI 1.53.1) ────── -// CPU 型号与核心数对应表(仅 Windows x64) +// ── 设备指纹(形态与哈希逐字对齐官方 CLI;1.53.1 对齐,1.54.0 复核未变) ── +// CPU 型号与核数对应表(仅 Windows x64)。 +// ⚠️ 上网的 components.cpuCount 必须是 **threads** 而不是 cores —— +// CLI 的 gatherRawSignals 取的是 `os.cpus().length`,即**逻辑处理器数**。 +// 这一对是明文上传的(components 里只有 machineId/mac/osUser/hostname/gitEmail 走哈希), +// 所以 cpuModel 与 cpuCount 可以被服务端交叉核对:填物理核数等于宣称「这台机器关了超线程」, +// 而原表 15 项全是物理核数 —— 100% 的指纹都落在这个罕见表述上,是群体分布层面的特征。 +// (实现参考 @jinyu2022 的 PR #35,数字逐项复核过。) const FINGERPRINT_CPUS = [ - { model: '12th Gen Intel(R) Core(TM) i7-12650H', cores: 10 }, // TEMP-REVERT - { model: '12th Gen Intel(R) Core(TM) i5-12400F', cores: 6 }, - { model: '12th Gen Intel(R) Core(TM) i9-12900K', cores: 16 }, - { model: '13th Gen Intel(R) Core(TM) i7-13700K', cores: 16 }, - { model: '13th Gen Intel(R) Core(TM) i5-13600K', cores: 14 }, - { model: '13th Gen Intel(R) Core(TM) i9-13900K', cores: 24 }, - { model: 'Intel(R) Core(TM) Ultra 7 155H', cores: 16 }, - { model: 'Intel(R) Core(TM) Ultra 9 285H', cores: 16 }, - { model: 'Intel(R) Core(TM) i9-14900K', cores: 24 }, - { model: 'Intel(R) Core(TM) i7-14700K', cores: 20 }, - { model: 'AMD Ryzen 7 7800X3D', cores: 8 }, - { model: 'AMD Ryzen 9 7950X', cores: 16 }, - { model: 'AMD Ryzen 5 7600', cores: 6 }, - { model: 'AMD Ryzen 9 7900X', cores: 12 }, - { model: 'AMD Ryzen 7 5800X3D', cores: 8 }, + { model: '12th Gen Intel(R) Core(TM) i7-12650H', cores: 10, threads: 16 }, // 6P+4E + { model: '12th Gen Intel(R) Core(TM) i5-12400F', cores: 6, threads: 12 }, + { model: '12th Gen Intel(R) Core(TM) i9-12900K', cores: 16, threads: 24 }, // 8P+8E + { model: '13th Gen Intel(R) Core(TM) i7-13700K', cores: 16, threads: 24 }, // 8P+8E + { model: '13th Gen Intel(R) Core(TM) i5-13600K', cores: 14, threads: 20 }, // 6P+8E + { model: '13th Gen Intel(R) Core(TM) i9-13900K', cores: 24, threads: 32 }, // 8P+16E + { model: 'Intel(R) Core(TM) Ultra 7 155H', cores: 16, threads: 22 }, // 6P+8E+2LPE(Meteor Lake 有超线程) + { model: 'Intel(R) Core(TM) Ultra 9 285H', cores: 16, threads: 16 }, // 6P+8E+2LPE(Arrow Lake 取消超线程) + { model: 'Intel(R) Core(TM) i9-14900K', cores: 24, threads: 32 }, // 8P+16E + { model: 'Intel(R) Core(TM) i7-14700K', cores: 20, threads: 28 }, // 8P+12E + { model: 'AMD Ryzen 7 7800X3D', cores: 8, threads: 16 }, + { model: 'AMD Ryzen 9 7950X', cores: 16, threads: 32 }, + { model: 'AMD Ryzen 5 7600', cores: 6, threads: 12 }, + { model: 'AMD Ryzen 9 7900X', cores: 12, threads: 24 }, + { model: 'AMD Ryzen 7 5800X3D', cores: 8, threads: 16 }, ]; const FINGERPRINT_MEMS = [8, 16, 24, 32, 48, 64]; const FINGERPRINT_TZS = [ @@ -92,6 +98,18 @@ const FINGERPRINT_TZS = [ ]; const FINGERPRINT_MAC_COUNT_RANGE = [2, 3, 4, 5]; // 随机 2~5 个 MAC +// ── 为什么指纹必须由 apiKey 确定性派生(而不是随机) ────────────── +// 指纹代表「这个账号对应的那台设备」。原实现每进程随机生成一台新机器, +// 会在三个场景下毫无理由地漂移:① 进程重启 ② 多实例各算各的 +// ③ session 过期清理连带 keyStateStore.delete。设备身份无故更换本身就是信号。 +// +// 也不能退化成「按 key 取哈希桶选设备」:桶方案的熵上限就是桶数,key 数一旦 +// 超过桶数必然出现多个 key 共用指纹,而上游 device_fingerprints 表对 +// (userId, thumbmark) 建了唯一索引 —— 共用指纹等于「多账号同机」的直接证据。 +// 派生方案每个 key 都是独立设备,碰撞概率 2^-256。 +// +// 原始信号值(MachineGuid / MAC / 主机名 / git 邮箱)只存在于本进程内存, +// 出网的永远只有它们的哈希。 // CLI 的根盐(buildMachineFingerprint 常量 sb) const FP_SALT = 'command-code:device-fingerprint:v1'; // 设备档案:指纹 / config.environment / config.workingDir / x-project-slug / lifecycle.os 共用同一份, @@ -179,7 +197,7 @@ function generateFingerprint(apiKey) { arch: DEVICE_PROFILE.arch, osRelease: DEVICE_PROFILE.osRelease, cpuModel: cpuEntry.model, - cpuCount: cpuEntry.cores, + cpuCount: cpuEntry.threads, // 逻辑处理器数,对齐 CLI 的 os.cpus().length memGiB, isContainer: DEVICE_PROFILE.isContainer, timezone: tz, @@ -221,12 +239,19 @@ async function checkProtocolDrift() { checkProtocolDrift(); // 启动时立即检查 setInterval(checkProtocolDrift, CC_VERSION_REFRESH_MS); -// 请求体大小上限:默认 100MB,可用环境变量 CC_MAX_BODY_MB 覆盖(正整数,单位 MB) +// 请求体大小上限:默认 100MB,可用环境变量 CC_MAX_BODY_MB 覆盖(正整数,单位 MB)。 +// 默认值保持 100MB(573e260 为修 issue #7 设定)——不要下调,理由是有实际证据的: +// #7 的真实触发场景是多模态长会话,21 张 Base64 图片累积到约 10.11 MiB 的合法请求, +// 下调到 4~8MB 会把这类请求整体挡在门外。#7 的「Connection error」症状由 +// 「超限返回 413 + 排空连接」这条路径解决,与阈值取值无关;但阈值决定了功能边界, +// 所以默认值必须容纳真实的多模态上下文。 // ⚠️ 内存特性(issue #20 实测):请求体在转发到上游前会同时存在多份副本 —— // chunks[] / Buffer.concat / utf8 字符串 / JSON.parse 对象树 / buildCcRequest 重建对象树 / JSON.stringify 序列化体。 // 实测峰值 ≈ body 大小 × 5.1~7.4(7MB→+52MB,20MB→+116MB;而 413 拒绝路径只要 ×1.05)。 // 故 100MB 上限意味着「单个请求」最坏可吃 ~550MB,且该上限是每请求的、不是全局的。 -// 公网/多用户部署请在反向代理层同时限制 body 大小与在途请求数(见 README「内存与部署」)。 +// 正因为阈值必须留足,#20 的第二半必须补齐:并发侧要有约束。 +// 进程内可用 CC_MAX_INFLIGHT,边缘侧用反向代理 limit_conn +// (见 README「内存与部署」)。 const MAX_BODY_SIZE = (() => { const mb = Number.parseInt(process.env.CC_MAX_BODY_MB ?? '', 10); return Number.isFinite(mb) && mb > 0 ? mb * 1024 * 1024 : 100 * 1024 * 1024; @@ -345,14 +370,17 @@ function ensureSession(apiKey) { return sessionId; } -// 定期清理过期 session 和 key 状态,防止 Map 无限增长 +// 定期清理过期 session,防止 Map 无限增长。 +// 注意:这里**不再**连带删除 keyStateStore。原实现写的是「同时清理该 key 的指纹状态」, +// 但指纹是随机生成的,删掉就等于该 key 每 12h 换一台"电脑" —— 真实用户不会这样。 +// 指纹状态现在有自己的空闲淘汰(见下方),且因为指纹是派生出来的, +// 即使被淘汰、下次重新派生得到的仍是同一台设备,不构成漂移。 setInterval(() => { const now = Date.now(); let cleaned = 0; for (const [key, entry] of sessionStore) { if (now >= entry.expiresAt) { sessionStore.delete(key); - keyStateStore.delete(key); // 同时清理该 key 的指纹状态 cleaned++; } } @@ -374,12 +402,20 @@ function getSessionId(incomingHeaders, apiKey, promptCacheKey) { return ensureSession(apiKey); } -// 每个请求独立 thread ID -function newThreadId() { return randomUUID(); } +// CC 线上信封的 threadId 只接受合法 UUID —— 对应 CLI 的 toWireThreadId: +// uuid.safeParse 失败即返回 undefined,该键随之被 JSON.stringify 丢弃。 +// 非 UUID 的 session 值一律不发,避免上游看到真 CLI 永远不会产生的 threadId。 +function isWireUuid(v) { + return typeof v === 'string' + && /^[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}$/i.test(v); +} // ── 每 Key 独立状态(fingerprint + 初始化节流) ── -// 每个 API Key 拥有自己的设备指纹和初始化定时器 -const keyStateStore = new Map(); // apiKey → { fingerprint, nextInitAt } +// 指纹现在是确定性派生的,所以这个 Map 只是缓存:淘汰它不会改变该 key 的设备身份, +// 只是下次多算一次 HMAC。淘汰按「空闲」而非「session 过期」判定, +// 这样活跃 key 的 nextInitAt 不会被重置(否则会重复发送指纹/lifecycle 预请求)。 +const KEY_STATE_IDLE_MS = 24 * 60 * 60 * 1000; // 24h 未活动即淘汰 +const keyStateStore = new Map(); // apiKey → { fingerprint, nextInitAt, lastSeen } function getOrCreateKeyState(apiKey) { let state = keyStateStore.get(apiKey); @@ -387,13 +423,28 @@ function getOrCreateKeyState(apiKey) { state = { fingerprint: generateFingerprint(apiKey), nextInitAt: 0, + lastSeen: 0, }; keyStateStore.set(apiKey, state); - log('info', 'Fingerprint generated for key', { keyPrefix: apiKey.slice(0, 8) }); + log('info', 'Fingerprint derived for key', { + keyPrefix: apiKey.slice(0, 8), + thumbmark: state.fingerprint.thumbmark.slice(0, 12), + }); } + state.lastSeen = Date.now(); return state; } +// 空闲淘汰:纯内存卫生,不影响设备身份(见上) +setInterval(() => { + const now = Date.now(); + let evicted = 0; + for (const [key, state] of keyStateStore) { + if (now - state.lastSeen > KEY_STATE_IDLE_MS) { keyStateStore.delete(key); evicted++; } + } + if (evicted > 0) log('info', 'Key state evicted (idle)', { evicted, remaining: keyStateStore.size }); +}, 60 * 60 * 1000); // 每小时 + // ── 初始化预请求(fingerprint + lifecycle,首次 + 每 8h+2h 抖动) ──── const INIT_REFRESH_MS = 8 * 60 * 60 * 1000; // 8h const INIT_JITTER_MS = 2 * 60 * 60 * 1000; // 2h 抖动 @@ -407,6 +458,11 @@ async function ensureInitialized(apiKey, signal) { // 并行发两个预请求 const headers = { 'Content-Type': 'application/json', + // 官方 CLI 的指纹预请求与生成请求共用同一个 header 常量表(lb = vy = "cli"), + // 因此预请求同样是 User-Agent: cli。此前只有 forwardToCC 设了它, + // 预请求走 Node 默认的 "node" —— 同一账号的指纹注册与生成请求来自两种 UA, + // 是可直接观测的破绽。 + 'User-Agent': 'cli', 'x-cli-environment': 'production', 'Authorization': `Bearer ${apiKey}`, 'x-command-code-version': CC_VERSION, @@ -551,6 +607,7 @@ function buildCcRequest(openaiReq) { const chatMessages = messages.filter(m => m.role !== 'system' && m.role !== 'developer'); // Build tool_call_id → tool_name reverse lookup + // 名字原样透传(理由见 toWireToolName 删除处的注释)。 const toolNameMap = {}; for (const msg of chatMessages) { if (msg.role === 'assistant' && msg.tool_calls) { @@ -617,6 +674,7 @@ function buildCcRequest(openaiReq) { return { role: 'assistant', content: parts }; } if (msg.role === 'tool') { + // toolName 与上面 tool-call 里的一致(同一张 map),不重命名 return { role: 'tool', content: [{ @@ -686,9 +744,10 @@ function buildCcRequest(openaiReq) { body.params.reasoning_effort = reasoning_effort; } // CLI 总是下发 tools(没有工具时是空数组)—— 空数组与缺键在 wire 上可观测,这里对齐 - // CLI 的 toWireTools:只有 name / description / input_schema,没有 type 字段 + // CLI 的 toWireTools:只有 name / description / input_schema,没有 type 字段; + // 且**不做**任何名字重写(理由见下面「工具名重写整段删除」的注释) body.params.tools = (tools || []).map(t => ({ - name: toWireToolName(t.function?.name || t.name || ''), + name: t.function?.name || t.name || '', description: t.function?.description || t.description || '', input_schema: t.function?.parameters || t.input_schema || { type: 'object', properties: {} }, })); @@ -711,14 +770,31 @@ function buildCcRequest(openaiReq) { return body; } -// CLI 发送前会重写部分工具名(resolveToolNameAlias / ow 表) -const TOOL_NAME_ALIASES = { - bash_output: 'shell_output', - task_output: 'shell_output', - tool_search: 'search_tools', - read_multiple_files: 'read_file', -}; -function toWireToolName(name) { return TOOL_NAME_ALIASES[name] || name; } +// ── 工具名为什么一个都不重命名(已删掉的别名表的墓志铭) ────────────── +// 上游 d063b47 曾引入一张 4 项别名表并作用于 params.tools[].name。查 CLI 源码后 +// (command-code@1.54.0 dist/cli.mjs)确认:**wire 协议里没有工具重命名这回事**。 +// +// CLI 里确实存在两个改名的函数,但都不适用于反代: +// +// toWireToolName(e){return e===rw?nw:e} // rw="tool_search" → nw="search_tools" +// —— 只作用在 toWireMessages(tool-call 与 tool-result,两边同名), +// 不作用在 toWireTools(声明原样下发)。它的存在前提是 CLI 自己退役过 +// tool_search 这个名字:createRetiredToolSearchTool 给的 visible:()=>false, +// 该工具从不进 params.tools,只有重放旧会话时历史里才会残留这个旧名。 +// +// resolveToolNameAlias(ow 表) // bash_output/task_output/read_multiple_files +// —— 被工具执行器调用:模型喊了退役名时本地按新名跑,回一句给模型看的 +// "Repair note",并按 defaults 补参(task_output 会补 wait:"exit")。 +// 这是执行语义、不是 wire 变换。 +// +// 反代没有这个前提:params.tools 由下游客户端给出,proxy 没有 catalog、没有退役名, +// 请求里出现的每个名字对 proxy 来说都是当前名。若强行重命名,一旦客户端恰好声明了 +// 一个叫 tool_search 的工具,就会变成「声明 tool_search、消息 search_tools」—— +// 下游按自己声明的名字派发不到工具(issue #36 / #37 的根因)。 +// +// 因此这里全程原样透传。若将来真要支持「重放真实 CLI 旧会话」,正确做法是**入站** +// 归一化 + 显式开关,并连 defaults 一起补,与 resolveToolNameAlias 同语义; +// 绝不要做成「上行改、下行不改」。 // CLI 的 toWireToolOutput:只取文本块,用 '\n' 拼接 function toWireToolOutputValue(content) { @@ -1069,8 +1145,22 @@ function readBody(req) { settled = true; chunks.length = 0; const mb = Math.round(MAX_BODY_SIZE / 1024 / 1024); - const err = new Error(`Request body exceeds ${mb}MB limit`); + // 只报上限等于让人去猜自己超了多少 —— 客户端要据此决定"拆请求"还是"去申请提额"。 + // nginx 开了 proxy_request_buffering 时会带 Content-Length,据此给出真实体积; + // 没有该头(chunked)时退回"已收到多少",并标注它是下界。 + const declared = Number.parseInt(req.headers['content-length'] ?? '', 10); + const known = Number.isFinite(declared) && declared > 0; + const bytes = known ? declared : totalSize; + const sizeNote = ` (body is ${(bytes / 1048576).toFixed(1)}MB${known ? '' : '+'})`; + log('warn', 'Request body rejected (too large)', { + path: req.url, + limitMB: mb, + bodyMB: +(bytes / 1048576).toFixed(1), + exact: known, + }); + const err = new Error(`Request body exceeds ${mb}MB limit${sizeNote}`); err.statusCode = 413; + err.bodyBytes = bytes; reject(err); return; } @@ -1306,7 +1396,7 @@ async function forwardToCC(body, apiKey, incomingHeaders = {}, signal, promptCac const sessionId = getSessionId(incomingHeaders, apiKey, promptCacheKey); // CLI 的 toWireThreadId:只有合法 UUID 才放进信封,否则整个键省略。 // 同时按 CLI 的键顺序重排:config, memory, taste, skills, permissionMode, threadId, mode, params - if (/^[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}$/i.test(String(sessionId))) { + if (isWireUuid(sessionId)) { const ordered = {}; for (const k of ['config', 'memory', 'taste', 'skills', 'permissionMode']) ordered[k] = body[k]; ordered.threadId = sessionId; @@ -1369,6 +1459,10 @@ async function handleChatCompletions(req, res) { // 构建 CC 请求体 const ccBody = buildCcRequest(openaiReq); + // issue #20:ccBody 建好后,openaiReq 这棵 20MB 级对象树只剩 prompt_cache_key 还被用到。 + // 先取出该值再断开引用,让这一份副本可以更早被 GC 回收(原来是整段请求期间一直活着)。 + const promptCacheKey = openaiReq.prompt_cache_key; + openaiReq = null; // AbortController 用于客户端断连时真正打断 CC 上游(pi-commandcode-provider 模式) // 每次尝试都换一个新的(已 abort 的 signal 不可复用) @@ -1419,7 +1513,7 @@ async function handleChatCompletions(req, res) { // 首次初始化(fingerprint + lifecycle) await ensureInitialized(apiKey, abortController.signal); // 转发到 CC API(传入客户端 headers,用于提取 session ID) - const ccResponse = await forwardToCC(ccBody, apiKey, req.headers, abortController.signal, openaiReq.prompt_cache_key); + const ccResponse = await forwardToCC(ccBody, apiKey, req.headers, abortController.signal, promptCacheKey); if (!ccResponse.ok) { const errorText = await ccResponse.text().catch(() => ''); @@ -1703,7 +1797,7 @@ async function handleChatCompletions(req, res) { // 非流式路径会掉进 default 打成 'Unknown CC event type' —— 上游每个响应都会发, // 于是线上刷屏。它们本身不携带内容(内容在 text-delta),纯粹是噪音。 case 'text-start': case 'text-end': case 'start': case 'start-step': - case 'reasoning-start': case 'reasoning-end': case 'finish-step': + case 'reasoning-start': case 'reasoning-end': case 'provider-metadata': case 'tool-input-start': case 'tool-input-delta': case 'tool-input-end': case 'tool-error': // Silent - no user-visible content @@ -2375,8 +2469,12 @@ async function handleMessages(req, res) { const model = anthropicReq.model || 'claude-sonnet-4-6'; // Convert Anthropic → OpenAI → CC - const openaiReq = convertAnthropicToOpenAI(anthropicReq); + let openaiReq = convertAnthropicToOpenAI(anthropicReq); const ccBody = buildCcRequest(openaiReq); + // issue #20:ccBody 已建好,原始请求树(anthropicReq)与中间树(openaiReq)都不再被引用, + // 显式断开以便尽早回收 —— 否则它们会和 ccBody 一起活到整段请求结束。 + openaiReq = null; + anthropicReq = null; const abortController = new AbortController(); let aborted = false; @@ -2613,7 +2711,7 @@ async function handleMessages(req, res) { // 非流式路径会掉进 default 打成 'Unknown CC event type' —— 上游每个响应都会发, // 于是线上刷屏。它们本身不携带内容(内容在 text-delta),纯粹是噪音。 case 'text-start': case 'text-end': case 'start': case 'start-step': - case 'reasoning-start': case 'reasoning-end': case 'finish-step': + case 'reasoning-start': case 'reasoning-end': case 'provider-metadata': case 'tool-input-start': case 'tool-input-delta': case 'tool-input-end': case 'tool-error': // Silent - no user-visible content @@ -3164,6 +3262,10 @@ function createResponsesSseTranslator(model, responseId, created) { // 这期间一个字节都不出网就会被中间层(实测 EdgeOne 源站 ~15s)或客户端首字节超时掐掉 start: startResponse, get started() { return createdSent; }, + // 是否真的产出过 output item(开过 item 或收尾过 item 都算)。 + // #54 之后 created/in_progress 在收到 200 时就先行发出,started 恒为 true, + // 零输出防护不能再拿它当判据 —— 只能看「有没有实际内容」。 + get hasOutput() { return outputIndex > 0 || doneItems.length > 0; }, get stopReason() { return finishReason; }, parseLine(line) { const trimmed = line.trim(); @@ -3447,10 +3549,22 @@ async function handleResponses(req, res) { } const failed = translator.fail(translator.upstreamError.body.error.message); if (failed.length) await writeEvents(failed); - } else if (translator.outputTokens === 0 && !translator.started) { + } else if (translator.outputTokens === 0 && !translator.hasOutput) { try { if (!abortController.signal.aborted) abortController.abort(); } catch (e2) {} - sendResponsesError(res, 429, 'rate_limit_error', - 'Empty response from upstream (zero output tokens)', 10); + if (!started) { + sendResponsesError(res, 429, 'rate_limit_error', + 'Empty response from upstream (zero output tokens)', 10); + return; + } + // created 已随 200 先行发出,响应头按 200 提交后状态码改不回 429 —— + // 这里再调 sendResponsesError 会抛 ERR_HTTP_HEADERS_SENT(issue #56)。 + // 按本文件既有失败口径走 response.failed:空响应绝不能经 finish() 包装成 + // response.completed 谎报成功(与 #38/#39 修掉的静默截断同类)。 + const failed = translator.fail('Empty response from upstream (zero output tokens)'); + if (failed.length) await writeEvents(failed); + // 就地收尾:下面的 return 会跳过流式分支尾部的 res.end(), + // 漏掉这条客户端会挂在永不结束的 SSE 上。 + if (!res.writableEnded) res.end(); return; } else { if (!started) { res.writeHead(200, SSE_HEADERS); started = true; } @@ -3552,7 +3666,7 @@ async function handleResponses(req, res) { // provider-metadata / tool-input-* / tool-error)全部掉进 default 打成 // 'Unknown CC event type',线上刷屏、把真正的错误淹掉。 case 'text-start': case 'text-end': case 'start': case 'start-step': - case 'reasoning-start': case 'reasoning-end': case 'finish-step': + case 'reasoning-start': case 'reasoning-end': case 'provider-metadata': case 'tool-input-start': case 'tool-input-delta': case 'tool-input-end': case 'tool-error': // Silent - no user-visible content @@ -3750,6 +3864,7 @@ server.listen(CFG.port, CFG.host, () => { upstreamRetry: UPSTREAM_RETRY_MAX > 0 ? `${UPSTREAM_RETRY_MAX} retries, base ${UPSTREAM_RETRY_BASE_MS}ms (only before first byte)` : 'disabled (CC_UPSTREAM_RETRY_MAX=0)', + fingerprint: `derived from API key (salt: ${CFG.fingerprintSalt ? 'set' : 'unset'})`, }); if (CLIENT_DRAIN_TIMEOUT_MS > 0) { log('info', 'Client drain timeout enabled', { timeoutMs: CLIENT_DRAIN_TIMEOUT_MS }); diff --git a/test/endpoints.test.mjs b/test/endpoints.test.mjs new file mode 100644 index 0000000..878adde --- /dev/null +++ b/test/endpoints.test.mjs @@ -0,0 +1,103 @@ +// 端点契约:四个路由在正常路径下的行为。 +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import { setup } from './helpers.mjs'; + +test('POST /v1/chat/completions 流式:返回 OpenAI SSE 且内容正确', async () => { + const s = await setup(); + try { + const r = await s.proxy.post('/v1/chat/completions', + { model: 'deepseek/deepseek-v4-flash', messages: [{ role: 'user', content: 'hi' }], stream: true }, + { Authorization: 'Bearer user_test' }); + assert.equal(r.status, 200); + const text = await r.text(); + assert.match(text, /data: /); + assert.ok(text.includes('hello'), 'SSE 应包含上游 text-delta 的内容'); + assert.match(text, /\[DONE\]/, '流应以 [DONE] 结束'); + } finally { await s.close(); } +}); + +test('POST /v1/chat/completions 非流式:返回 chat.completion 对象', async () => { + const s = await setup(); + try { + const r = await s.proxy.post('/v1/chat/completions', + { model: 'deepseek/deepseek-v4-flash', messages: [{ role: 'user', content: 'hi' }] }, + { Authorization: 'Bearer user_test' }); + assert.equal(r.status, 200); + const j = await r.json(); + assert.equal(j.object, 'chat.completion'); + assert.equal(j.choices[0].message.content, 'hello'); + assert.ok(j.usage, 'usage 必须存在'); + } finally { await s.close(); } +}); + +test('POST /v1/messages 流式:返回 Anthropic SSE 事件序列', async () => { + const s = await setup(); + try { + const r = await s.proxy.post('/v1/messages', + { model: 'deepseek/deepseek-v4-flash', max_tokens: 100, messages: [{ role: 'user', content: 'hi' }], stream: true }, + { 'x-api-key': 'user_test' }); + assert.equal(r.status, 200); + const text = await r.text(); + for (const ev of ['message_start', 'content_block_start', 'message_stop']) { + assert.ok(text.includes(ev), 'Anthropic SSE 应包含 ' + ev); + } + } finally { await s.close(); } +}); + +test('POST /v1/responses 流式:返回具名 SSE 事件', async () => { + const s = await setup(); + try { + const r = await s.proxy.post('/v1/responses', + { model: 'deepseek/deepseek-v4-flash', input: 'hi', stream: true }, + { Authorization: 'Bearer user_test' }); + assert.equal(r.status, 200); + const text = await r.text(); + assert.ok(text.includes('response.completed'), '应包含 response.completed'); + // 规范要求每个事件都带 sequence_number + assert.ok(text.includes('sequence_number'), '每个事件都必须带 sequence_number'); + } finally { await s.close(); } +}); + +test('GET /v1/models 走 /provider/v1/models', async () => { + const s = await setup({ env: { CC_USE_PROVIDER_MODELS: 'true' }, onRequest: (req, res) => { + if (req.url === '/provider/v1/models') { + res.writeHead(200, { 'Content-Type': 'application/json' }); + res.end(JSON.stringify({ data: [{ id: 'test/model-1', object: 'model' }] })); + } + }}); + try { + const r = await s.proxy.get('/v1/models', { headers: { Authorization: 'Bearer user_test' } }); + assert.equal(r.status, 200); + const j = await r.json(); + assert.ok(Array.isArray(j.data)); + assert.equal(j.data[0].id, 'test/model-1'); + } finally { await s.close(); } +}); + +test('GET /health 与 / 返回存活状态', async () => { + const s = await setup(); + try { + for (const path of ['/health', '/']) { + const r = await s.proxy.get(path); + assert.equal(r.status, 200, path + ' 应返回 200'); + } + } finally { await s.close(); } +}); + +test('未知路由返回 404', async () => { + const s = await setup(); + try { + const r = await s.proxy.get('/nope'); + assert.equal(r.status, 404); + } finally { await s.close(); } +}); + +test('缺少 API key 返回 401', async () => { + const s = await setup(); + try { + const r = await s.proxy.post('/v1/chat/completions', + { model: 'deepseek/deepseek-v4-flash', messages: [{ role: 'user', content: 'hi' }] }); + assert.equal(r.status, 401); + } finally { await s.close(); } +}); diff --git a/test/envelope.test.mjs b/test/envelope.test.mjs new file mode 100644 index 0000000..05dc24d --- /dev/null +++ b/test/envelope.test.mjs @@ -0,0 +1,138 @@ +// 信封与请求头契约:断言 /alpha/generate 的**键序**与 header 集合。 +// 依据官方 CLI 源码(command-code@1.54.0 dist/cli.mjs,未混淆): +// buildCommandAuthHeaders(e) → { 'Content-Type', 'User-Agent': "cli", [CLI_VERSION], +// [CLI_ENVIRONMENT], [PROJECT_SLUG], [TASTE_LEARNING], [SESSION_ID], Authorization, ... } +// 常量表 Sy 共 10 项,**没有 x-co-flag**。 +// 信封键序:config, memory, taste, skills, permissionMode, threadId, mode, params。 +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import { setup } from './helpers.mjs'; + +const AUTH = { Authorization: 'Bearer user_test' }; +const UUID = '3f2504e0-4f89-11d3-9a0c-0305e82c3301'; + +async function generate(proxy, mock, path, body, headers = AUTH) { + const r = await proxy.post(path, body, headers); + await r.text(); + const g = mock.lastGenerate(); + assert.ok(g, '应至少产生一条 /alpha/generate'); + return { status: r.status, raw: g.raw, body: g.body, headers: g.headers }; +} + +test('信封键序与 CLI 一致:config…permissionMode, threadId, mode, params', async () => { + const s = await setup(); + try { + const { body } = await generate(s.proxy, s.mock, '/v1/chat/completions', + { model: 'm', stream: true, messages: [{ role: 'user', content: 'hi' }] }, + { ...AUTH, 'x-session-id': UUID }); + assert.deepEqual(Object.keys(body), + ['config', 'memory', 'taste', 'skills', 'permissionMode', 'threadId', 'mode', 'params']); + assert.equal(body.threadId, UUID, 'threadId 必须与 x-session-id 同值'); + } finally { await s.close(); } +}); + +test('session 非 UUID 时 threadId 整键省略,键序仍为 CLI 顺序', async () => { + const s = await setup(); + try { + const { body } = await generate(s.proxy, s.mock, '/v1/chat/completions', + { model: 'm', stream: true, messages: [{ role: 'user', content: 'hi' }] }, + { ...AUTH, 'x-session-id': 'my-cache-key-001' }); + assert.ok(!('threadId' in body), '非 UUID 不得下发 threadId'); + assert.deepEqual(Object.keys(body), + ['config', 'memory', 'taste', 'skills', 'permissionMode', 'mode', 'params']); + } finally { await s.close(); } +}); + +test('skills 是 null(CLI 发字面 null,不是空串)', async () => { + const s = await setup(); + try { + const { body } = await generate(s.proxy, s.mock, '/v1/chat/completions', + { model: 'm', stream: true, messages: [{ role: 'user', content: 'hi' }] }); + assert.equal(body.skills, null); + assert.equal(body.memory, null); + assert.equal(body.taste, null); + assert.equal(body.mode, 'agent', '信封 mode 默认 agent'); + } finally { await s.close(); } +}); + +test('请求头集合与 CLI 的 buildCommandAuthHeaders 一致(无 x-co-flag)', async () => { + const s = await setup(); + try { + const { headers } = await generate(s.proxy, s.mock, '/v1/chat/completions', + { model: 'm', stream: true, messages: [{ role: 'user', content: 'hi' }] }); + assert.equal(headers['user-agent'], 'cli'); + assert.ok(!('x-co-flag' in headers), 'CLI 常量表 Sy 里没有 x-co-flag,不得发送'); + assert.equal(headers['x-cli-environment'], 'production'); + assert.equal(headers['x-taste-learning'], 'false', 'toString() 的字符串,不是布尔'); + assert.equal(headers['x-project-slug'], 'c-users-dev-projects-app', + 'x-project-slug = slugify(workingDir),盘符保留(@sindresorhus/slugify 的结果)'); + assert.ok(headers['x-command-code-version']); + assert.ok(headers.traceparent); + } finally { await s.close(); } +}); + +test('config.workingDir 与 x-project-slug 同源,且不泄露宿主 cwd', async () => { + const s = await setup(); + try { + const { body, headers } = await generate(s.proxy, s.mock, '/v1/chat/completions', + { model: 'm', stream: true, messages: [{ role: 'user', content: 'hi' }] }); + const slugify = p => p.toLowerCase().replace(/[^a-z0-9]+/g, '-').replace(/^-+|-+$/g, '') || 'root'; + assert.equal(headers['x-project-slug'], slugify(body.config.workingDir), + 'slug 必须等于 slugify(workingDir),否则两者自相矛盾'); + assert.ok(!body.config.workingDir.includes(process.cwd()), + '不得把宿主机真实 cwd 交给上游'); + assert.equal(body.config.environment, 'win32', 'environment 是平台词,不是 "win32-x64, Node.js v…"'); + } finally { await s.close(); } +}); + +test('tools 总是下发:无工具时是空数组而非缺键(对齐 CLU 的 toWireTools)', async () => { + const s = await setup(); + try { + const { body } = await generate(s.proxy, s.mock, '/v1/chat/completions', + { model: 'm', stream: true, messages: [{ role: 'user', content: 'hi' }] }); + assert.ok(Array.isArray(body.params.tools), 'params.tools 必须存在'); + assert.equal(body.params.tools.length, 0); + } finally { await s.close(); } +}); + +test('tools 形态:只有 name/description/input_schema,没有 type 字段', async () => { + const s = await setup(); + try { + const { body } = await generate(s.proxy, s.mock, '/v1/chat/completions', { + model: 'm', stream: true, messages: [{ role: 'user', content: 'hi' }], + tools: [{ type: 'function', function: { name: 'get_weather', description: 'w', parameters: { type: 'object', properties: {} } } }], + }); + assert.deepEqual(Object.keys(body.params.tools[0]), ['name', 'description', 'input_schema']); + assert.equal(body.params.tools[0].name, 'get_weather'); + } finally { await s.close(); } +}); + +test('多模态 image_url 转成 CC image 块并带 mimeType', async () => { + const s = await setup(); + try { + const { body } = await generate(s.proxy, s.mock, '/v1/chat/completions', { + model: 'm', stream: true, + messages: [{ role: 'user', content: [ + { type: 'text', text: 'what is this' }, + { type: 'image_url', image_url: { url: 'data:image/png;base64,iVBORw0KGgo=' } }, + ] }], + }); + const img = body.params.messages[0].content.find(p => p.type === 'image'); + assert.ok(img, '应产出 image 块'); + assert.equal(img.mimeType, 'image/png', 'CLI 的 toWireMessages 会补 mimeType'); + assert.equal(img.image, 'data:image/png;base64,iVBORw0KGgo='); + } finally { await s.close(); } +}); + +test('lifecycle metadata 的 mode 用另一个枚举(CC_CLI_SESSION_MODE)', async () => { + const s = await setup({ env: { CC_CLI_SESSION_MODE: 'non-interactive', CC_CLI_MODE: 'compact' } }); + try { + const r = await s.proxy.post('/v1/chat/completions', + { model: 'm', stream: true, messages: [{ role: 'user', content: 'hi' }] }, AUTH); + await r.text(); + const lc = s.mock.seen.find(x => x.url === '/alpha/lifecycle-events'); + assert.equal(JSON.parse(lc.raw).metadata.mode, 'non-interactive'); + assert.equal(s.mock.lastGenerate().body.mode, 'compact', + '信封 mode 与 lifecycle mode 是两个独立配置项'); + } finally { await s.close(); } +}); diff --git a/test/errors-limits.test.mjs b/test/errors-limits.test.mjs new file mode 100644 index 0000000..b2720f8 --- /dev/null +++ b/test/errors-limits.test.mjs @@ -0,0 +1,170 @@ +// 错误映射 + 请求体上限 + 在途上限。 +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import { setup } from './helpers.mjs'; + +const AUTH = { Authorization: 'Bearer user_test' }; +const CHAT = { model: 'm', stream: true, messages: [{ role: 'user', content: 'hi' }] }; + +// CC_STATUS_MAP 的语义契约(逐条来自 proxy.mjs 的映射表): +// 400/401/404 原样透传;422 -> 400;403 -> 401;402 -> 429(payment → rate limit) +// 500/502 -> 502;503 -> 503;未列出的状态 -> 502 upstream_error +const MAPPING = [ + [400, 400], [401, 401], [404, 404], + [422, 400], [403, 401], [402, 429], + [500, 502], [502, 502], [503, 503], + [418, 502], // 未在表中 → upstream_error +]; + +for (const [upstream, expected] of MAPPING) { + test('状态映射:上游 ' + upstream + ' -> 下游 ' + expected, async () => { + const s = await setup({ status: upstream, errorBody: JSON.stringify({ error: { message: 'mock' } }) }); + try { + const r = await s.proxy.post('/v1/chat/completions', CHAT, AUTH); + assert.equal(r.status, expected, '上游 ' + upstream + ' 应映射为 ' + expected); + const j = await r.json(); + assert.ok(j.error && j.error.type, '响应体应含 error.type'); + } finally { await s.close(); } + }); +} + +test('上游 402(payment required)映射为 429 而非 402', async () => { + const s = await setup({ status: 402, errorBody: JSON.stringify({ error: { message: 'insufficient credits' } }) }); + try { + const r = await s.proxy.post('/v1/chat/completions', CHAT, AUTH); + assert.equal(r.status, 429, '402 是 CC 的余额语义,下游按限流处理'); + const j = await r.json(); + assert.match(j.error.message, /insufficient credits/, '上游错误体应透传进 message'); + } finally { await s.close(); } +}); + +test('上游 429 映射为 rate_limit_error 并带 retry_after', async () => { + const s = await setup({ status: 429, errorBody: JSON.stringify({ error: { message: 'slow down' } }) }); + try { + const r = await s.proxy.post('/v1/chat/completions', CHAT, AUTH); + const j = await r.json(); + assert.equal(r.status, 429); + assert.equal(j.error.type, 'rate_limit_error'); + } finally { await s.close(); } +}); + +test('上游 5xx 映射为 502/503 类服务端错误', async () => { + const s = await setup({ status: 500, errorBody: JSON.stringify({ error: { message: 'boom' } }) }); + try { + const r = await s.proxy.post('/v1/chat/completions', CHAT, AUTH); + assert.ok(r.status >= 500, '上游 500 应映射成 5xx,实际 ' + r.status); + } finally { await s.close(); } +}); + +test('Anthropic 端点:上游错误以 Anthropic 错误体返回', async () => { + const s = await setup({ status: 429, errorBody: JSON.stringify({ error: { message: 'slow down' } }) }); + try { + const r = await s.proxy.post('/v1/messages', + { model: 'm', max_tokens: 10, stream: true, messages: [{ role: 'user', content: 'hi' }] }, + { 'x-api-key': 'user_test' }); + const j = await r.json(); + assert.ok(j.error || j.type, 'Anthropic 错误体应有 error 或 type 字段'); + } finally { await s.close(); } +}); + +test('非法 JSON 请求体返回 400', async () => { + const s = await setup(); + try { + const r = await s.proxy.post('/v1/chat/completions', '{not json', AUTH); + assert.equal(r.status, 400); + } finally { await s.close(); } +}); + +test('超过 CC_MAX_BODY_MB 的请求返回 413,且之后小请求仍可用', async () => { + const s = await setup({ env: { CC_MAX_BODY_MB: '1' } }); + try { + const big = JSON.stringify({ model: 'm', stream: true, + messages: [{ role: 'user', content: 'x'.repeat(2 * 1024 * 1024) }] }); + const r1 = await s.proxy.post('/v1/chat/completions', big, AUTH); + assert.equal(r1.status, 413, '超限请求应返回 413(而非直接 reset)'); + + // issue #7:超限后连接必须保持可排空,后续请求不受影响 + const r2 = await s.proxy.post('/v1/chat/completions', + { model: 'm', stream: true, messages: [{ role: 'user', content: 'small' }] }, AUTH); + assert.equal(r2.status, 200, '超限拒绝后小请求仍应正常'); + await r2.text(); + } finally { await s.close(); } +}); + +test('默认上限放行多模态量级的请求(issue #7 场景,~9MB)', async () => { + const s = await setup(); + try { + const body = JSON.stringify({ model: 'm', stream: true, + messages: [{ role: 'user', content: 'x'.repeat(9 * 1024 * 1024) }] }); + const r = await s.proxy.post('/v1/chat/completions', body, AUTH); + assert.equal(r.status, 200, '默认 100MB 上限必须容纳 #7 的多模态长会话'); + await r.text(); + } finally { await s.close(); } +}); + +test('CC_MAX_INFLIGHT 限流:超限返回 503 + Retry-After,且名额会释放', async () => { + // 让上游把请求挂住,以便观察在途名额 + const s = await setup({ env: { CC_MAX_INFLIGHT: '1' }, + onRequest: async (req, res) => { if (req.url === '/alpha/generate') await new Promise(r => setTimeout(r, 1200)); } }); + try { + const first = s.proxy.post('/v1/chat/completions', CHAT, AUTH); + await new Promise(r => setTimeout(r, 400)); // 确保第一个已占住名额 + + const r2 = await s.proxy.post('/v1/chat/completions', CHAT, AUTH); + assert.equal(r2.status, 503, '超出在途上限应返回 503'); + assert.equal(r2.headers.get('retry-after'), '5', '应带 Retry-After'); + const j = await r2.json(); + assert.equal(j.error.type, 'server_busy'); + + const r1 = await first; + assert.equal(r1.status, 200, '已占住名额的请求应正常完成'); + await r1.text(); + + // 名额释放后新请求应通行 + const r3 = await s.proxy.post('/v1/chat/completions', CHAT, AUTH); + assert.equal(r3.status, 200, '名额释放后应恢复通行'); + await r3.text(); + } finally { await s.close(); } +}); + +test('CC_MAX_INFLIGHT 不限制探活端点', async () => { + const s = await setup({ env: { CC_MAX_INFLIGHT: '1' }, + onRequest: async (req, res) => { if (req.url === '/alpha/generate') await new Promise(r => setTimeout(r, 1200)); } }); + try { + const first = s.proxy.post('/v1/chat/completions', CHAT, AUTH); + await new Promise(r => setTimeout(r, 400)); + // 探活端点被占满时仍必须可用,否则编排系统会误判容器已死 + for (const path of ['/health', '/']) { + const r = await s.proxy.get(path); + assert.equal(r.status, 200, path + ' 不应受在途上限影响'); + } + const r1 = await first; await r1.text(); + } finally { await s.close(); } +}); + +// 413 只报上限等于让人去猜。这条断言锁住"必须报出实际体积", +// 客户端才能判断是该拆请求还是该去申请提额。 +test('413 报错包含实际请求体积(客户端要知道超了多少)', async () => { + const s = await setup({ env: { CC_MAX_BODY_MB: '1' } }); + try { + const body = JSON.stringify({ model: 'm', stream: true, + messages: [{ role: 'user', content: 'x'.repeat(3 * 1024 * 1024) }] }); + const r = await s.proxy.post('/v1/chat/completions', body, AUTH); + assert.equal(r.status, 413); + const j = await r.json(); + assert.match(j.error.message, /exceeds 1MB limit/); + assert.match(j.error.message, /body is \d+\.\d+MB/, + '必须报出实际体积,不能只说上限。实际消息:' + j.error.message); + + // 日志是经 stdout 异步送到测试进程的,可能晚于 HTTP 响应到达 —— + // 直接断言会在快机器上偶发失败(实测 Node 20/22 失败、18 通过)。 + // 轮询到有界超时,既保留这条覆盖又不引入竞态。 + let sawLog = false; + for (let i = 0; i < 40 && !sawLog; i++) { + sawLog = s.proxy.logs().includes('Request body rejected (too large)'); + if (!sawLog) await new Promise(r => setTimeout(r, 50)); + } + assert.ok(sawLog, '应有一条可 grep 的 warn 日志,便于运维统计实际体积分布'); + } finally { await s.close(); } +}); + diff --git a/test/fingerprint.test.mjs b/test/fingerprint.test.mjs new file mode 100644 index 0000000..285b0d5 --- /dev/null +++ b/test/fingerprint.test.mjs @@ -0,0 +1,145 @@ +// 指纹 / lifecycle 预请求协议契约。 +// 这三条预请求是 CC 侧设备识别的入口,字段形状与次序都属于「行为可观察」的部分。 +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import { setup } from './helpers.mjs'; + +const AUTH = { Authorization: 'Bearer user_fp' }; +const CHAT = { model: 'm', stream: true, messages: [{ role: 'user', content: 'hi' }] }; + +// CLI 侧 FINGERPRINT_IB 的 components 字段全集(对齐 command-code CLI 实现) +const CLI_FIELDS = ['arch', 'collectorVersion', 'cpuCount', 'cpuModel', 'gitEmailHash', + 'hostnameHash', 'isContainer', 'macHashes', 'machineIdHash', 'memGiB', 'osRelease', + 'osUserHash', 'platform', 'runtime', 'timezone']; + +async function firstRequest(s) { + const r = await s.proxy.post('/v1/chat/completions', CHAT, AUTH); + await r.text(); + return s.mock.seen; +} + +test('初始化会发出 fingerprint/record 与 lifecycle-events 两条预请求', async () => { + const s = await setup(); + try { + const seen = await firstRequest(s); + const paths = seen.map(x => x.url); + assert.ok(paths.includes('/alpha/fingerprint/record'), '应发出 fingerprint/record'); + assert.ok(paths.includes('/alpha/lifecycle-events'), '应发出 lifecycle-events'); + assert.ok(paths.includes('/alpha/generate'), '应发出 generate'); + } finally { await s.close(); } +}); + +test('lifecycle 事件类型为 cli_session_exists', async () => { + const s = await setup(); + try { + await firstRequest(s); + const lc = s.mock.seen.find(x => x.url === '/alpha/lifecycle-events'); + const body = JSON.parse(lc.raw); + assert.equal(body.eventType, 'cli_session_exists'); + } finally { await s.close(); } +}); + +test('指纹 components 字段集合与 CLI 完全一致(不多不少)', async () => { + const s = await setup(); + try { + await firstRequest(s); + const fp = s.mock.seen.find(x => x.url === '/alpha/fingerprint/record'); + const body = JSON.parse(fp.raw); + const actual = Object.keys(body.components).sort(); + assert.deepEqual(actual, [...CLI_FIELDS].sort(), + '字段集合偏离 CLI 是实现指纹的典型破绽'); + } finally { await s.close(); } +}); + +test('指纹 runtime=cli、collectorVersion=1,thumbmark/machineIdHash 为 64 位 hex', async () => { + const s = await setup(); + try { + await firstRequest(s); + const body = JSON.parse(s.mock.seen.find(x => x.url === '/alpha/fingerprint/record').raw); + assert.equal(body.components.runtime, 'cli', 'runtime 必须自称 cli'); + assert.equal(body.components.collectorVersion, 1); + assert.match(body.thumbmark, /^[0-9a-f]{64}$/, 'thumbmark 应为 sha256 hex'); + assert.match(body.components.machineIdHash, /^[0-9a-f]{64}$/); + assert.match(body.components.hostnameHash, /^[0-9a-f]{64}$/); + assert.match(body.components.osUserHash, /^[0-9a-f]{64}$/); + } finally { await s.close(); } +}); + +test('macHashes 为 2~5 个 hex 串(CLI 的取值区间)', async () => { + const s = await setup(); + try { + await firstRequest(s); + const body = JSON.parse(s.mock.seen.find(x => x.url === '/alpha/fingerprint/record').raw); + const macs = body.components.macHashes; + assert.ok(Array.isArray(macs)); + assert.ok(macs.length >= 2 && macs.length <= 5, 'macHashes 数量应在 2~5,实际 ' + macs.length); + for (const m of macs) assert.match(m, /^[0-9a-f]+$/); + } finally { await s.close(); } +}); + +test('每种指纹只上报一次(进程内去重)', async () => { + const s = await setup(); + try { + await firstRequest(s); + // 第二次请求不应再触发预请求 + const r2 = await s.proxy.post('/v1/chat/completions', CHAT, AUTH); + await r2.text(); + const fpCount = s.mock.seen.filter(x => x.url === '/alpha/fingerprint/record').length; + assert.equal(fpCount, 1, '同一进程内指纹只应上报一次,实际 ' + fpCount); + } finally { await s.close(); } +}); + +test('不同 API key 各自初始化(互不复用指纹状态)', async () => { + const s = await setup(); + try { + const r1 = await s.proxy.post('/v1/chat/completions', CHAT, { Authorization: 'Bearer user_a' }); + await r1.text(); + const r2 = await s.proxy.post('/v1/chat/completions', CHAT, { Authorization: 'Bearer user_b' }); + await r2.text(); + const fpCount = s.mock.seen.filter(x => x.url === '/alpha/fingerprint/record').length; + assert.equal(fpCount, 2, '每个 key 应各自初始化一次,实际 ' + fpCount); + } finally { await s.close(); } +}); + +// components.cpuCount 是明文上传的,且可与同样明文的 cpuModel 交叉核对。 +// CLI 的 gatherRawSignals 取 os.cpus().length —— 逻辑处理器数,不是物理核心数。 +// 这张表是对「CLI 语义」的锁定:改了型号/核数就得同步改这里的期望值。 +const EXPECTED_THREADS = { + '12th Gen Intel(R) Core(TM) i7-12650H': 16, + '12th Gen Intel(R) Core(TM) i5-12400F': 12, + '12th Gen Intel(R) Core(TM) i9-12900K': 24, + '13th Gen Intel(R) Core(TM) i7-13700K': 24, + '13th Gen Intel(R) Core(TM) i5-13600K': 20, + '13th Gen Intel(R) Core(TM) i9-13900K': 32, + 'Intel(R) Core(TM) Ultra 7 155H': 22, + 'Intel(R) Core(TM) Ultra 9 285H': 16, // Arrow Lake 取消超线程,核数 == 线程数 + 'Intel(R) Core(TM) i9-14900K': 32, + 'Intel(R) Core(TM) i7-14700K': 28, + 'AMD Ryzen 7 7800X3D': 16, + 'AMD Ryzen 9 7950X': 32, + 'AMD Ryzen 5 7600': 12, + 'AMD Ryzen 9 7900X': 24, + 'AMD Ryzen 7 5800X3D': 16, +}; + +test('cpuCount 是逻辑处理器数(os.cpus().length),且与 cpuModel 自洽', async () => { + const s = await setup(); + try { + const keys = ['user_a', 'user_b', 'user_c', 'user_d', 'user_e', 'user_f', 'user_g', 'user_h']; + for (const k of keys) { + const r = await s.proxy.post('/v1/chat/completions', CHAT, { Authorization: 'Bearer ' + k }); + await r.text(); + } + const fps = s.mock.seen.filter(x => x.url === '/alpha/fingerprint/record') + .map(x => JSON.parse(x.raw).components); + assert.ok(fps.length >= 6, '应覆盖到足够多的指纹样本'); + for (const c of fps) { + const expected = EXPECTED_THREADS[c.cpuModel]; + assert.ok(expected !== undefined, 'cpuModel 必须在已知表内:' + c.cpuModel); + assert.equal(c.cpuCount, expected, + `${c.cpuModel} 的 cpuCount 应为逻辑处理器数 ${expected}(CLI 取 os.cpus().length),` + + '填物理核数等于宣称这台机器关了超线程 —— 而 cpuModel 与 cpuCount 都是明文,可被交叉核对'); + } + } finally { await s.close(); } +}); + diff --git a/test/fork.test.mjs b/test/fork.test.mjs new file mode 100644 index 0000000..e09614a --- /dev/null +++ b/test/fork.test.mjs @@ -0,0 +1,246 @@ +// fork 专属行为的回归护栏。这些改动不在上游,若被上游同步覆盖会静默丢失。 +// 每条都对应一个明确的行为契约,不是实现细节。 +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import http from 'node:http'; +import net from 'node:net'; +import { setup, startProxy, startMockUpstream, allocPort } from './helpers.mjs'; + +const AUTH = { Authorization: 'Bearer user_test' }; +const UUID = 'aaaaaaaa-bbbb-4ccc-8ddd-eeeeeeeeeeee'; +const CHAT = { model: 'm', stream: true, messages: [{ role: 'user', content: 'hi' }] }; + +// ── 逆向对齐:上游 User-Agent 与 threadId ────────────────── +// CLI 侧 vy = "cli"(源码常量),threadId 与 x-session-id 同值 +// (toWireThreadId 对合法 UUID 直接透传)。 +test('fork: 上游请求 User-Agent 为 cli(对齐官方 CLI 常量)', async () => { + const s = await setup(); + try { + const r = await s.proxy.post('/v1/chat/completions', CHAT, { ...AUTH, 'x-session-id': UUID }); + await r.text(); + const g = s.mock.lastGenerate(); + assert.equal(g.headers['user-agent'], 'cli', + '官方 CLI 发送 User-Agent: cli;发送 Node 默认 UA 是明显破绽'); + } finally { await s.close(); } +}); + +test('fork: 预请求(fingerprint/lifecycle)同样带 User-Agent: cli', async () => { + const s = await setup(); + try { + const r = await s.proxy.post('/v1/chat/completions', CHAT, { ...AUTH, 'x-session-id': UUID }); + await r.text(); + for (const path of ['/alpha/fingerprint/record', '/alpha/lifecycle-events']) { + const req = s.mock.seen.find(x => x.url === path); + assert.ok(req, path + ' 应被发出'); + assert.equal(req.headers['user-agent'], 'cli', path + ' 的 UA 也必须是 cli'); + } + } finally { await s.close(); } +}); + +test('fork: threadId 与 x-session-id 同值', async () => { + const s = await setup(); + try { + const r = await s.proxy.post('/v1/chat/completions', CHAT, { ...AUTH, 'x-session-id': UUID }); + await r.text(); + const g = s.mock.lastGenerate(); + assert.equal(g.headers['x-session-id'], UUID, 'session 头应透传'); + assert.equal(g.body.threadId, UUID, + 'sessionId 为合法 UUID 时,threadId 必须与之同值(CLI 的 toWireThreadId 行为)'); + } finally { await s.close(); } +}); + +test('fork: session 头非 UUID 时省略 threadId,但 header 原样透传', async () => { + const s = await setup(); + try { + // 两个契约不同: + // x-session-id header —— getSessionId 接受任意 >=8 字符,原样发出 + // body.threadId —— 仅当 sessionId 是合法 UUID 时才写(isWireUuid), + // 否则省略该字段,对齐 CLI 的 toWireThreadId + const r = await s.proxy.post('/v1/chat/completions', CHAT, + { ...AUTH, 'x-session-id': 'not-a-uuid-but-long-enough' }); + await r.text(); + const g = s.mock.lastGenerate(); + assert.equal(g.headers['x-session-id'], 'not-a-uuid-but-long-enough', + 'session 头应原样透传(不限于 UUID)'); + assert.equal(g.body.threadId, undefined, + '非 UUID 时 threadId 必须省略 —— 填非法值会让上游的 UUID 校验失败'); + } finally { await s.close(); } +}); + +test('fork: 无 session 头时回落 per-key session,threadId 仍与之同值', async () => { + const s = await setup(); + try { + const r = await s.proxy.post('/v1/chat/completions', CHAT, AUTH); // 不带 session 头 + await r.text(); + const g = s.mock.lastGenerate(); + const sid = g.headers['x-session-id']; + assert.ok(sid, '应回落到 per-key session'); + assert.equal(g.body.threadId, sid, '回落路径下两者同样必须同值'); + } finally { await s.close(); } +}); + +// ── issue #18:上游 HTTP(S) 代理(零依赖 CONNECT 隧道)── +// Bun 的 node:http 基于 fetch 实现:CONNECT 方法发起即报 "fetch() URL is invalid", +// 且 http.request 的 createConnection 会被忽略(自建连接直连),隧道无法落地。 +// 所以两个「必须走隧道」的用例只在 Node 跑;「不走代理」的负向断言两个运行器都成立。 +const IS_BUN = !!process.versions.bun; +const notOnBun = IS_BUN ? test.skip : test; + +/** 录制型 CONNECT 代理:记录每次 CONNECT 的 target,并做裸字节转发。 */ +async function startRecordingProxy() { + const port = await allocPort(); + const connects = []; + const server = http.createServer((req, res) => { res.writeHead(405); res.end(); }); + server.on('connect', (req, clientSocket, head) => { + connects.push(req.url); + const [h, p] = req.url.split(':'); + const up = net.connect(Number(p || 443), h, () => { + clientSocket.write('HTTP/1.1 200 Connection Established\r\n\r\n'); + if (head?.length) up.write(head); + up.pipe(clientSocket); clientSocket.pipe(up); + }); + up.on('error', () => clientSocket.destroy()); + clientSocket.on('error', () => up.destroy()); + }); + await new Promise(r => server.listen(port, '127.0.0.1', r)); + return { port, connects, close: () => new Promise(r => server.close(r)) }; +} + +notOnBun('#18: 配置 CC_UPSTREAM_PROXY 后上游请求经 CONNECT 隧道', async () => { + const rec = await startRecordingProxy(); + const mock = await startMockUpstream(); + const proxy = await startProxy({ upstreamPort: mock.port, + env: { CC_UPSTREAM_PROXY: 'http://127.0.0.1:' + rec.port } }); + try { + const r = await proxy.post('/v1/chat/completions', CHAT, AUTH); + await r.text(); + assert.ok(rec.connects.length >= 1, '应建立 CONNECT 隧道,实际 ' + JSON.stringify(rec.connects)); + // generate + fingerprint + lifecycle 三条都要走代理 + assert.equal(mock.generateCount(), 1, 'generate 应经隧道到达上游'); + assert.equal(mock.lastGenerate().headers['user-agent'], 'cli', '经代理时 UA 仍为 cli'); + } finally { + await proxy.kill(); await mock.close(); await rec.close(); + } +}); + +notOnBun('#18: 预请求也走代理(避免同一账号从两个 IP 注册)', async () => { + const rec = await startRecordingProxy(); + const mock = await startMockUpstream(); + const proxy = await startProxy({ upstreamPort: mock.port, + env: { CC_UPSTREAM_PROXY: 'http://127.0.0.1:' + rec.port } }); + try { + const r = await proxy.post('/v1/chat/completions', CHAT, AUTH); + await r.text(); + // 三条预请求 + generate 共 3 次 CONNECT(generate/fingerprint/lifecycle) + assert.ok(rec.connects.length >= 3, + 'fingerprint 与 lifecycle 也必须走代理,否则账号会从两个 IP 注册。实际 CONNECT 数: ' + rec.connects.length); + } finally { + await proxy.kill(); await mock.close(); await rec.close(); + } +}); + +test('#18: 未配置代理时不建立任何 CONNECT', async () => { + const rec = await startRecordingProxy(); + const mock = await startMockUpstream(); + const proxy = await startProxy({ upstreamPort: mock.port }); // 不设 CC_UPSTREAM_PROXY + try { + const r = await proxy.post('/v1/chat/completions', CHAT, AUTH); + await r.text(); + assert.equal(rec.connects.length, 0, '默认应直连,不经任何代理'); + } finally { + await proxy.kill(); await mock.close(); await rec.close(); + } +}); + +test('#18: 探活端点不经过上游代理', async () => { + const rec = await startRecordingProxy(); + const mock = await startMockUpstream(); + const proxy = await startProxy({ upstreamPort: mock.port, + env: { CC_UPSTREAM_PROXY: 'http://127.0.0.1:' + rec.port } }); + try { + const before = rec.connects.length; + const r = await proxy.get('/health'); + assert.equal(r.status, 200); + assert.equal(rec.connects.length, before, '/health 是本地端点,不应触发上游 CONNECT'); + } finally { + await proxy.kill(); await mock.close(); await rec.close(); + } +}); + +// ── 设备指纹派生 ───────────────────────────────────────── +test('fork: 同一 API key 在同一盐下得到稳定 thumbmark(重启后不变)', async () => { + const mock = await startMockUpstream(); + const env = { CC_FINGERPRINT_SALT: 'ci-salt' }; + const p1 = await startProxy({ upstreamPort: mock.port, env }); + let first; + try { + const r = await p1.post('/v1/chat/completions', CHAT, AUTH); + await r.text(); + first = JSON.parse(mock.seen.find(x => x.url === '/alpha/fingerprint/record').raw); + } finally { await p1.kill(); } + + const p2 = await startProxy({ upstreamPort: mock.port, env }); + try { + const r = await p2.post('/v1/chat/completions', CHAT, AUTH); + await r.text(); + const fps = mock.seen.filter(x => x.url === '/alpha/fingerprint/record'); + const second = JSON.parse(fps[fps.length - 1].raw); + assert.equal(second.thumbmark, first.thumbmark, + 'derived 模式下同一 key 重启后必须得到同一设备(否则每次重启都像换了台机器)'); + } finally { await p2.kill(); await mock.close(); } +}); + +test('fork: 不同 API key 得到不同 thumbmark', async () => { + const mock = await startMockUpstream(); + const proxy = await startProxy({ upstreamPort: mock.port, env: { CC_FINGERPRINT_SALT: 'ci-salt' } }); + try { + for (const k of ['user_a', 'user_b', 'user_c']) { + const r = await proxy.post('/v1/chat/completions', CHAT, { Authorization: 'Bearer ' + k }); + await r.text(); + } + const tms = mock.seen.filter(x => x.url === '/alpha/fingerprint/record') + .map(x => JSON.parse(x.raw).thumbmark); + assert.equal(tms.length, 3, '三个 key 应各上报一次'); + assert.equal(new Set(tms).size, 3, '不同 key 必须映射到不同设备指纹(不能退化成哈希桶)'); + } finally { await proxy.kill(); await mock.close(); } +}); + +test('fork: 不同盐得到不同部署指纹', async () => { + const mock = await startMockUpstream(); + const grab = async (salt) => { + const p = await startProxy({ upstreamPort: mock.port, env: { CC_FINGERPRINT_SALT: salt } }); + try { + const r = await p.post('/v1/chat/completions', CHAT, AUTH); + await r.text(); + const fps = mock.seen.filter(x => x.url === '/alpha/fingerprint/record'); + return JSON.parse(fps[fps.length - 1].raw).thumbmark; + } finally { await p.kill(); } + }; + try { + const a = await grab('salt-one'); + const b = await grab('salt-two'); + assert.notEqual(a, b, '不同部署(不同盐)不应共享设备指纹'); + } finally { await mock.close(); } +}); + +test('fork: 伪造信号值的形状与真实机器一致(MachineGuid/MAC/主机名)', async () => { + const mock = await startMockUpstream(); + const proxy = await startProxy({ upstreamPort: mock.port, env: { CC_FINGERPRINT_SALT: 'ci-salt' } }); + try { + // 原始信号值不上网,只能从派生源反推:这里直接校验上网的哈希之间有正确的数量关系。 + // 真正要锁的是「形状」这一层 —— 出网载荷里没有明文,所以断言落在组件集合上。 + const r = await proxy.post('/v1/chat/completions', CHAT, AUTH); + await r.text(); + const fp = JSON.parse(mock.seen.find(x => x.url === '/alpha/fingerprint/record').raw); + const c = fp.components; + assert.match(fp.thumbmark, /^[0-9a-f]{64}$/, 'thumbmark 必须是 64 位 hex(sha256)'); + assert.match(c.machineIdHash, /^[0-9a-f]{64}$/, 'machineIdHash 必须是 sha256'); + assert.ok(Array.isArray(c.macHashes) && c.macHashes.length >= 2 && c.macHashes.length <= 5, + 'macHashes 数量应落在 FINGERPRINT_MAC_COUNT_RANGE 内'); + assert.equal(new Set(c.macHashes).size, c.macHashes.length, 'MAC 必须已去重'); + assert.match(c.hostnameHash, /^[0-9a-f]{64}$/); + assert.match(c.gitEmailHash, /^[0-9a-f]{64}$/); + assert.equal(c.collectorVersion, 1); + assert.equal(c.runtime, 'cli', 'CLI 的 runtime 恒为 cli'); + } finally { await proxy.kill(); await mock.close(); } +}); diff --git a/test/helpers.mjs b/test/helpers.mjs index 3342bb8..5feb59b 100644 --- a/test/helpers.mjs +++ b/test/helpers.mjs @@ -63,7 +63,10 @@ export async function startMockUpstream(opts = {}) { req.on('end', async () => { const raw = Buffer.concat(chunks).toString('utf8'); seen.push({ url: req.url, method: req.method, headers: req.headers, raw }); - if (opts.onRequest) await opts.onRequest(req, res, seen[seen.length - 1]); + // onRequest 返回 true 表示它自己接管了响应(可以只写一半就挂住, + // 用来模拟"上游有数据但长时间没有新数据",触发代理的空闲超时)。 + const handled = opts.onRequest ? await opts.onRequest(req, res, seen[seen.length - 1]) : false; + if (handled === true) return; if (res.writableEnded) return; const status = opts.status ?? 200; if (status !== 200) { diff --git a/test/regressions.test.mjs b/test/regressions.test.mjs new file mode 100644 index 0000000..657db63 --- /dev/null +++ b/test/regressions.test.mjs @@ -0,0 +1,70 @@ +// 回归护栏:为已修复的 issue 各留一条断言,防止再次退化。 +// 每条都注明来源 issue —— 删掉某条前请先读对应的 issue。 +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import { setup } from './helpers.mjs'; + +const AUTH = { Authorization: 'Bearer user_test' }; + +async function wire(s, path, body, headers = AUTH) { + const r = await s.proxy.post(path, body, headers); + await r.text(); + const g = s.mock.lastGenerate(); + return { status: r.status, params: g ? g.body.params : null }; +} + +// issue #17:无 system prompt 时发空格占位,阻止 CC 上游注入 ~7.5K token 默认提示词 +// 形态是 CLI 的 toWireSystem 块数组(见 wire.test.mjs 的说明),占位因此是单个空格文本块。 +test('#17 chat:无 system prompt 时 params.system 为占位块而非空/缺省', async () => { + const s = await setup(); + try { + const { params } = await wire(s, '/v1/chat/completions', + { model: 'm', stream: true, messages: [{ role: 'user', content: 'hi' }] }); + assert.deepEqual(params.system, [{ type: 'text', text: ' ' }], + '缺省会触发上游注入默认提示词(#17),占位必须是单个空格文本块'); + } finally { await s.close(); } +}); + +// issue #7:超限必须走 413 + 排空,而不是直接 reset(客户端会看到 Connection error) +test('#7 超限请求返回 413 且连接可继续使用(不 reset)', async () => { + const s = await setup({ env: { CC_MAX_BODY_MB: '1' } }); + try { + const big = JSON.stringify({ model: 'm', stream: true, + messages: [{ role: 'user', content: 'x'.repeat(2 * 1024 * 1024) }] }); + const r = await s.proxy.post('/v1/chat/completions', big, AUTH); + assert.equal(r.status, 413, '必须是 HTTP 413 响应,而不是连接被 reset'); + const j = await r.json(); + assert.ok(j.error, '应为结构化错误体,便于客户端识别'); + } finally { await s.close(); } +}); + +// issue #25:Anthropic input_tokens 只计非缓存部分(与 cache_read 相加 = 总输入) +test('#25 messages:Anthropic usage 的 input_tokens 不含缓存部分', async () => { + const s = await setup({ ndjson: [ + '{"type":"text-start"}', + '{"type":"text-delta","text":"hi"}', + '{"type":"text-end"}', + // inputTokens 是总数(含缓存),cacheRead 是其子集 + '{"type":"finish-step","finishReason":"stop","usage":{"inputTokens":1000,"outputTokens":10,"cachedInputTokens":800,"inputTokenDetails":{"noCacheTokens":200,"cacheReadTokens":800,"cacheWriteTokens":0}}}', + ] }); + try { + const r = await s.proxy.post('/v1/messages', + { model: 'm', max_tokens: 100, stream: true, messages: [{ role: 'user', content: 'hi' }] }, + { 'x-api-key': 'user_test' }); + const text = await r.text(); + // SSE 事件格式:event: \ndata: \n\n —— 按行取,不能用非贪婪 \{.*?\}(会在首个 } 截断) + const deltas = text.split('\n\n') + .map(block => { + const ev = /^event: (\S+)/m.exec(block)?.[1]; + const data = /^data: (.*)$/m.exec(block)?.[1]; + if (ev !== 'message_delta' || !data) return null; + try { return JSON.parse(data); } catch { return null; } + }) + .filter(Boolean); + const usage = deltas.find(d => d.usage)?.usage; + assert.ok(usage, 'message_delta 应携带 usage'); + assert.equal(usage.input_tokens, 200, + 'input_tokens 只能是非缓存部分(200),不能是总数(1000)—— 否则下游相加会约两倍'); + assert.equal(usage.cache_read_input_tokens, 800); + } finally { await s.close(); } +}); diff --git a/test/responses-zero-output.test.mjs b/test/responses-zero-output.test.mjs new file mode 100644 index 0000000..2327db6 --- /dev/null +++ b/test/responses-zero-output.test.mjs @@ -0,0 +1,59 @@ +// issue #56:#54(2eccdbf)把流式 /v1/responses 改成「上游一 200 就先发 response.created」, +// translator.started(= createdSent)自此恒为 true,零输出防护 +// translator.outputTokens === 0 && !translator.started +// 成了死代码 —— 空响应会经 finish() 包装成 response.completed 谎报成功 +//(旧版 cce214d 是 HTTP 429 + rate_limit_error)。 +// 修复口径:判据换成「是否真的产出过 output item」(hasOutput); +// 响应头已按 200 提交后状态码改不回 429(再调 sendResponsesError 会抛 +// ERR_HTTP_HEADERS_SENT),按本文件既有失败口径走 response.failed。 +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import { setup } from './helpers.mjs'; + +const AUTH = { Authorization: 'Bearer user_test' }; + +// 与 issue 里复现用的同一组上游输出:start + finish(outputTokens=0),中间零内容 +const EMPTY_STREAM = [ + '{"type":"start"}', + '{"type":"finish","finishReason":"stop","totalUsage":{"inputTokens":5,"outputTokens":0,"cachedInputTokens":0}}', +]; + +test('#56 流式:空响应必须报 response.failed,不能再谎报 response.completed', async () => { + const s = await setup({ ndjson: EMPTY_STREAM }); + try { + const r = await s.proxy.post('/v1/responses', { model: 'm', stream: true, input: 'hi' }, AUTH); + assert.equal(r.status, 200, 'created 已先行发出,HTTP 状态只能停在 200'); + const text = await r.text(); // 能正常读完 = 流有收尾;漏 res.end() 这里会挂起 + assert.ok(text.includes('event: response.failed'), '必须显式发 response.failed'); + assert.ok(text.includes('"status":"failed"'), 'response.status 必须是 failed'); + assert.ok(text.includes('"code":"upstream_error"'), '错误码按本文件既有失败口径取 upstream_error'); + assert.ok(text.includes('Empty response from upstream (zero output tokens)'), '错误消息要说明是空响应'); + assert.ok(!text.includes('response.completed'), + '零输出绝不能发 response.completed —— 那是把空响应谎报成功(#38/#39 同类问题)'); + assert.ok(!text.includes('Cannot write headers after they are sent'), + '响应头已提交后不能再走 sendResponsesError(issue 里实测的 ERR_HTTP_HEADERS_SENT 惨案)'); + } finally { await s.close(); } +}); + +test('#56 流式:有真实输出时行为不变(response.completed 正常发出)', async () => { + const s = await setup(); // 默认 ndjson:hello + outputTokens 3 + try { + const r = await s.proxy.post('/v1/responses', { model: 'm', stream: true, input: 'hi' }, AUTH); + const text = await r.text(); + assert.equal(r.status, 200); + assert.ok(text.includes('event: response.completed'), '正常输出必须照常 completed'); + assert.ok(text.includes('"status":"completed"')); + assert.ok(!text.includes('response.failed'), '不能误伤正常响应'); + } finally { await s.close(); } +}); + +test('#56 非流式:空响应仍是 HTTP 429(该路径守卫未被 #54 波及)', async () => { + const s = await setup({ ndjson: EMPTY_STREAM }); + try { + const r = await s.proxy.post('/v1/responses', { model: 'm', input: 'hi' }, AUTH); + assert.equal(r.status, 429, '非流式没有提前发响应头,旧语义(429)应原样保留'); + const json = await r.json(); + assert.equal(json.error.type, 'rate_limit_error'); + assert.equal(json.error.message, 'Empty response from upstream (zero output tokens)'); + } finally { await s.close(); } +}); diff --git a/test/stream-end.test.mjs b/test/stream-end.test.mjs index 3999fb6..f721504 100644 --- a/test/stream-end.test.mjs +++ b/test/stream-end.test.mjs @@ -202,3 +202,119 @@ test('#38 回归:tool-calls 仍报 tool_use / tool_calls', async () => { assert.equal(a.json.stop_reason, 'tool_use'); } finally { await s.close(); } }); + +// 回归:#38 排查期间发现的日志噪音。上游每个响应都会发一串无内容事件 +// (text-start / text-end / start / start-step / reasoning-start / reasoning-end / +// provider-metadata / tool-input-* / tool-error)。三条非流式路径原先缺少静默列表, +// 全部掉进 default 打成 'Unknown CC event type',线上刷屏并把真正的错误淹掉。 +test('标准 NDJSON 序列不产生任何 Unknown CC event type 警告(三协议 × 流式/非流式)', async () => { + const s = await setup(); + try { + const chat = { model: 'm', messages: [{ role: 'user', content: 'hi' }] }; + const msg = { model: 'm', max_tokens: 50, messages: [{ role: 'user', content: 'hi' }] }; + await (await s.proxy.post('/v1/chat/completions', { ...chat, stream: true }, AUTH)).text(); + await (await s.proxy.post('/v1/chat/completions', chat, AUTH)).text(); + await (await s.proxy.post('/v1/messages', { ...msg, stream: true }, { 'x-api-key': 'user_test' })).text(); + await (await s.proxy.post('/v1/messages', msg, { 'x-api-key': 'user_test' })).text(); + await (await s.proxy.post('/v1/responses', { model: 'm', stream: true, input: 'hi' }, AUTH)).text(); + await (await s.proxy.post('/v1/responses', { model: 'm', input: 'hi' }, AUTH)).text(); + + const logs = s.proxy.logs(); + assert.ok(!logs.includes('Unknown CC event type'), + '不应出现 Unknown CC event type 警告,实际日志片段:\n' + + logs.split('\n').filter(l => l.includes('Unknown CC')).join('\n')); + } finally { await s.close(); } +}); + + +// 上游 error 事件自带 statusCode 时必须用它 —— CLI 的 readStreamErrorEvent 读的就是这个字段, +// 取值链是 parseEmbeddedErrorJSON(message)?.status ?? error.statusCode ?? null。 +// 原实现只看 message 里的 "" 前缀,statusCode 全被丢掉 → 429/503 塌成 502。 +test('#38 error 事件带 statusCode 时按其映射(429 而非 502)', async () => { + const s = await setup({ ndjson: [ + '{"type":"text-start"}', + '{"type":"text-delta","text":"partial"}', + '{"type":"error","error":{"message":"providers are currently at capacity","statusCode":429}}', + ] }); + try { + const r = await s.proxy.post('/v1/chat/completions', CHAT, AUTH); + const j = await r.json(); + assert.equal(r.status, 429, 'statusCode 是上游给的,不能抹成 502'); + assert.equal(j.error.type, 'rate_limit_error'); + assert.equal(j.retry_after, 30, '429 要带退避提示,否则客户端不知道等多久'); + } finally { await s.close(); } +}); + +test('#38 error 事件带 statusCode 时按其映射(503 而非 502)', async () => { + const s = await setup({ ndjson: [ + '{"type":"text-start"}', + '{"type":"error","error":{"message":"service unavailable","statusCode":503}}', + ] }); + try { + const r = await s.proxy.post('/v1/chat/completions', CHAT, AUTH); + assert.equal(r.status, 503); + } finally { await s.close(); } +}); + +test('#38 error 事件没有 statusCode 时仍回落 502(保持原行为)', async () => { + const s = await setup({ ndjson: [ + '{"type":"text-start"}', + '{"type":"error","error":{"message":"something broke"}}', + ] }); + try { + const r = await s.proxy.post('/v1/chat/completions', CHAT, AUTH); + assert.equal(r.status, 502); + } finally { await s.close(); } +}); + +test('#38 message 里的 "" 前缀优先于 statusCode(对齐 CLI 的取值链)', async () => { + const s = await setup({ ndjson: [ + '{"type":"text-start"}', + '{"type":"error","error":{"message":"<400> bad request","statusCode":503}}', + ] }); + try { + const r = await s.proxy.post('/v1/chat/completions', CHAT, AUTH); + assert.equal(r.status, 400, ' 前缀是最优先的取值来源'); + } finally { await s.close(); } +}); + + +// 流空闲超时后必须以 end() 收尾。原先走的是 res.write(err) 紧跟 res.destroy(): +// write 是异步的,destroy 会把未刷出的缓冲丢掉并发 RST,反向代理那里就是 +// "upstream prematurely closed connection" → 502,或者客户端看到 connection error。 +// 断言方式:客户端必须能**完整读到**已产生的 delta 与超时错误事件 —— destroy 会让 +// 这条读挂掉(ECONNRESET / 截断),end 则正常收束。 +test('流空闲超时:已产生的内容 + 错误事件都能完整送达(不能 destroy 客户端 socket)', async () => { + const s = await setup({ + env: { CC_STREAM_IDLE_MS: '300' }, + onRequest: (req, res) => { + res.writeHead(200, { 'Content-Type': 'text/event-stream' }); + res.write('{"type":"text-start"}\n'); + res.write('{"type":"text-delta","text":"partial-content"}\n'); + return true; // 接管后挂住:不再发任何数据 → 触发空闲超时 + }, + }); + try { + const r = await s.proxy.post('/v1/chat/completions', { ...CHAT, stream: true }, AUTH); + const text = await r.text(); + assert.equal(r.status, 200); + assert.ok(text.includes('partial-content'), '已发出的内容不能因为收尾方式而丢失'); + assert.ok(text.includes('rate_limit_error'), '超时错误事件必须完整送进流里'); + } finally { await s.close(); } +}); + + +// 反代场景的 keep-alive 时序:Node 的 keepAliveTimeout 必须**大于**反代的 +// upstream keepalive_timeout。否则反代会复用后端已关闭的连接,写请求体时吃 EPIPE, +// 而 POST 是非幂等、nginx 默认不重试 → 客户端直接 502。 +test('启动时显式设置 keepAliveTimeout 并打出(反代 keepalive_timeout 必须小于它)', async () => { + const s = await setup(); + try { + const logs = s.proxy.logs(); + assert.ok(/keepAliveTimeout[":\s]+65000ms/.test(logs), + '启动横幅必须打出 keepAliveTimeout,便于和反代配置对齐。实际:\n' + + logs.split('\n').filter(l => l.includes('CC Proxy started')).join('\n')); + assert.ok(logs.includes('keepalive_timeout'), '横幅里要提示反代侧的对应设置'); + } finally { await s.close(); } +}); + diff --git a/test/wire.test.mjs b/test/wire.test.mjs new file mode 100644 index 0000000..384f07f --- /dev/null +++ b/test/wire.test.mjs @@ -0,0 +1,243 @@ +// wire 协议契约:断言发往 CC 上游 /alpha/generate 的请求体形状。 +// 这类断言是挡住「静默丢消息 / 静默改语义」回归的关键 —— 只看 HTTP 状态码看不出来。 +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import { setup } from './helpers.mjs'; + +const AUTH = { Authorization: 'Bearer user_test' }; + +/** 取本次请求发往上游的 params.messages */ +async function wireMessages(proxy, mock, path, body, headers = AUTH) { + const r = await proxy.post(path, body, headers); + await r.text(); + const g = mock.lastGenerate(); + assert.ok(g, '应至少产生一条 /alpha/generate(实际: ' + mock.seen.map(s => s.url).join(',') + ')'); + return { status: r.status, params: g.body.params, config: g.body.config, headers: g.headers }; +} + +// params.system 的形态是 **块数组**,不是字符串 —— 对齐 CLI 的 toWireSystem: +// toWireSystem(e) { const t = e.length - 1; +// return e.map((e, n) => ({ type: 'text', text: n < t ? e.text + '\n' : e.text, +// ...(e.cache ? { cache_control: { type: 'ephemeral' } } : {}) })); } +// 早先「数组会被上游拒绝」的判断源自一次误诊(真因是 content 为字符串时整条 user 消息 +// 丢失,见 convertAnthropicToOpenAI 的注释),已更正。 +test('chat:system 提升到 params.system,形态为 CLI 的块数组', async () => { + const s = await setup(); + try { + const { params } = await wireMessages(s.proxy, s.mock, '/v1/chat/completions', { + model: 'deepseek/deepseek-v4-flash', stream: true, + messages: [{ role: 'system', content: 'you are terse' }, { role: 'user', content: 'hi' }], + }); + assert.deepEqual(params.system, [{ type: 'text', text: 'you are terse' }], + '单个 system 段应序列化为一个文本块(末块不加 \\n)'); + assert.equal(params.messages.length, 1, 'system 不应留在 messages 里'); + assert.equal(params.messages[0].role, 'user'); + } finally { await s.close(); } +}); + +test('chat:多个 system 段时非末块补 \\n(对齐 toWireSystem)', async () => { + const s = await setup(); + try { + const { params } = await wireMessages(s.proxy, s.mock, '/v1/chat/completions', { + model: 'm', stream: true, + messages: [ + { role: 'system', content: 'first' }, + { role: 'system', content: 'second' }, + { role: 'user', content: 'hi' }, + ], + }); + assert.deepEqual(params.system, [ + { type: 'text', text: 'first\n' }, + { type: 'text', text: 'second' }, + ]); + } finally { await s.close(); } +}); + +test('chat:system 块上的 cache_control 原样下发(CLI 的 systemSections[].cache)', async () => { + const s = await setup(); + try { + const { params } = await wireMessages(s.proxy, s.mock, '/v1/chat/completions', { + model: 'm', stream: true, + messages: [ + { role: 'system', content: [{ type: 'text', text: 'cached prefix', cache_control: { type: 'ephemeral' } }] }, + { role: 'user', content: 'hi' }, + ], + }); + assert.deepEqual(params.system, [ + { type: 'text', text: 'cached prefix', cache_control: { type: 'ephemeral' } }, + ]); + } finally { await s.close(); } +}); + +test('chat:user 内容包成 [{type:text}] 结构', async () => { + const s = await setup(); + try { + const { params } = await wireMessages(s.proxy, s.mock, '/v1/chat/completions', { + model: 'm', stream: true, messages: [{ role: 'user', content: 'hello' }], + }); + assert.deepEqual(params.messages[0], { role: 'user', content: [{ type: 'text', text: 'hello' }] }); + } finally { await s.close(); } +}); + +test('chat:assistant 历史按 [reasoning, text, tool-call] 次序回传', async () => { + const s = await setup(); + try { + const { params } = await wireMessages(s.proxy, s.mock, '/v1/chat/completions', { + model: 'm', stream: true, messages: [ + { role: 'user', content: 'q' }, + { role: 'assistant', reasoning_content: 'thinking', content: 'answer', + tool_calls: [{ id: 'c1', type: 'function', function: { name: 'f', arguments: '{"a":1}' } }] }, + { role: 'tool', tool_call_id: 'c1', content: 'result' }, + ], + }); + const asst = params.messages.find(m => m.role === 'assistant'); + assert.deepEqual(asst.content.map(p => p.type), ['reasoning', 'text', 'tool-call'], + 'CC 校验历史中的 reasoning,且次序必须与 CLI 一致'); + assert.equal(asst.content[0].text, 'thinking'); + } finally { await s.close(); } +}); + +test('chat:多模态 image_url 转成 CC image 结构', async () => { + const s = await setup(); + try { + const { params } = await wireMessages(s.proxy, s.mock, '/v1/chat/completions', { + model: 'm', stream: true, messages: [{ role: 'user', content: [ + { type: 'text', text: 'look' }, + { type: 'image_url', image_url: { url: 'data:image/png;base64,AAA' } }, + ] }], + }); + const parts = params.messages[0].content; + assert.equal(parts.find(p => p.type === 'image').image, 'data:image/png;base64,AAA'); + } finally { await s.close(); } +}); + +test('messages:Anthropic thinking block 回传为 reasoning_content', async () => { + const s = await setup(); + try { + const { params } = await wireMessages(s.proxy, s.mock, '/v1/messages', { + model: 'm', max_tokens: 100, stream: true, messages: [ + { role: 'user', content: 'q' }, + { role: 'assistant', content: [ + { type: 'thinking', thinking: 'pondering' }, + { type: 'text', text: 'answer' }, + ] }, + { role: 'user', content: 'again' }, + ], + }, { 'x-api-key': 'user_test' }); + const asst = params.messages.find(m => m.role === 'assistant'); + assert.deepEqual(asst.content.map(p => p.type), ['reasoning', 'text'], + 'Anthropic 的 thinking 必须转成 reasoning 回传,否则 CC 拒绝多轮'); + assert.equal(asst.content[0].text, 'pondering'); + } finally { await s.close(); } +}); + +test('responses:带 type 的 input item 正常转换', async () => { + const s = await setup(); + try { + const { params } = await wireMessages(s.proxy, s.mock, '/v1/responses', { + model: 'm', stream: true, + input: [{ type: 'message', role: 'user', content: 'hello' }], + }); + assert.deepEqual(params.messages[0], { role: 'user', content: [{ type: 'text', text: 'hello' }] }); + } finally { await s.close(); } +}); + +test('responses:input 为字符串时等价于单条 user 消息', async () => { + const s = await setup(); + try { + const { params } = await wireMessages(s.proxy, s.mock, '/v1/responses', { + model: 'm', stream: true, input: 'hello', + }); + assert.deepEqual(params.messages[0], { role: 'user', content: [{ type: 'text', text: 'hello' }] }); + } finally { await s.close(); } +}); + +test('responses:function_call_output 映射成 tool 消息', async () => { + const s = await setup(); + try { + const { params } = await wireMessages(s.proxy, s.mock, '/v1/responses', { + model: 'm', stream: true, input: [ + { type: 'message', role: 'user', content: 'q' }, + { type: 'function_call', call_id: 'c1', name: 'f', arguments: '{}' }, + { type: 'function_call_output', call_id: 'c1', output: 'out' }, + ], + }); + assert.ok(params.messages.some(m => m.role === 'tool'), 'function_call_output 应产出 tool 消息'); + } finally { await s.close(); } +}); + +// ── 工具名完全不重命名(issue #36 / #37) ────────────── +// wire 协议里没有工具重命名这回事。CLI 的 toWireToolName 只服务于「重放自家退役 +// 工具名的旧会话」(tool_search 在 CLI 里 visible:()=>false,从不进声明);反代没有 +// catalog、没有退役名,因此声明与消息都必须原样透传 —— 只要有一处改名,下游就会按 +// 自己声明的名字派发不到工具。 +test('tools 声明不做名字重写(CLI 的 toWireTools 是原样 map)', async () => { + const s = await setup(); + try { + const { params } = await wireMessages(s.proxy, s.mock, '/v1/chat/completions', { + model: 'm', stream: true, messages: [{ role: 'user', content: 'q' }], + tools: ['bash_output', 'read_multiple_files', 'tool_search'].map(n => ({ + type: 'function', function: { name: n, description: '', parameters: { type: 'object', properties: {} } }, + })), + }); + assert.deepEqual(params.tools.map(t => t.name), ['bash_output', 'read_multiple_files', 'tool_search'], + '客户端声明的名字必须原样下发,否则客户端按自己的声明找不到工具'); + } finally { await s.close(); } +}); + +test('tool_search 在 tool-call 里也**不**被重写(不套用 CLI 自家的退役名归一化)', async () => { + const s = await setup(); + try { + const { params } = await wireMessages(s.proxy, s.mock, '/v1/chat/completions', { + model: 'm', stream: true, messages: [ + { role: 'user', content: 'q' }, + { role: 'assistant', content: null, + tool_calls: [{ id: 'c1', type: 'function', function: { name: 'tool_search', arguments: '{}' } }] }, + { role: 'tool', tool_call_id: 'c1', content: 'r' }, + ], + tools: [{ type: 'function', function: { name: 'tool_search', parameters: { type: 'object', properties: {} } } }], + }); + const call = params.messages.find(m => m.role === 'assistant').content.find(p => p.type === 'tool-call'); + assert.equal(call.toolName, 'tool_search', + '客户端声明的就是 tool_search,消息里必须还是它;改成 search_tools 下游就派发不到'); + assert.equal(params.tools[0].name, 'tool_search', '声明与消息必须同名'); + } finally { await s.close(); } +}); + +test('tool-result 的 toolName 与 tool-call 一致(CLI 用同一张 map)', async () => { + const s = await setup(); + try { + const { params } = await wireMessages(s.proxy, s.mock, '/v1/chat/completions', { + model: 'm', stream: true, messages: [ + { role: 'user', content: 'q' }, + { role: 'assistant', content: null, + tool_calls: [{ id: 'c1', type: 'function', function: { name: 'tool_search', arguments: '{}' } }] }, + { role: 'tool', tool_call_id: 'c1', name: 'tool_search', content: 'r' }, + ], + }); + const call = params.messages.find(m => m.role === 'assistant').content.find(p => p.type === 'tool-call'); + const res = params.messages.find(m => m.role === 'tool').content[0]; + assert.equal(res.toolName, call.toolName, + '调用名与结果名对不上会被上游判为无效的工具结果'); + assert.equal(res.toolName, 'tool_search'); + } finally { await s.close(); } +}); + +test('普通工具名在 tools 声明与 messages 里都不动', async () => { + const s = await setup(); + try { + const { params } = await wireMessages(s.proxy, s.mock, '/v1/chat/completions', { + model: 'm', stream: true, messages: [ + { role: 'user', content: 'q' }, + { role: 'assistant', content: null, + tool_calls: [{ id: 'c1', type: 'function', function: { name: 'get_weather', arguments: '{}' } }] }, + { role: 'tool', tool_call_id: 'c1', content: 'r' }, + ], + tools: [{ type: 'function', function: { name: 'get_weather', parameters: { type: 'object', properties: {} } } }], + }); + assert.equal(params.tools[0].name, 'get_weather'); + assert.equal(params.messages.find(m => m.role === 'assistant').content.find(p => p.type === 'tool-call').toolName, 'get_weather'); + assert.equal(params.messages.find(m => m.role === 'tool').content[0].toolName, 'get_weather'); + } finally { await s.close(); } +}); +