diff --git a/.agents/plugins/marketplace.json b/.agents/plugins/marketplace.json index 4823167..1346a9b 100644 --- a/.agents/plugins/marketplace.json +++ b/.agents/plugins/marketplace.json @@ -9,20 +9,20 @@ "source": { "source": "url", "url": "https://github.com/partme-ai/partme-codeguard-plugin.git", - "ref": "v0.14.13" + "ref": "v0.14.14" }, "policy": { "installation": "AVAILABLE", "authentication": "ON_USE" }, "category": "Developer Tools", - "version": "0.14.13", + "version": "0.14.14", "description": "Evidence-backed code checks and Git content gates for AI assistants, with Maven/Gradle module impact analysis. Save hooks provide feedback; unverified checks are explicit.", - "icon": "https://cdn.jsdelivr.net/gh/full-stack-plugins/codeguard-plugin@v0.14.13/assets/official-logo.png", + "icon": "https://cdn.jsdelivr.net/gh/full-stack-plugins/codeguard-plugin@v0.14.14/assets/official-logo.png", "interface": { "displayName": "代码规范守卫", "shortDescription": "Trustworthy code checks and Java impact analysis", - "logo": "https://cdn.jsdelivr.net/gh/full-stack-plugins/codeguard-plugin@v0.14.13/assets/official-logo.png" + "logo": "https://cdn.jsdelivr.net/gh/full-stack-plugins/codeguard-plugin@v0.14.14/assets/official-logo.png" } } ] diff --git a/.codex-plugin/plugin.json b/.codex-plugin/plugin.json index 6a2f449..9db9a4e 100644 --- a/.codex-plugin/plugin.json +++ b/.codex-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "codeguard", - "version": "0.14.13+codex.20260923", + "version": "0.14.14+codex.20260923", "description": "Evidence-backed code checks and Git content gates for AI assistants, with Maven/Gradle module impact analysis. Save hooks provide feedback; unverified checks are explicit.", "author": { "name": "Full Stack Skills / PartMe.AI", diff --git a/.zcode-plugin/plugin.json b/.zcode-plugin/plugin.json index 9daa777..b5abe8d 100644 --- a/.zcode-plugin/plugin.json +++ b/.zcode-plugin/plugin.json @@ -5,7 +5,7 @@ "en": "CodeGuard", "zh-CN": "代码规范检查" }, - "version": "0.14.13", + "version": "0.14.14", "description": "Evidence-backed code checks and Git content gates for AI assistants, with Maven/Gradle module impact analysis. Save hooks provide feedback; unverified checks are explicit.", "description_i18n": { "en": "Evidence-backed code checks and Git content gates for AI assistants, with Maven/Gradle module impact analysis. Save hooks provide feedback; unverified checks are explicit.", diff --git a/docs/current-architecture.md b/docs/current-architecture.md index 2fb7914..7009fb5 100644 --- a/docs/current-architecture.md +++ b/docs/current-architecture.md @@ -26,9 +26,9 @@ flowchart TD Entry -. 会话与去重 .-> State[storage / hook_state / cache] ``` -`execution` 只记录进程证据,不把非零退出自动认定为代码违规。`verdict`、CVE 与 Dockerfile 报告解析决定 PASS、FAIL 或 UNVERIFIED;入口仅按各自协议呈现和聚合退出码。Git 提交检查使用 index 或预测暂存快照,推送检查使用 HEAD;保存钩子只给反馈。计划、未执行和未验证不能宣传为通过。 +`execution` 只记录进程证据,不把非零退出自动认定为代码违规;stdout/stderr 合计按默认 16 MiB 捕获预算并发读取,超限终止执行、保留预算内诊断且返回 `output_limit`/UNVERIFIED,不把片段当成完整日志。`verdict`、CVE 与 Dockerfile 报告解析决定 PASS、FAIL 或 UNVERIFIED;入口仅按各自协议呈现和聚合退出码。Git 提交检查使用 index 或预测暂存快照,推送检查使用 HEAD;保存钩子只给反馈。计划、未执行和未验证不能宣传为通过。 Git 准确快照的二进制读取独立于文本检查执行器:先严格解析 index/HEAD 对象列表的 NUL 帧、模式、ID、类型/阶段和唯一文件名,并与独立路径列举核对;`cat-file` 两阶段响应再逐项核对 ID、blob 类型、大小与边界,按 SHA-1/SHA-256 blob 对象格式复算内容哈希。缺项、截断、尾部脏数据或同长度内容替换均不交付临时树,而是明确标记 Git UNVERIFIED。原始 index 不在观察过程中改写。该校验不是恶意 Git 进程沙箱或完整 Git/Shell 模拟。 -`language_check` 将 `PlanExecution` 的实际命令依序映射为应用层有界 `execution_trace`(检查、修复、复检阶段);`check_application` 将其进一步压成 MCP 安全元数据,不回显原始 argv、环境覆盖或捕获输出。MCP `auto_fix` 以 `fingerprint.file_content` 对仓内改动文件在修复前、formatter 后、复检后分别采集有总量预算的流式身份,仓外路径、符号链接或读取故障不启动 formatter;顶层 `fixed` 只归因于 formatter 且复检后仍保留的变化。公开 `fix_results` 将命令成功 `formatter_succeeded` 与逐项修复确认 `fixed` 分离;只有唯一成功执行的 formatter 且内容变化可验证时才把全局变化归给该项,多 formatter 执行不猜测归属。后续身份不可采集、formatter 失败后仍有内容变化或检查器改写目标文件时,保留实际执行证据、整体标为 UNVERIFIED,不把未知副作用声称为已确认修复。`fix_results` 只公开状态与安全元数据;formatter stderr 尾部写入私有 `.codeguard-fix.log` 后只公开路径,日志不可用时不退回公开原文。失败日志保留各已执行检查的完整输出,可能含敏感文本,应排除出版本控制。项目日志和门禁截断日志共用 `storage.write_private_text` 的私有原子落盘;预置日志文件链接不被跟随,默认 `out` 目录为链接或不可写时不返回虚假日志路径,也不覆盖检查结论。终止命令仍决定既有状态、退出码和旧字段;未运行的计划命令不进入证据。 +`language_check` 将 `PlanExecution` 的实际命令依序映射为应用层有界 `execution_trace`(检查、修复、复检阶段);`check_application` 将其进一步压成 MCP 安全元数据,不回显原始 argv、环境覆盖或捕获输出。MCP `auto_fix` 以 `fingerprint.file_content` 对仓内改动文件在修复前、formatter 后、复检后分别采集有总量预算的流式身份,仓外路径、符号链接或读取故障不启动 formatter;顶层 `fixed` 只归因于 formatter 且复检后仍保留的变化。公开 `fix_results` 将命令成功 `formatter_succeeded` 与逐项修复确认 `fixed` 分离;只有唯一成功执行的 formatter 且内容变化可验证时才把全局变化归给该项,多 formatter 执行不猜测归属。后续身份不可采集、formatter 失败后仍有内容变化或检查器改写目标文件时,保留实际执行证据、整体标为 UNVERIFIED,不把未知副作用声称为已确认修复。`fix_results` 只公开状态与安全元数据;formatter stderr 尾部写入私有 `.codeguard-fix.log` 后只公开路径,日志不可用时不退回公开原文。失败日志保留各已执行检查在捕获预算内的输出;输出超限时日志只含片段,不能称为完整诊断,且可能含敏感文本,应排除出版本控制。项目日志和门禁截断日志共用 `storage.write_private_text` 的私有原子落盘;预置日志文件链接不被跟随,默认 `out` 目录为链接或不可写时不返回虚假日志路径,也不覆盖检查结论。终止命令仍决定既有状态、退出码和旧字段;未运行的计划命令不进入证据。 `run_fix` 的历史 `fixed` 字段仍表示 formatter 命令以 0 退出,供旧调用方决定是否复检;它本身不是内容变化证明。CLI `fix` 因而只报告 formatter 执行成功与文件变化未验证,不再把退出码 0 呈现为已确认修复。 Git 命令的仓库定位与拟暂存解析共用入口传入的 cwd;显式 `cd`/`git -C` 无法绑定 Git 工作树时,硬门禁给出目标未验证并阻断,不退回到调用者仓或无关子仓。未给出显式目标且调用目录非仓时仍保留 workspace 子仓兜底。静态命令解析不等于完整 Shell 执行模拟。 workspace 兜底产生的每个子仓是合成目标,暂存观察也绑定该子仓根目录;不能继续以非 Git 的 workspace cwd 解析,从而把脏子仓的已知违规变成未知放行。 diff --git a/kimi.plugin.json b/kimi.plugin.json index 65b12fc..50df0c7 100644 --- a/kimi.plugin.json +++ b/kimi.plugin.json @@ -1,6 +1,6 @@ { "name": "codeguard", - "version": "0.14.13", + "version": "0.14.14", "description": "Evidence-backed code checks and Git content gates for AI assistants, with Maven/Gradle module impact analysis. Save hooks provide feedback; unverified checks are explicit.", "author": { "name": "Full Stack Skills / PartMe.AI" diff --git a/openspec/changes/refactor-codeguard-architecture/design.md b/openspec/changes/refactor-codeguard-architecture/design.md index a81d9e6..4cf4572 100644 --- a/openspec/changes/refactor-codeguard-architecture/design.md +++ b/openspec/changes/refactor-codeguard-architecture/design.md @@ -48,10 +48,12 @@ flowchart TD ### 2. 进程结果与代码判定分离 -不可变执行结果记录 argv、cwd、退出码、stdout、stderr、启动/超时故障。执行器不把 rc=1 判断成代码违规、不跑安装、不隐式 shell。`lint_verdict` 和 CVE 报告解析负责各工具语义。保留旧 tuple 包装以兼容已有消费者,但真实执行逻辑仅一份。 +不可变执行结果记录 argv、cwd、退出码、stdout、stderr、启动/超时/输出超限故障。执行器不把 rc=1 判断成代码违规、不跑安装、不隐式 shell。`lint_verdict` 和 CVE 报告解析负责各工具语义。保留旧 tuple 包装以兼容已有消费者,但真实执行逻辑仅一份。 超时保留部分输出;使用 UTF-8 替换无法解码的字节,防止 decode error 把整个门禁变成未知异常。缺命令=127、超时=124、其他启动 OS 错误=126;与已有 UNVERIFIED 语义一致。执行器不将非预期程序错误吞成正常进程结果;并行门禁在每个语言任务的应用边界把未预期异常显式标为 UNVERIFIED,继续保留其它语言已确认的结论,且不把异常原文注入宿主。 +第二十八批审计确认 `subprocess.run(capture_output=True)` 会先无界收集检查器输出,应用层截断不能保护进程内存。统一执行器改为两流并发读取,默认合计最多保留 16 MiB;输出超限时停止进程,返回独立 `output_limit` 故障与退出码 125,保留预算内的诊断前缀,不声称存在完整日志。POSIX 上使用独立进程组终止子进程;Windows 仍只终止直接进程,子孙生命周期需实机验证。调用方可以为确定性测试传入更小预算,但正常预算内的 argv、cwd、输出和退出契约不变。这是外部工具资源护栏,不是执行沙箱;Git 二进制快照入口另有自身预算,不经过文本执行器。 + ### 3. 策略差异必须显式,不能强行拉平 保存只检查单文件;CLI 仓检查可全量;Git 精确检查必须基于 index/HEAD;软提醒可缓存工作树结果。共用执行和计划基础,不将这些差异藏在一个越来越长的布尔参数列表中。已存在的基线比较能力保留,证据质量问题另做缺陷场景,不能把整条能力删除来换取测试通过。 diff --git a/openspec/changes/refactor-codeguard-architecture/specs/execution-kernel/spec.md b/openspec/changes/refactor-codeguard-architecture/specs/execution-kernel/spec.md index f252e88..3eb5fef 100644 --- a/openspec/changes/refactor-codeguard-architecture/specs/execution-kernel/spec.md +++ b/openspec/changes/refactor-codeguard-architecture/specs/execution-kernel/spec.md @@ -8,7 +8,7 @@ ### Requirement: Tool execution SHALL preserve process evidence -执行结果 MUST 保留实际 argv、工作目录、原始退出码、标准输出和标准错误。参数 MUST 以 argv 传递,不隐式增加 shell;工具是否发现代码违规由工具判定策略解释,不能由进程执行器推断。 +执行结果 MUST 保留实际 argv、工作目录、原始退出码、标准输出和标准错误;若外部工具输出超过明确的捕获预算,则 MUST 终止该执行并保留预算内已捕获的诊断,以独立故障标识报告 UNVERIFIED,不得伪造完整输出或 PASS。参数 MUST 以 argv 传递,不隐式增加 shell;工具是否发现代码违规由工具判定策略解释,不能由进程执行器推断。 #### Scenario: A checker exits nonzero - **WHEN** 同一检查器经 CLI、保存 hook、Git 门禁或 CVE 执行并以非零退出 @@ -18,6 +18,10 @@ - **WHEN** argv 参数包含空格、分号或命令替换字符 - **THEN** 它作为单个字面参数到达工具,不作为额外 shell 命令执行 +#### Scenario: A checker exceeds the output budget +- **WHEN** 外部检查器向 stdout/stderr 持续写入超过进程执行预算的内容 +- **THEN** 执行器有界地保留两个流的已有诊断,终止该检查,返回明确的输出超限故障与非成功退出;下游不得把截断的成功前缀判为 PASS 或宣称已有完整诊断日志 + ### Requirement: Execution failures SHALL remain observable 超时 MUST 保留已捕获输出并记录超时故障;缺失命令与权限/工作目录故障 MUST 返回明确不可验证的执行结果。非 UTF-8 输出 MUST 保留可解码文本并替换非法字节,不能让编码异常吞掉整次检查。不得捕获用户取消或任意编程异常作为成功结果。 @@ -45,7 +49,7 @@ ### Requirement: Check plans SHALL retain execution context 检查计划 MUST 显式区分 repo、delta 与 save,保存物化后的命令、工作目录和逐命令环境覆盖。执行 MUST 按序进行,首个非零结果终止普通检查批次,保留已执行证据;未执行的命令不得计为成功。空计划不得生成 PASS。Java 原生计划给出的环境覆盖 MUST 在子进程中生效,不改变宿主环境。修复后的检查 MUST 复用原计划,不重新扩大范围。Git 门禁的逐文件基线豁免是独立策略,不由通用执行器决定。 -语言检查与修复的应用结果 MUST 保留实际运行的每条命令身份、目录、退出码、故障标识和有界输出摘要;MCP 适配不得只保留终止命令,但其公开执行轨迹 MUST 只包含阶段、序号、程序名、退出码、故障标识和输出长度,不得新增暴露原始 argv、命令环境覆盖或任意检查器输出。失败日志 MUST 保留同一检查批次中此前已执行命令的完整输出,不能只记录最后一条。证据不得把尚未运行的计划命令写成已执行。 +语言检查与修复的应用结果 MUST 保留实际运行的每条命令身份、目录、退出码、故障标识和有界输出摘要;MCP 适配不得只保留终止命令,但其公开执行轨迹 MUST 只包含阶段、序号、程序名、退出码、故障标识和输出长度,不得新增暴露原始 argv、命令环境覆盖或任意检查器输出。失败日志 MUST 保留同一检查批次中此前已执行命令在捕获预算内的全部输出,不能只记录最后一条;输出超限必须明确标识未验证,不得宣称日志完整。证据不得把尚未运行的计划命令写成已执行。 MCP `auto_fix` 的 `fix_results` 公开面 MUST 同样只保留状态、退出码、安全元数据和可用诊断路径,不得因复制内部修复字典而回显 formatter 原始 argv 或 stderr。内部兼容 `fixed` 只表示 formatter 命令成功退出;公开结果 MUST 将此事实单独标为 `formatter_succeeded`,公开的逐项 `fixed` 只有在内容变化可唯一归因于这一条成功执行、且复检后身份稳定时才为真。多条 formatter 执行、读取故障或已知失败后有内容变化时,不得把全局差异猜给某条成功命令。实际修复诊断仍应在可写时以私有原子日志保存;日志写入失败不得改变修复和复检结论,也不得回退为公开原文。 #### Scenario: Java commands use the selected JDK diff --git a/openspec/changes/refactor-codeguard-architecture/tasks.md b/openspec/changes/refactor-codeguard-architecture/tasks.md index d700509..5890aec 100644 --- a/openspec/changes/refactor-codeguard-architecture/tasks.md +++ b/openspec/changes/refactor-codeguard-architecture/tasks.md @@ -33,6 +33,6 @@ - [x] 5.1 增加自动架构依赖/循环检查并接入 CI,临时源码树注入反向依赖/函数内导入/相对导入环后确实拒绝;本地已验证,远端 CI 待发布阶段。不以文件行数当合格标准。 - [x] 5.2 四个历史 skipped 已全部恢复;Dockerfile CLI 接入统一报告解析与扫描应用,14 项真实进程边界测试通过;基线比较按诊断内容和次数核验并要求基线实际 FAIL,三项误豁免用例先红后绿。完整 433 项单测零跳过、142 项回归零失败。 - [x] 5.3 新增当前架构 Mermaid 与扩展指南;报告版本读取插件 manifest,双语 README 指向当前证据,旧设计标明历史范围;双语标题/链接/版本及本地链接存在性测试通过。发布版本仍需在 5.5 检查。 -- [x] 5.4 全量单测/真实 hook 回归/ruff/语言 schema/vendor 离线与在线/本 change strict 和逐条规格追溯通过;第二十七批本地 546 项单测、143 项真实 Hook 回归、57 项/11 规则,证据见 verification.md。仅证明本地契约,不等同三宿主现场验收。 -- [x] 5.5 按仓规升级版本、生成市场元数据、授权的 PR/CI/不可变 tag/Release 闭环;第二十七批已发布 v0.14.13(插件 PR #60、市场 PR #17、源码 PR/main CI、注释 tag/正式 Release 均核对;市场 PR 无 CI 检查)。宿主安装运行独立记录,未授权不得改宿主;发布证据见 verification.md。 +- [x] 5.4 全量单测/真实 hook 回归/ruff/语言 schema/vendor 离线与在线/本 change strict 和逐条规格追溯通过;第二十八批本地 549 项单测、143 项真实 Hook 回归、57 项/11 规则,证据见 verification.md。仅证明本地契约,不等同三宿主现场验收。 +- [ ] 5.5 按仓规升级版本、生成市场元数据、授权的 PR/CI/不可变 tag/Release 闭环;第二十七批已发布 v0.14.13(插件 PR #60、市场 PR #17、源码 PR/main CI、注释 tag/正式 Release 均核对;市场 PR 无 CI 检查)。第二十八批输出预算护栏仍未 bump/推送/发布,需独立闭环;宿主安装运行另行记录,未授权不得改宿主。 - [ ] 5.6 完成全目标审计,代码/规格/证据一致后同步并归档 change,才可将整体优化标记完成。 diff --git a/openspec/changes/refactor-codeguard-architecture/verification.md b/openspec/changes/refactor-codeguard-architecture/verification.md index b4f0b61..1bb67ca 100644 --- a/openspec/changes/refactor-codeguard-architecture/verification.md +++ b/openspec/changes/refactor-codeguard-architecture/verification.md @@ -285,3 +285,10 @@ - 再沿同一 CodeGraph 路径审查 `run_fix → auto_fix → _public_fix_results`:内部 `fixed=rc==0` 被直接拷进 MCP 逐项结果,导致 no-op formatter、多个 formatter 或失败后部分改写的公开修复结论不可信。新增三个身份/归因测试先 RED;公开结果保留 `formatter_succeeded` 命令事实,并且仅唯一成功执行且复检后身份稳定时才声明逐项 `fixed`。多 formatter 不猜归属;失败且内容变化则整体 UNVERIFIED。完整单测 **546/546、0 skipped**,真实 Hook 回归 **143/0/0**;Ruff、架构检查、语言 schema **57 项/11 规则**、vendor 离线与在线、OpenSpec strict、diff whitespace 通过。内部兼容字段与复检策略未变;此处仍仅是隔离 worktree 的本地证据,未证明三宿主安装运行或版本发布。 - 第二十七批随后发布为 v0.14.13:源码 PR #60 合并到 `f683bc7b9888da97277eadce853774adf5fcaa15`,市场 PR #17 合并到 `947fc7d65b9a7f31360c1eeb96a02e62d1e8b196`。源码 PR 的 `vendor-check` 和合并后 main 的 `skills-check` 均成功;市场 PR 无 CI 检查,不称其 CI 已通过。受保护注释 tag `v0.14.13` 解引用到源码 main 合并提交,正式 GitHub Release 非 draft、非 prerelease,Codex/ZCode/Kimi 市场元数据及中英文导航均为 0.14.13。此发布不等于三宿主已安装运行、Windows、联网 CVE 或大型 Maven/Gradle 的现场验收;task 5.6 全目标审计、规格同步与归档仍未完成。 + +## 第二十八批本地验证:外部进程输出资源边界 + +- 在最新 v0.14.13 main 的隔离工作树建立临时 CodeGraph 索引(141 文件、2,226 节点、5,017 边);`execute` 影响分析覆盖 77 个符号,沿语言检查、Git 门禁、CVE、Dockerfile 与 Hook 路径追出公共执行器的无界 `capture_output=True`。应用层虽截断报告,仍在收集阶段承担无限输出的内存风险。目标真实子进程用例先因缺少捕获预算参数 RED。 +- 统一执行器现并发读取 stdout/stderr,默认共享 16 MiB 捕获预算;超限停止进程,保留预算内两流诊断,返回 `output_limit`/125 并由语言判定显式转 UNVERIFIED。退出码 125 由工具自己返回时不伪造执行器故障。POSIX 独立进程组有助于停止子孙进程;Windows 子孙进程回收尚未实机证明。已有 argv 字面值、工作目录、非法 UTF-8、超时部分诊断、非零双流和旧 tuple 协议继续通过。 +- 首轮全量发现旧 SkipGate 重试用例 mock `subprocess.run`,执行器改用 `Popen` 后该 mock 不再触及业务边界;改为在 `repository_policy.execute` 注入相同的首次超时/两次超时证据,两项断言仍核验重试和审计。最终隔离工作树本地完整单测 **549/549、0 skipped**,真实 Hook 回归 **143/0/0**;Ruff、架构门禁、语言 schema **57 项/11 规则**、vendor 离线与在线、本 change strict、diff whitespace 通过。 +- 该护栏只约束文本检查执行器;Git 二进制快照使用独立 `git_snapshot.git()`,其 `cat-file` 传输、覆盖层累计磁盘预算与恶意 Git 输出仍需另行审计。本批未 bump、推送或发布;三宿主现场、Windows、在线 CVE 与大型 Maven/Gradle 项目没有获得新证据。 diff --git a/scripts/check_architecture.py b/scripts/check_architecture.py index 12406f5..f26f2e9 100644 --- a/scripts/check_architecture.py +++ b/scripts/check_architecture.py @@ -25,7 +25,8 @@ "codeguard.models": {"__future__", "dataclasses", "pathlib", "typing"}, "codeguard.verdict": {"__future__", "re", "posixpath", "collections", "collections.abc", "pathlib", "codeguard.models"}, - "codeguard.execution": {"__future__", "os", "subprocess", "collections", "pathlib", "codeguard.models"}, + "codeguard.execution": {"__future__", "os", "signal", "subprocess", "threading", "time", + "collections", "pathlib", "codeguard.models"}, # 报告仍决定日志路径与呈现;私有原子落盘交给 storage,不称纯模型。 "codeguard.reporting": {"__future__", "re", "json", "pathlib", "hashlib", "tempfile", "codeguard.models", "codeguard.verdict", "codeguard.storage"}, diff --git a/scripts/codeguard/execution.py b/scripts/codeguard/execution.py index 9bee706..9766510 100644 --- a/scripts/codeguard/execution.py +++ b/scripts/codeguard/execution.py @@ -2,50 +2,130 @@ from __future__ import annotations import os +import signal import subprocess +import threading +import time from collections.abc import Mapping, Sequence from pathlib import Path from .models import CheckPlan, PlanExecution, ProcessResult +DEFAULT_MAX_OUTPUT_BYTES = 16 * 1024 * 1024 +_READ_SIZE = 64 * 1024 -def _text(value: str | bytes | None) -> str: - # TimeoutExpired 的输出即使 text=True 也可能是 bytes。 - if isinstance(value, bytes): - return value.decode("utf-8", errors="replace") - return value or "" + +def _text(value: bytes) -> str: + """只在有界捕获完成后解码,保留非法 UTF-8 的可读上下文。""" + return value.decode("utf-8", errors="replace") def execute(argv: Sequence[str], cwd: Path, timeout: float, *, - env: Mapping[str, str] | None = None, stdin_null: bool = False) -> ProcessResult: + env: Mapping[str, str] | None = None, stdin_null: bool = False, + max_output_bytes: int | None = None) -> ProcessResult: """按 argv 在指定目录执行,不启 shell、不更改全局 cwd/环境。 - 归一化可预期的 OS 启动错误与超时;程序错误和用户取消仍向上传播。 + 归一化可预期的 OS 启动错误、超时与输出超限;程序错误和用户取消仍向上传播。 普通工具的退出码(包括工具自己返回的 124/127)原样保留,由领域 - 判定解释;failure 只描述执行器实际观察到的启动/超时故障。 + 判定解释;failure 只描述执行器实际观察到的故障。 """ + if max_output_bytes is None: + max_output_bytes = DEFAULT_MAX_OUTPUT_BYTES + if max_output_bytes < 1: + raise ValueError("输出捕获预算必须为正数") command = tuple(argv) root = Path(cwd) try: - proc = subprocess.run( - list(command), cwd=root, capture_output=True, check=False, - text=True, encoding="utf-8", errors="replace", timeout=timeout, + proc = subprocess.Popen( + list(command), cwd=root, stdout=subprocess.PIPE, stderr=subprocess.PIPE, env={**os.environ, **env} if env is not None else None, stdin=subprocess.DEVNULL if stdin_null else None, + start_new_session=os.name == "posix", ) - except subprocess.TimeoutExpired as exc: - stderr = _text(exc.stderr) - if stderr and not stderr.endswith("\n"): - stderr += "\n" - return ProcessResult(command, root, 124, _text(exc.stdout), - stderr + f"timeout after {timeout}s", "timeout") except FileNotFoundError as exc: return ProcessResult(command, root, 127, stderr=f"command not found: {exc}", failure="not_found") except OSError as exc: return ProcessResult(command, root, 126, stderr=f"cannot start command: {exc}", failure="os_error") - return ProcessResult(command, root, proc.returncode, _text(proc.stdout), _text(proc.stderr)) + + captured = (bytearray(), bytearray()) + guard = threading.Lock() + overflow = threading.Event() + reader_error = threading.Event() + total = 0 + + def drain(stream, output: bytearray) -> None: + nonlocal total + try: + with stream: + while chunk := stream.read1(_READ_SIZE): + with guard: + keep = min(len(chunk), max_output_bytes - total) + output.extend(chunk[:keep]) + total += keep + if keep < len(chunk): + overflow.set() + except OSError: + reader_error.set() + + readers = [threading.Thread(target=drain, args=(stream, output), daemon=True) + for stream, output in zip((proc.stdout, proc.stderr), captured, strict=True)] + for reader in readers: + reader.start() + + def stop_process() -> None: + if os.name == "posix": + try: + os.killpg(proc.pid, signal.SIGKILL) + except ProcessLookupError: + pass + elif proc.poll() is None: + proc.kill() + + fault = None + try: + deadline = time.monotonic() + timeout + while proc.poll() is None: + if reader_error.is_set(): + fault = "os_error" + break + if overflow.is_set(): + fault = "output_limit" + break + remaining = deadline - time.monotonic() + if remaining <= 0: + fault = "timeout" + break + overflow.wait(min(0.02, remaining)) + if fault is not None: + stop_process() + proc.wait() + for reader in readers: + reader.join(timeout=1) + if any(reader.is_alive() for reader in readers): + stop_process() + for reader in readers: + reader.join(timeout=1) + fault = fault or "os_error" + if overflow.is_set() and fault is None: + fault = "output_limit" + if reader_error.is_set() and fault is None: + fault = "os_error" + except BaseException: + stop_process() + proc.wait() + raise + + stdout, stderr = (_text(bytes(stream)) for stream in captured) + if fault is not None: + message = {"timeout": f"timeout after {timeout}s", + "output_limit": f"output limit exceeded ({max_output_bytes} bytes)", + "os_error": "cannot finish reading command output"}[fault] + stderr += ("\n" if stderr and not stderr.endswith("\n") else "") + message + return ProcessResult(command, root, {"timeout": 124, "output_limit": 125, + "os_error": 126}.get(fault, proc.returncode), + stdout, stderr, fault) def execute_plan(plan: CheckPlan, timeout: float) -> PlanExecution: diff --git a/scripts/codeguard/gate_checks.py b/scripts/codeguard/gate_checks.py index fdc7559..2e699cd 100644 --- a/scripts/codeguard/gate_checks.py +++ b/scripts/codeguard/gate_checks.py @@ -81,7 +81,8 @@ def check_language(root: Path, cfg: dict, lang: str, *, scope: str, timeout = cfg.get("lint_timeout_seconds", 120) for index, command in enumerate(plan.commands): outcome = execute(command.argv, plan.cwd, timeout) - status, reason = lint_verdict(outcome.returncode, list(command.argv), outcome.stdout + outcome.stderr) + status, reason = lint_verdict(outcome.returncode, list(command.argv), + outcome.stdout + outcome.stderr, failure=outcome.failure) # 当前或基线工具状态未知都不授予豁免;不是先比较文本再猜工具是否运行成功。 if status == FAIL and delta and "{file}" in " ".join(base): stale = baseline_stale_finding(root, files[index], base, outcome.stdout + outcome.stderr, diff --git a/scripts/codeguard/language_check.py b/scripts/codeguard/language_check.py index 44211a4..b1bdcb2 100644 --- a/scripts/codeguard/language_check.py +++ b/scripts/codeguard/language_check.py @@ -115,7 +115,7 @@ def run_check(languages: list[str], project_root: Path, outcome = execution.terminal rc, out, err = outcome.as_tuple() lint_cmd = list(outcome.argv) - status, reason = lint_verdict(rc, lint_cmd, out + err) + status, reason = lint_verdict(rc, lint_cmd, out + err, failure=outcome.failure) if status == FAIL and fix: repairs = run_fix([lang], project_root, timeout=timeout + 60, files=files) for repair in repairs: @@ -127,7 +127,7 @@ def run_check(languages: list[str], project_root: Path, outcome = execution.terminal rc, out, err = outcome.as_tuple() lint_cmd = list(outcome.argv) - status, reason = lint_verdict(rc, lint_cmd, out + err) + status, reason = lint_verdict(rc, lint_cmd, out + err, failure=outcome.failure) results.append(result(lang, status, reason, exit_code=rc, unverified=reason if status == UNVERIFIED else "", command=lint_cmd, java_plan=java_plan, diff --git a/scripts/codeguard/models.py b/scripts/codeguard/models.py index 1866903..777f993 100644 --- a/scripts/codeguard/models.py +++ b/scripts/codeguard/models.py @@ -46,7 +46,7 @@ class ProcessResult: returncode: int stdout: str = "" stderr: str = "" - failure: Literal["timeout", "not_found", "os_error"] | None = None + failure: Literal["timeout", "not_found", "os_error", "output_limit"] | None = None def as_tuple(self) -> tuple[int, str, str]: """兼容旧脚本的 (rc, stdout, stderr) 返回约定。""" diff --git a/scripts/codeguard/verdict.py b/scripts/codeguard/verdict.py index 0a89289..daec4ba 100644 --- a/scripts/codeguard/verdict.py +++ b/scripts/codeguard/verdict.py @@ -110,8 +110,13 @@ def finding_signatures(output: str) -> set[str]: return sigs -def lint_verdict(rc: int, command: list[str], output: str = "") -> tuple[str, str]: +def lint_verdict(rc: int, command: list[str], output: str = "", *, + failure: str | None = None) -> tuple[str, str]: """按工具契约归一结论;只允许明确的成功返回 PASS。""" + if failure == "output_limit": + return UNVERIFIED, "检查器输出超过捕获上限,结果不完整" + if failure in ("timeout", "not_found", "os_error") and rc == 0: + return UNVERIFIED, "工具链执行异常,结果不完整" if rc == 0: return PASS, "检查执行成功" # ShellCheck 不支持 zsh/dash 之外的方言(SC1071 是 error 级固有限制): diff --git a/tests/test_execution_kernel.py b/tests/test_execution_kernel.py index 3fc085f..3f2cb35 100644 --- a/tests/test_execution_kernel.py +++ b/tests/test_execution_kernel.py @@ -9,8 +9,10 @@ import subprocess import sys import tempfile +import time import unittest from pathlib import Path +from unittest.mock import patch ROOT = Path(__file__).resolve().parents[1] sys.path.insert(0, str(ROOT / "scripts")) @@ -20,6 +22,7 @@ import dockerfile_security import post_tool_lint import run_per_language +from codeguard.execution import execute from verdict import UNVERIFIED, lint_verdict @@ -66,6 +69,46 @@ def test_timeout_keeps_partial_diagnostics(self): self.assertIn("timeout", err) self.assertEqual(UNVERIFIED, lint_verdict(rc, argv, out + err)[0]) + def test_excessive_output_is_bounded_and_unverified(self): + argv = [sys.executable, "-u", "-c", ( + "import sys,time; print('stdout before flood',flush=True); " + "print('stderr before flood',file=sys.stderr,flush=True); " + "time.sleep(0.1); sys.stdout.write('x'*200000); sys.stdout.flush(); time.sleep(30)" + )] + started = time.monotonic() + outcome = execute(argv, self.root, 5, max_output_bytes=1024) + self.assertLess(time.monotonic() - started, 3) + self.assertEqual("output_limit", outcome.failure) + self.assertEqual(125, outcome.returncode) + self.assertIn("stdout before flood", outcome.stdout) + self.assertIn("stderr before flood", outcome.stderr) + self.assertLessEqual(len(outcome.stdout.encode()) + + len(outcome.stderr.split("output limit exceeded")[0].encode()), 1024) + self.assertEqual((UNVERIFIED, "检查器输出超过捕获上限,结果不完整"), + lint_verdict(outcome.returncode, argv, + outcome.stdout + outcome.stderr, failure=outcome.failure)) + + def test_language_check_keeps_output_limit_as_unverified(self): + from codeguard import execution, language_check + + argv = [sys.executable, "-c", "import sys; sys.stdout.write('x'*200000)"] + with (patch.object(language_check, "LANG_COMMANDS", {"tiny": {"gate": argv, "lint": argv}}), + patch.object(language_check, "project_uses_linter", return_value=True), + patch.object(execution, "DEFAULT_MAX_OUTPUT_BYTES", 1024)): + rows = language_check.run_check(["tiny"], self.root) + self.assertEqual(1, len(rows)) + self.assertEqual("UNVERIFIED", rows[0]["status"]) + self.assertEqual("检查器输出超过捕获上限,结果不完整", rows[0]["reason"]) + self.assertEqual("output_limit", rows[0]["execution_trace"][0]["failure"]) + + def test_checker_exit_125_is_not_an_output_limit_without_executor_evidence(self): + argv = [sys.executable, "-c", "raise SystemExit(125)"] + outcome = execute(argv, self.root, 5, max_output_bytes=1024) + self.assertEqual(125, outcome.returncode) + self.assertIsNone(outcome.failure) + self.assertEqual("UNVERIFIED", lint_verdict(outcome.returncode, argv, + outcome.stdout + outcome.stderr)[0]) + @unittest.skipIf(os.name == "nt", "POSIX executable permission contract") def test_non_executable_is_unverified_not_an_uncaught_error(self): tool = self.root / "not-executable" diff --git a/tests/test_gate_hardening_batch2.py b/tests/test_gate_hardening_batch2.py index cbb182e..1eb56c4 100644 --- a/tests/test_gate_hardening_batch2.py +++ b/tests/test_gate_hardening_batch2.py @@ -20,7 +20,8 @@ import gate_lib import pre_tool_git_guard import scope -from codeguard import spec_validation +from codeguard import repository_policy, spec_validation +from codeguard.models import ProcessResult from git_snapshot import validation_tree from verdict import FAIL, UNVERIFIED, finding_signatures, lint_verdict @@ -313,18 +314,19 @@ class SkipGateRetryTests(GitRepoCase): def test_retry_after_timeout_then_success(self): _git(self.root, "config", "codeguard.skipGate", "true") calls = {"n": 0} - real_run = subprocess.run + real_execute = repository_policy.execute def flaky(*args, **kwargs): calls["n"] += 1 if calls["n"] == 1: - raise subprocess.TimeoutExpired("git", 3) - return real_run(*args, **kwargs) + return ProcessResult(("git", "config", "--get", "codeguard.skipGate"), self.root, + 124, stderr="timeout after 3s", failure="timeout") + return real_execute(*args, **kwargs) from unittest.mock import patch os.environ["CODEGUARD_HOME"] = str(Path(self._td.name) / "home") try: - with patch.object(subprocess, "run", side_effect=flaky): + with patch.object(repository_policy, "execute", side_effect=flaky): self.assertTrue(gate_lib.skip_gate_via_git_config(self.root), "第一次超时后重试成功必须算豁免") state = (Path(self._td.name) / "home" / "session_state.json") @@ -337,8 +339,9 @@ def test_two_failures_audited(self): home = Path(self._td.name) / "home" os.environ["CODEGUARD_HOME"] = str(home) try: - with patch.object(subprocess, "run", - side_effect=subprocess.TimeoutExpired("git", 3)): + failure = ProcessResult(("git", "config", "--get", "codeguard.skipGate"), self.root, + 124, stderr="timeout after 3s", failure="timeout") + with patch.object(repository_policy, "execute", return_value=failure): self.assertFalse(gate_lib.skip_gate_via_git_config(self.root)) state = json.loads((home / "session_state.json").read_text(encoding="utf-8")) self.assertIn("skipGate-read-error", state["_skip"]["kinds"])