From 91dc5dd1ca8d65426412dd8649dd75ca42d511a0 Mon Sep 17 00:00:00 2001 From: coso Date: Fri, 27 Mar 2026 19:22:46 +0800 Subject: [PATCH] feat: finalize latest v0.97.0 updates --- .github/workflows/harness-nightly.yml | 14 +- .github/workflows/release.yml | 31 +- .gitignore | 4 + docs/aiprompts/commands.md | 4 +- docs/aiprompts/governance.md | 3 +- docs/aiprompts/playwright-e2e.md | 23 +- docs/aiprompts/quality-workflow.md | 4 +- docs/design/settings-redesign.md | 2 - docs/test/README.md | 16 +- docs/test/e2e-tests.md | 8 +- docs/test/harness-evals.md | 49 +- docs/test/testing-strategy-2026.md | 36 +- package.json | 5 + scripts/check-doc-freshness.mjs | 146 ++ scripts/governance-graph.mjs | 84 +- scripts/harness-eval-history-record.mjs | 343 +++ scripts/harness-eval-runner.mjs | 280 ++- scripts/harness-eval-trend-report.mjs | 70 +- scripts/harness-replay-promote.mjs | 173 +- scripts/lib/doc-freshness-core.mjs | 380 +++ scripts/lib/doc-freshness-core.test.ts | 86 + scripts/lib/generated-slop-report-core.mjs | 1004 ++++++++ .../lib/generated-slop-report-core.test.ts | 324 +++ .../lib/harness-eval-history-record.test.ts | 87 + .../lib/harness-eval-history-window.test.ts | 226 ++ .../lib/harness-review-decision-evals.test.ts | 447 ++++ scripts/lib/legacy-surface-report-core.mjs | 688 ++++++ .../lib/legacy-surface-report-core.test.ts | 46 + scripts/lib/legacy-surface-report-summary.mjs | 176 ++ .../lib/legacy-surface-report-summary.test.ts | 86 + scripts/report-generated-slop.mjs | 293 +++ scripts/report-legacy-surfaces.mjs | 2119 +---------------- src-tauri/src/app/runner.rs | 1 + .../commands/aster_agent_cmd/command_api.rs | 3 +- .../command_api/runtime_api.rs | 51 +- src-tauri/src/commands/aster_agent_cmd/dto.rs | 26 + src-tauri/src/commands/aster_agent_cmd/mod.rs | 13 +- .../commands/aster_agent_cmd/runtime_turn.rs | 46 +- .../services/artifact_document_validator.rs | 310 ++- .../runtime_review_decision_service.rs | 356 ++- .../components/HarnessStatusPanel.test.tsx | 366 +++ .../chat/components/HarnessStatusPanel.tsx | 372 ++- .../Inputbar/hooks/useInputbarController.ts | 7 + .../Inputbar/hooks/useStickyA2UIForm.test.tsx | 125 + .../Inputbar/hooks/useStickyA2UIForm.ts | 60 + .../agent/chat/components/Inputbar/index.tsx | 3 +- .../RuntimeReviewDecisionDialog.tsx | 471 ++++ .../chat/hooks/agentStreamRuntimeHandler.ts | 21 +- .../chat/hooks/runtimeWarningPresentation.ts | 52 + .../agent/chat/hooks/useAgentStream.ts | 21 +- .../chat/hooks/useAsterAgentChat.test.tsx | 35 + .../chat/utils/threadReliabilityView.test.ts | 44 + .../agent/chat/utils/threadReliabilityView.ts | 15 +- .../workspace/ArtifactWorkbenchShell.test.tsx | 156 +- .../chat/workspace/ArtifactWorkbenchShell.tsx | 319 ++- .../useWorkspaceCanvasLayoutRuntime.test.tsx | 131 + .../useWorkspaceCanvasLayoutRuntime.ts | 24 + src/lib/api/agent.test.ts | 144 ++ src/lib/api/agentRuntime.ts | 125 + src/lib/dev-bridge/mockPriorityCommands.ts | 1 + src/lib/governance/agentCommandCatalog.json | 83 +- src/lib/governance/legacySurfaceCatalog.json | 1364 +++++++++++ .../governance/legacySurfaceCatalog.test.ts | 32 + src/lib/tauri-mock/core.ts | 109 + 64 files changed, 9744 insertions(+), 2399 deletions(-) create mode 100644 scripts/check-doc-freshness.mjs create mode 100644 scripts/harness-eval-history-record.mjs create mode 100644 scripts/lib/doc-freshness-core.mjs create mode 100644 scripts/lib/doc-freshness-core.test.ts create mode 100644 scripts/lib/generated-slop-report-core.mjs create mode 100644 scripts/lib/generated-slop-report-core.test.ts create mode 100644 scripts/lib/harness-eval-history-record.test.ts create mode 100644 scripts/lib/harness-eval-history-window.test.ts create mode 100644 scripts/lib/harness-review-decision-evals.test.ts create mode 100644 scripts/lib/legacy-surface-report-core.mjs create mode 100644 scripts/lib/legacy-surface-report-core.test.ts create mode 100644 scripts/lib/legacy-surface-report-summary.mjs create mode 100644 scripts/lib/legacy-surface-report-summary.test.ts create mode 100644 scripts/report-generated-slop.mjs create mode 100644 src/components/agent/chat/components/Inputbar/hooks/useStickyA2UIForm.test.tsx create mode 100644 src/components/agent/chat/components/Inputbar/hooks/useStickyA2UIForm.ts create mode 100644 src/components/agent/chat/components/RuntimeReviewDecisionDialog.tsx create mode 100644 src/components/agent/chat/hooks/runtimeWarningPresentation.ts create mode 100644 src/components/agent/chat/workspace/useWorkspaceCanvasLayoutRuntime.test.tsx create mode 100644 src/lib/governance/legacySurfaceCatalog.json create mode 100644 src/lib/governance/legacySurfaceCatalog.test.ts diff --git a/.github/workflows/harness-nightly.yml b/.github/workflows/harness-nightly.yml index 03d8e89e9..d034aaba1 100644 --- a/.github/workflows/harness-nightly.yml +++ b/.github/workflows/harness-nightly.yml @@ -33,12 +33,11 @@ jobs: - name: Generate harness eval summary run: | - mkdir -p "./artifacts/history" node scripts/harness-eval-runner.mjs \ --output-json "./artifacts/harness-eval-summary.json" \ - --output-markdown "./artifacts/harness-eval-summary.md" - cp "./artifacts/harness-eval-summary.json" "./artifacts/history/$(date -u +%Y%m%dT%H%M%SZ)-harness-eval-summary.json" - ls -1t "./artifacts/history"/*.json 2>/dev/null | tail -n +31 | xargs -r rm -f + --output-markdown "./artifacts/harness-eval-summary.md" \ + --record-history-dir "./artifacts/history" \ + --history-retain 30 - name: Generate harness eval trend run: | @@ -47,6 +46,13 @@ jobs: --output-json "./artifacts/harness-eval-trend.json" \ --output-markdown "./artifacts/harness-eval-trend.md" + - name: Generate harness cleanup report + run: | + node scripts/report-generated-slop.mjs \ + --trend-input "./artifacts/harness-eval-trend.json" \ + --output-json "./artifacts/harness-cleanup-report.json" \ + --output-markdown "./artifacts/harness-cleanup-report.md" + - name: Upload harness eval artifact uses: actions/upload-artifact@v4 with: diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index fa065577e..6c76eadf5 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -52,7 +52,6 @@ jobs: TAURI_SIGNING_PRIVATE_KEY_RAW: ${{ secrets.TAURI_SIGNING_PRIVATE_KEY }} TAURI_SIGNING_PRIVATE_KEY_PASSWORD: ${{ secrets.TAURI_SIGNING_PRIVATE_KEY_PASSWORD }} NODE_OPTIONS: --max-old-space-size=8192 - SCCACHE_NO_DAEMON: "1" steps: - name: Checkout @@ -61,6 +60,11 @@ jobs: fetch-depth: 0 ref: ${{ github.event.inputs.source_ref || github.ref }} + - name: Setup pnpm + uses: pnpm/action-setup@v4 + with: + version: 9 + - name: Setup Node.js uses: actions/setup-node@v4 with: @@ -68,11 +72,6 @@ jobs: cache: "pnpm" cache-dependency-path: "pnpm-lock.yaml" - - name: Setup pnpm - uses: pnpm/action-setup@v4 - with: - version: 9 - - name: Setup Rust uses: dtolnay/rust-toolchain@stable with: @@ -384,7 +383,11 @@ jobs: run: | set -euo pipefail TARGET_TRIPLE="${{ matrix.target }}" - mapfile -t bundle_dirs < <(find . -type d -path "*/target/${TARGET_TRIPLE}/release/bundle" | sort) + bundle_dirs=() + while IFS= read -r bundle_dir; do + [ -n "$bundle_dir" ] || continue + bundle_dirs+=("$bundle_dir") + done < <(find . -type d -path "*/target/${TARGET_TRIPLE}/release/bundle" | sort) if [ "${#bundle_dirs[@]}" -eq 0 ]; then echo "Bundle dir not created yet for ${TARGET_TRIPLE}" exit 0 @@ -425,7 +428,11 @@ jobs: run: | set -euo pipefail TARGET_TRIPLE="${{ matrix.target }}" - mapfile -t bundle_dirs < <(find . -type d -path "*/target/${TARGET_TRIPLE}/release/bundle" | sort) + bundle_dirs=() + while IFS= read -r bundle_dir; do + [ -n "$bundle_dir" ] || continue + bundle_dirs+=("$bundle_dir") + done < <(find . -type d -path "*/target/${TARGET_TRIPLE}/release/bundle" | sort) if [ "${#bundle_dirs[@]}" -eq 0 ]; then echo "Bundle dir not created yet for ${TARGET_TRIPLE}" exit 0 @@ -527,10 +534,14 @@ jobs: exit 1 fi - mapfile -d '' assets < <( + assets=() + while IFS= read -r asset; do + [ -n "$asset" ] || continue + assets+=("$asset") + done < <( find "$BUNDLE_DIR" -type f \ \( -name "*.dmg" -o -name "*.app.tar.gz" -o -name "*.app.tar.gz.sig" -o -name "latest*.json" \) \ - -print0 | sort -z + | sort ) if [ "${#assets[@]}" -eq 0 ]; then diff --git a/.gitignore b/.gitignore index 40a8e75dc..1a8142dc1 100644 --- a/.gitignore +++ b/.gitignore @@ -90,3 +90,7 @@ target-codex* tmp/ governance/ +!src/lib/governance/ +!src/lib/governance/*.json +!src/lib/governance/*.mjs +!src/lib/governance/*.test.ts diff --git a/docs/aiprompts/commands.md b/docs/aiprompts/commands.md index 4489d0017..506a1d2f4 100644 --- a/docs/aiprompts/commands.md +++ b/docs/aiprompts/commands.md @@ -194,7 +194,7 @@ npm run verify:local - **运行时证据导出主链**:继续收敛到 `agent_runtime_export_evidence_pack`,用于把 runtime / timeline / artifacts 打包成最小问题证据 - **运行时 replay 样本主链**:继续收敛到 `agent_runtime_export_replay_case`,复用 handoff bundle + evidence pack 生成 `input / expected / grader / evidence-links` - **运行时外部分析交接主链**:继续收敛到 `agent_runtime_export_analysis_handoff`,复用 handoff bundle + evidence pack + replay case 生成 `analysis-brief.md / analysis-context.json / copy_prompt`,供外部 Claude Code / Codex 直接诊断与最小修复;当前 GUI 入口位于 `HarnessStatusPanel` -- **运行时人工审核记录主链**:继续收敛到 `agent_runtime_export_review_decision_template`,复用 `analysis handoff` 生成 `review-decision.md / review-decision.json`,把开发者的接受 / 延后 / 拒绝与回归要求回挂到工作区;当前 GUI 入口位于 `HarnessStatusPanel` +- **运行时人工审核记录主链**:继续收敛到 `agent_runtime_export_review_decision_template` + `agent_runtime_save_review_decision`;前者复用 `analysis handoff` 生成 `review-decision.md / review-decision.json` 模板,后者把开发者的接受 / 延后 / 拒绝与回归要求回写到同一份工作区制品;当前 GUI 入口位于 `HarnessStatusPanel` - **会话主题上下文主链**:`getSession` 返回的 `execution_runtime.recent_theme / recent_session_mode` 负责承接最近一次运行态主题上下文;当前端已命中同一 steady-state theme/workbench mode 时,不应继续每回合重复携带 `harness.theme / harness.session_mode` - **会话运行阶段上下文主链**:`getSession` 返回的 `execution_runtime.recent_gate_key / recent_run_title` 负责承接最近一次 Theme Workbench 运行阶段上下文;当前端已命中同一 steady-state gate/run 时,不应继续每回合重复携带 `harness.gate_key / harness.run_title` - **会话内容上下文主链**:`getSession` 返回的 `execution_runtime.recent_content_id` 负责承接最近一次运行态 `content_id`;当前端已命中同一 steady-state 内容时,不应继续每回合重复携带 `harness.content_id` @@ -233,3 +233,5 @@ npm run verify:local - `docs/aiprompts/governance.md` - `docs/aiprompts/quality-workflow.md` - `docs/aiprompts/credential-pool.md` +- `src/lib/governance/agentCommandCatalog.json` +- `src/lib/governance/legacySurfaceCatalog.json` diff --git a/docs/aiprompts/governance.md b/docs/aiprompts/governance.md index 55e44637e..141a4b2e1 100644 --- a/docs/aiprompts/governance.md +++ b/docs/aiprompts/governance.md @@ -132,10 +132,11 @@ npm run test:contracts - 扫描已被判定为 `deprecated` / `dead-candidate` 的前端入口 - 检查旧 Tauri 命令是否仍被限制在指定 API 网关 - 找出已经零引用、可进入删除候选的兼容壳 + - 规则事实源优先看 `src/lib/governance/legacySurfaceCatalog.json` - `test:contracts` - 检查前端 `safeInvoke(...)` / `invoke(...)` 的实际调用 - 检查 Rust `tauri::generate_handler!` 的实际注册 - - 检查 `agentCommandCatalog` 中的治理口径 + - 检查 `src/lib/governance/agentCommandCatalog.json` 中的命令治理口径 - 检查 `mockPriorityCommands` 与 `defaultMocks` 是否同步 原则只有一句: diff --git a/docs/aiprompts/playwright-e2e.md b/docs/aiprompts/playwright-e2e.md index bb00350e7..a1bb4213b 100644 --- a/docs/aiprompts/playwright-e2e.md +++ b/docs/aiprompts/playwright-e2e.md @@ -182,14 +182,27 @@ npm run test:contracts - `input / expected / grader / evidence-links` 文件列表出现 - replay 区块能显示 handoff / evidence 的关联根路径 - 打开目录后工作区内确实生成 `.lime/harness/sessions//replay` -8. 如果这轮继续开发外部分析交接,再点击 `导出分析交接` 与 `一键复制给 AI`,确认: +8. 如果这轮继续开发 replay -> eval 主链,再点击 `复制回归命令`,确认: + - 剪贴板内容同时包含 `npm run harness:eval:promote -- ...`、`npm run harness:eval` 与 `npm run harness:eval:trend` + - promote 命令里的 `session-id / slug / title` 已自动带出,不需要手工补参数 + - 该入口只是复制仓库已有主命令,不是 Lime 内部自动 promotion +9. 如果这轮继续开发外部分析交接,再点击 `导出分析交接` 与 `一键复制给 AI`,确认: - `analysis-brief.md / analysis-context.json` 文件列表出现 - 复制内容直接来自后端 `copy_prompt`,不需要前端再手写 prompt - analysis 区块能显示 handoff / evidence / replay 的关联目录 -9. 如果这轮继续开发人工审核记录,再点击 `导出人工审核记录`,确认: - - `review-decision.md / review-decision.json` 文件列表出现 - - 区块能显示默认状态、审核清单与关联 analysis 文件 - - 打开目录后工作区内确实生成 `.lime/harness/sessions//review` +10. 如果这轮继续开发人工审核记录,再点击 `导出人工审核记录`,确认: + - `review-decision.md / review-decision.json` 文件列表出现 + - 区块能显示当前状态、审核清单与关联 analysis 文件 + - 打开目录后工作区内确实生成 `.lime/harness/sessions//review` +11. 如果这轮继续开发人工审核保存闭环,再点击 `填写人工审核结果`,至少填写: + - `决策状态` + - `决策摘要` + - `审核人` + - `风险等级` +12. 保存后确认: + - 区块里的“当前人工审核结论”立即刷新为最新状态、审核人和摘要 + - `review-decision.md / review-decision.json` 仍然保持同一目录,不会新开平级目录 + - 如页面桥接到了真实后端,重新点击 `导出人工审核记录` 后,已保存结论不会被刷回 `pending_review` ### 话题内容上下文恢复验证 diff --git a/docs/aiprompts/quality-workflow.md b/docs/aiprompts/quality-workflow.md index 14a9ca4f9..9d7d2c26d 100644 --- a/docs/aiprompts/quality-workflow.md +++ b/docs/aiprompts/quality-workflow.md @@ -216,9 +216,9 @@ npm run bridge:health -- --timeout-ms 120000 - 切换到新的 Theme Workbench gate 或运行标题、但 runtime 尚未同步时,前端仍会保留显式 `gate_key / run_title` - 如果这次改动影响浏览器工作台里的站点采集链路,例如推荐区、资料自动选择、`report_hint` 展示、`lime_site_recommend`,或“优先写回当前 `content_id` 而不是新建资源文档”的主线收敛,除了契约检查,还应补对应 `*.test.tsx` 回归并执行 `verify:gui-smoke`。 - 如果这次改动影响浏览器资料 / 环境预设的真实来源,还应补一次浏览器模式实测,确认控制台不再出现 `[Mock] invoke: list_browser_profiles_cmd` 或 `[Mock] invoke: list_browser_environment_presets_cmd`。 -- 如果这次改动影响 `agent_runtime_export_handoff_bundle`、`agent_runtime_export_evidence_pack`、`agent_runtime_export_analysis_handoff`、`agent_runtime_export_review_decision_template` 或 `agent_runtime_export_replay_case` 这条 Harness 导出主链,除了契约检查,还应至少补: +- 如果这次改动影响 `agent_runtime_export_handoff_bundle`、`agent_runtime_export_evidence_pack`、`agent_runtime_export_analysis_handoff`、`agent_runtime_export_review_decision_template`、`agent_runtime_save_review_decision` 或 `agent_runtime_export_replay_case` 这条 Harness 导出 / 审核主链,除了契约检查,还应至少补: - `src/lib/api/agent.test.ts` 一类的网关回归,确认仍走统一 `agent_runtime_*` 主命令 - - `HarnessStatusPanel.test.tsx` 一类的 UI 回归,确认导出入口、状态与制品展示正常 + - `HarnessStatusPanel.test.tsx` 一类的 UI 回归,确认导出入口、保存弹窗、状态与制品展示正常 - 受影响 Rust 服务 / 命令的定向测试,确认 `.lime/harness/sessions//...` 一类制品仍能生成 ## CI 事实源 diff --git a/docs/design/settings-redesign.md b/docs/design/settings-redesign.md index dca077394..774236c80 100644 --- a/docs/design/settings-redesign.md +++ b/docs/design/settings-redesign.md @@ -1,7 +1,5 @@ # Lime 设置页面重构设计 -> 参考 LobeHub 的设置架构,为 Lime 设计现代化的设置界面 - ## 一、设计目标 1. **分类清晰**:将设置项按功能分组,便于用户快速定位 diff --git a/docs/test/README.md b/docs/test/README.md index 475aa8dc5..8908fef86 100644 --- a/docs/test/README.md +++ b/docs/test/README.md @@ -102,10 +102,12 @@ npm run verify:local:full npm run bridge:health -- --timeout-ms 120000 ``` -### 运行首条自包含 smoke +### 运行自包含 smoke ```bash npm run smoke:workspace-ready +npm run smoke:browser-runtime +npm run smoke:site-adapters ``` ### 运行 Harness eval 摘要 @@ -126,6 +128,18 @@ npm run harness:eval:promote -- --session-id "session-123" --slug "pending-reque npm run harness:eval:trend ``` +### 记录 Harness eval 历史窗口 + +```bash +node scripts/harness-eval-runner.mjs --record-history-dir "./artifacts/history" --history-retain 30 +``` + +### 运行 Harness cleanup / slop 报告 + +```bash +npm run harness:cleanup-report +``` + ### 当前浏览器续测入口 当前仓库的浏览器模式 E2E / 续测文档分两层: diff --git a/docs/test/e2e-tests.md b/docs/test/e2e-tests.md index b6b1a532d..463058a21 100644 --- a/docs/test/e2e-tests.md +++ b/docs/test/e2e-tests.md @@ -10,7 +10,9 @@ - `npm run tauri:dev:headless`:当前浏览器模式启动入口 - `npm run bridge:health -- --timeout-ms 120000`:当前 DevBridge 就绪检查入口 - `npm run test:bridge`:当前浏览器桥接最小自动校验入口 -- `npm run smoke:workspace-ready`:当前首条自包含 smoke,覆盖 DevBridge 就绪与默认 workspace 基础链路 +- `npm run smoke:workspace-ready`:当前自包含 smoke,覆盖 DevBridge 就绪与默认 workspace 基础链路 +- `npm run smoke:browser-runtime`:当前自包含 smoke,覆盖 browser runtime 的启动、状态读取、最小动作与审计关联键 +- `npm run smoke:site-adapters`:当前自包含 smoke,覆盖站点适配器目录状态、列表、推荐与检索主链 ### supplement @@ -90,6 +92,8 @@ npm run bridge:health -- --timeout-ms 120000 | 等待 Bridge 就绪 | `npm run bridge:health -- --timeout-ms 120000` | current | 当前标准健康检查 | | 校验桥接基础能力 | `npm run test:bridge` | current | `safeInvoke` / mock / tauri-mock 最小自动校验 | | Workspace 自包含 smoke | `npm run smoke:workspace-ready` | current | 验证 DevBridge、默认 workspace、路径回查链路 | +| Browser Runtime smoke | `npm run smoke:browser-runtime` | current | 验证 browser runtime 最短主链与审计关联键 | +| Site Adapter smoke | `npm run smoke:site-adapters` | current | 验证站点适配器目录、推荐与检索最短主链 | | 校验跨层命令契约 | `npm run test:contracts` | current | 检查前端命令、Rust 注册、catalog、mock 集合漂移 | | 浏览器续测细则 | `docs/aiprompts/playwright-e2e.md` | current | Playwright MCP 唯一详细事实源 | | 专项 bridge 排障 | `npm run bridge:e2e` | supplement | 适合排障,不是统一门禁 | @@ -112,7 +116,7 @@ npm run bridge:health -- --timeout-ms 120000 - 假设 `tauri-driver` 仍是推荐路径 - 假设浏览器 E2E 已进入 CI 标准门禁 -当前浏览器主链路 smoke 仍属于后续建设项,详见 `docs/test/testing-strategy-2026.md`。 +当前浏览器最小 smoke 基线已经具备;后续是否继续补 terminal / server / 专项 smoke,以 `docs/test/testing-strategy-2026.md` 的剩余优先级为准。 ## 7. 给后续 Agent 的交接要求 diff --git a/docs/test/harness-evals.md b/docs/test/harness-evals.md index 15f276065..2e260691d 100644 --- a/docs/test/harness-evals.md +++ b/docs/test/harness-evals.md @@ -17,6 +17,7 @@ Lime 当前不直接把“真实模型重放平台”一次做完,而是先固 - `grader.md` - `evidence-links.json` 其中 `input.json` 继续承载 `classification.suiteTags` 与 `classification.failureModes`。 + 如果样本已经完成人工审核,还可以额外挂载可选的 `review-decision.json` / `review-decision.md`,但它不是 replay case 的必填件。 3. **固定摘要出口** 由 `scripts/harness-eval-runner.mjs` 统一产出 JSON / Markdown 摘要,后续 nightly 与趋势报表都从这里接。 @@ -86,10 +87,16 @@ Lime 当前不直接把“真实模型重放平台”一次做完,而是先固 1. 读取 manifest 2. 解析固定 fixture 与工作区自动发现 case 3. 校验 replay case 最小四件套与关键 JSON 字段 -4. 输出统一 JSON / Markdown 摘要,并聚合 `suite tag / failure mode` 分布 +4. 输出统一 JSON / Markdown 摘要,并聚合 `suite tag / failure mode / review decision status / risk level` 分布 当前它**不直接执行真实模型重放**,而是先把“样本是否可评估、摘要是否可归档”工程化。 +如果 case 根目录内存在 `review-decision.json`,或者工作区 replay 同级 `../review/review-decision.json` 已存在,runner 会把它识别为同一条会话的可选人工审核增强信息,并写入: + +- `reviewDecisionRecordedCount` +- `reviewDecisionStatuses` +- `reviewRiskLevels` + 这符合 Lime 当前阶段的约束: - 先复用现有 `handoff bundle + evidence pack + replay export` @@ -115,12 +122,13 @@ node scripts/harness-replay-promote.mjs \ --slug "pending-request-runtime" ``` -这个命令会做四件事: +这个命令会做五件事: 1. 读取 replay 最小四件套。 2. 把工作区绝对路径脱敏成稳定占位路径,避免把本机路径直接写进仓库。 -3. 把样本复制到 `docs/test/harness-fixtures/replay//`。 -4. 把 case 回写到 `repo-promoted-replays` suite,成为 nightly 与 trend 的 current 样本。 +3. 如果同会话 `review/` 目录里已经有 `review-decision.json/md`,一起复制并脱敏到目标 fixture。 +4. 把样本复制到 `docs/test/harness-fixtures/replay//`,并把人工审核摘要回写到 manifest case。 +5. 把 case 回写到 `repo-promoted-replays` suite,成为 nightly 与 trend 的 current 样本。 默认原则: @@ -128,12 +136,24 @@ node scripts/harness-replay-promote.mjs \ - promotion 之后,样本不再只是“本机能看到”,而是仓库 current 主线的一部分。 - 仓库沉淀样本仍然复用原来的 handoff / evidence 形状,不另造 schema。 +如果当前已经在 Lime 工作台里导出了 Replay 样本,也可以直接在 `HarnessStatusPanel` 的 Replay 区块点击: + +- `复制回归命令` + +它会一次性复制三条现成命令: + +1. `npm run harness:eval:promote -- ...` +2. `npm run harness:eval` +3. `npm run harness:eval:trend` + +这样做的目的不是把 promotion 内建进 Lime,而是把仓库已有的 current 主命令直接挂到工作台,避免用户还要自己重新拼 `session-id / slug / title`,并且在 promotion 后立刻补上统一 trend 入口。 + ## Trend Report 做什么 `scripts/harness-eval-trend-report.mjs` 当前负责三件事: 1. 读取一个或多个 `harness eval summary` JSON -2. 生成 baseline / latest 对比、suite 级 delta,以及 `suite tag / failure mode` 聚合变化 +2. 生成 baseline / latest 对比、suite 级 delta,以及 `suite tag / failure mode / review decision status / risk level` 聚合变化 3. 输出 JSON / Markdown 趋势报告 如果没有显式提供输入,它会先调用 `harness-eval-runner` 生成当前 summary,再把它当作第一条 trend seed。 @@ -148,6 +168,10 @@ node scripts/harness-replay-promote.mjs \ 当前 nightly 还会恢复并追加 `artifacts/history/*.json` 历史窗口,用于让 trend 不只停留在单次 seed。 +在当前仓库主链里,nightly 还会基于 trend + doc freshness + governance report 继续生成 `harness-cleanup-report.json/md`,让回归样本、人工审核状态和治理建议进入同一份 nightly artifact。 + +从 `2026-03-27` 起,这个历史窗口不再依赖 workflow 里的 `cp / ls / xargs` 拼接,而是直接由 `scripts/harness-eval-runner.mjs --record-history-dir` 负责写入和裁剪,保证本地与 nightly 走同一条跨平台主链。 + ## 常用命令 ```bash @@ -171,6 +195,11 @@ node scripts/harness-eval-runner.mjs \ --output-json "./tmp/harness-eval-summary.json" \ --output-markdown "./tmp/harness-eval-summary.md" +# 记录本地 history window,供 trend / cleanup 复用 +node scripts/harness-eval-runner.mjs \ + --record-history-dir "./artifacts/history" \ + --history-retain 30 + # 从历史 summary 目录生成趋势报告 node scripts/harness-eval-trend-report.mjs \ --history-dir "./artifacts/history" \ @@ -188,6 +217,7 @@ Runner 摘要至少回答下面这些问题: - 哪些 case JSON 字段不完整 - 哪些 case 属于什么 suite tag / failure mode - 哪些 case 默认需要人工复核 +- 哪些 case 已经记录人工审核状态与风险等级 - 工作区 replay 是否已经开始形成增量样本 如果摘要回答不了这些问题,就说明 runner 还不算进入 current 主链。 @@ -197,6 +227,7 @@ Trend 报告至少还要回答: - baseline 和 latest 之间,ready / invalid / pending request 有没有变化 - 哪些 suite 在 latest 里变差了 - 哪些 failure mode / suite tag 在 latest 里增长或退化了 +- 哪些人工审核状态 / 风险等级在 latest 里新增、减少或发生迁移 - 当前只有 trend seed,还是已经开始形成真正的历史窗口 ## 与其他事实源的关系 @@ -207,16 +238,20 @@ Trend 报告至少还要回答: | [testing-strategy-2026.md](testing-strategy-2026.md) | 解释为什么 eval 工程化排在 smoke 之后 | | [../tech/harness/implementation-blueprint.md](../tech/harness/implementation-blueprint.md) | 解释 `P3-2 Eval runner` 在 Harness 主线中的位置 | | [../tech/harness/tooling-roadmap.md](../tech/harness/tooling-roadmap.md) | 解释 runner、nightly、trend 的后续工具面 | +| [../tech/harness/entropy-governance-workflow.md](../tech/harness/entropy-governance-workflow.md) | 解释 trend 怎样回挂到 cleanup / governance 建议 | | `scripts/harness-eval-runner.mjs` | 当前唯一的 runner 入口 | | `scripts/harness-eval-trend-report.mjs` | 当前 trend 聚合与 nightly 趋势出口 | +| `scripts/report-generated-slop.mjs` | 当前 cleanup/slop 聚合与治理建议入口 | +| `scripts/check-doc-freshness.mjs` | 当前 Harness 文档保鲜检查入口 | +| [../../.github/workflows/harness-nightly.yml](../../.github/workflows/harness-nightly.yml) | 当前 nightly summary / trend / cleanup artifact 主入口 | | [harness-evals.manifest.json](harness-evals.manifest.json) | 当前任务集与 suite 机可读事实源 | ## 下一刀 `P3-6` 做完之后,下一刀优先级建议固定为: -1. 把分类聚合直接挂到熵治理清单,形成 replay 驱动 cleanup 主线 -2. 继续补 observability 证据字段,让 grader 能消费更多 request / timeline / artifact 关联 +1. 把 cleanup report 接入更稳定的历史窗口,避免长期停留在 trend seed +2. 继续补 observability 证据字段,让 grader 和 cleanup report 都能消费更多 request / timeline / artifact 关联 3. 逐步提高 repo current 样本质量,而不是只增加数量 4. 再考虑是否引入真实模型执行或 transcript grading diff --git a/docs/test/testing-strategy-2026.md b/docs/test/testing-strategy-2026.md index f075bb067..b19b2c75d 100644 --- a/docs/test/testing-strategy-2026.md +++ b/docs/test/testing-strategy-2026.md @@ -40,31 +40,21 @@ - 旧权限表面治理护栏已经补齐:`src/lib/governance/legacyToolPermissionGuard.test.ts` + `npm run governance:legacy-report` - 跨层命令契约检查基础版已经落地:`npm run test:contracts` 已进入 `scripts/local-ci.mjs` - 命令契约延期例外已经收口:`agent_terminal_command_response`、`agent_term_scrollback_response` 已退出 `runtimeGatewayCommands`,改为 `dead-candidate` 治理监控 -- 首条自包含 smoke 已落地:`npm run smoke:workspace-ready` 可自动校验 DevBridge 就绪、默认 workspace 获取、目录修复与路径回查 +- 自包含 smoke 最小基线已落地:`npm run smoke:workspace-ready`、`npm run smoke:browser-runtime`、`npm run smoke:site-adapters` 都无需人工准备,且 `npm run verify:gui-smoke` 已默认串联这三条主链 smoke - 测试文档事实源已经收口:`docs/test/README.md`、`docs/test/e2e-tests.md`、`docs/aiprompts/playwright-e2e.md` 已按“索引 / 总览 / 详细事实源”分层 ## 3. 当前仍未解决的问题优先级 | 优先级 | 事项 | 为什么重要 | 当前证据 | 完成定义 | | ------ | ------------------------- | ---------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------- | -| P0 | 自包含 smoke 仍然不足 | 单测很多,但主链路仍缺少无需人工准备的自动回归 | 目前仅有 `smoke:workspace-ready` 属于自包含 smoke;`smoke:social-workbench` 仍依赖已有 session,`bridge:e2e` 更像排障脚本 | 至少补齐 3 条无需人工准备的 smoke;当前已完成 1 条,仍需补 server / terminal / browser runtime 等 2 条以上 | -| P1 | Agent eval 仍未完全工程化 | 价值高,但建立在前面基础门禁稳定之后 | 已补 `docs/test/harness-evals.md`、`harness-evals.manifest.json`、`scripts/harness-eval-runner.mjs`、`scripts/harness-eval-trend-report.mjs` 与 nightly 摘要 / trend 骨架,但真实执行与更多高价值样本仍缺 | 形成稳定任务集、可增长 replay 样本、grader、nightly 输出与趋势指标 | +| P1 | Agent eval 仍未完全工程化 | 价值高,且当前最缺的是把证据沉淀成长期回归资产 | 已补 `docs/test/harness-evals.md`、`harness-evals.manifest.json`、`scripts/harness-eval-runner.mjs`、`scripts/harness-eval-trend-report.mjs` 与 nightly 摘要 / trend 骨架,但真实执行与更多高价值样本仍缺 | 形成稳定任务集、可增长 replay 样本、grader、nightly 输出与趋势指标 | +| P2 | terminal / server 自包含 smoke 仍可继续扩面 | 最小 GUI smoke 基线已具备,但更细分主链仍缺专项守卫 | 当前 `workspace-ready / browser-runtime / site-adapters` 已覆盖 GUI 最小主链;`smoke:social-workbench` 仍依赖已有 session,terminal / server 还没有各自独立的自包含 smoke 入口 | 如后续需要继续扩面,应补 terminal 或 server 的独立 smoke,而不是继续把现有 3 条 current smoke 算成缺口 | ## 4. 建议执行顺序 -### 第 1 步:把 smoke 升级为自包含场景 +### 第 1 步:把 Agent eval 工程化 -先只挑 3 条最高价值场景,不要贪多: - -1. 应用启动 + workspace 可创建 / 打开 -2. server 基础链路可自动打通 -3. terminal 或 browser runtime 至少有一条基础链路可自动打通 - -验收标准是“本地和 CI 都能重复执行”,而不是“方便人工排障”。 - -### 第 2 步:把 Agent eval 工程化 - -这一步放在最后,不是因为不重要,而是它依赖前面的基础设施稳定: +现在可以把它提到第一优先级,因为前面的最小 GUI smoke 基线已经具备: - 有稳定门禁 - 有稳定契约检查 @@ -82,11 +72,21 @@ - 更多真实高价值 replay 样本 - 更长窗口的趋势报表 +### 第 2 步:按需继续扩自包含 smoke 覆盖面 + +如果后续还要补 smoke,不要重复把 `workspace-ready / browser-runtime / site-adapters` 记成“未完成”。 + +下一轮更合理的扩面方向是: + +1. terminal 基础链路 +2. server 基础链路 +3. 仍依赖人工前置状态的专项 smoke 去人工化 + ## 5. 当前建议 如果只看投入产出比,当前最值得先做的两刀是: -1. 把 smoke 升级为自包含场景 -2. 把 Agent eval 工程化 +1. 把 Agent eval 工程化 +2. 如需继续补 smoke,优先做 terminal / server 专项自包含场景 -这两步做完之后,再继续往 nightly 与趋势报表收口,收益会更高。 +这两步做完之后,再继续往 nightly、趋势报表与 replay promotion 收口,收益会更高。 diff --git a/package.json b/package.json index b51e5ca35..d3ebe2ced 100644 --- a/package.json +++ b/package.json @@ -43,8 +43,13 @@ "harness:eval": "node scripts/harness-eval-runner.mjs", "harness:eval:json": "node scripts/harness-eval-runner.mjs --format json", "harness:eval:promote": "node scripts/harness-replay-promote.mjs", + "harness:eval:history:record": "node scripts/harness-eval-history-record.mjs", "harness:eval:trend": "node scripts/harness-eval-trend-report.mjs", "harness:eval:trend:json": "node scripts/harness-eval-trend-report.mjs --format json", + "harness:cleanup-report": "node scripts/report-generated-slop.mjs", + "harness:cleanup-report:json": "node scripts/report-generated-slop.mjs --format json", + "harness:doc-freshness": "node scripts/check-doc-freshness.mjs", + "harness:doc-freshness:json": "node scripts/check-doc-freshness.mjs --json", "lint:rust": "cargo clippy --manifest-path \"src-tauri/Cargo.toml\"", "detect-translations": "tsx scripts/detect-missing-translations.ts", "detect-translations:fix": "tsx scripts/detect-missing-translations.ts --fix", diff --git a/scripts/check-doc-freshness.mjs b/scripts/check-doc-freshness.mjs new file mode 100644 index 000000000..1159885ee --- /dev/null +++ b/scripts/check-doc-freshness.mjs @@ -0,0 +1,146 @@ +#!/usr/bin/env node + +import fs from "node:fs"; +import path from "node:path"; +import process from "node:process"; +import { fileURLToPath } from "node:url"; + +import legacySurfaceCatalog from "../src/lib/governance/legacySurfaceCatalog.json" with { type: "json" }; +import { + DOC_FRESHNESS_SPECS, + buildDocFreshnessReport, + renderDocFreshnessText, +} from "./lib/doc-freshness-core.mjs"; + +export { + DOC_FRESHNESS_SPECS, + buildDocFreshnessReport, + renderDocFreshnessText, +} from "./lib/doc-freshness-core.mjs"; + +function parseArgs(argv) { + const options = { + format: "text", + help: false, + output: "", + }; + + for (let index = 0; index < argv.length; index += 1) { + const arg = argv[index]; + + if (arg === "--format" && argv[index + 1]) { + options.format = String(argv[index + 1]).trim(); + index += 1; + continue; + } + + if (arg === "--output" && argv[index + 1]) { + options.output = String(argv[index + 1]).trim(); + index += 1; + continue; + } + + if (arg === "--json") { + options.format = "json"; + continue; + } + + if (arg === "--help" || arg === "-h") { + options.help = true; + } + } + + return options; +} + +function printHelp() { + console.log(` +Lime Doc Freshness Check + +用法: + node scripts/check-doc-freshness.mjs + node scripts/check-doc-freshness.mjs --json + node scripts/check-doc-freshness.mjs --format json --output "./tmp/doc-freshness.json" + +选项: + --format FMT 标准输出格式:text | json + --json 等价于 --format json + --output PATH 将结果写入指定路径 + -h, --help 显示帮助 +`); +} + +function buildDeletedSurfaceTargets(repoRoot) { + return (legacySurfaceCatalog.imports ?? []) + .flatMap((entry) => entry.targets ?? []) + .filter((target) => { + const absolutePath = path.resolve(repoRoot, target); + return !fs.existsSync(absolutePath); + }); +} + +function readMonitoredDocuments(repoRoot) { + return DOC_FRESHNESS_SPECS.flatMap((spec) => { + const absolutePath = path.resolve(repoRoot, spec.path); + if (!fs.existsSync(absolutePath)) { + return []; + } + + return [ + { + path: spec.path, + content: fs.readFileSync(absolutePath, "utf8"), + }, + ]; + }); +} + +export function buildLiveDocFreshnessReport(repoRoot) { + return buildDocFreshnessReport({ + repoRoot, + documents: readMonitoredDocuments(repoRoot), + deletedSurfaceTargets: buildDeletedSurfaceTargets(repoRoot), + pathExists: (absolutePath) => fs.existsSync(absolutePath), + }); +} + +function renderOutput(report, format) { + if (format === "json") { + return `${JSON.stringify(report, null, 2)}\n`; + } + + return renderDocFreshnessText(report); +} + +function runDocFreshnessCli() { + const options = parseArgs(process.argv.slice(2)); + if (options.help) { + printHelp(); + return; + } + + const repoRoot = process.cwd(); + const report = buildLiveDocFreshnessReport(repoRoot); + + const output = renderOutput(report, options.format); + if (options.output) { + const targetPath = path.resolve(repoRoot, options.output); + fs.mkdirSync(path.dirname(targetPath), { recursive: true }); + fs.writeFileSync(targetPath, output, "utf8"); + console.log(`[lime] doc freshness output: ${targetPath}`); + } else { + process.stdout.write(output); + } + + if (report.summary.issueCount > 0) { + process.exit(1); + } +} + +const isMainModule = + process.argv[1] && + path.resolve(process.argv[1]) === fileURLToPath(import.meta.url); + +if (isMainModule) { + runDocFreshnessCli(); +} diff --git a/scripts/governance-graph.mjs b/scripts/governance-graph.mjs index 3bf621cf5..04d4fb4f7 100644 --- a/scripts/governance-graph.mjs +++ b/scripts/governance-graph.mjs @@ -8,7 +8,7 @@ import { fileURLToPath, pathToFileURL } from "node:url"; import { cruise } from "dependency-cruiser"; import extractTSConfig from "dependency-cruiser/config-utl/extract-ts-config"; import YAML from "yaml"; -import { buildLegacySurfaceReport } from "./report-legacy-surfaces.mjs"; +import { buildLegacySurfaceReport } from "./lib/legacy-surface-report-core.mjs"; import { buildRustModuleIndex, buildRustModulePathFromFile, @@ -138,7 +138,10 @@ function parseArgs(argv) { } if (arg === "--days" && argv[index + 1]) { - result.days = normalizePositiveNumber(argv[index + 1], DEFAULT_SINCE_DAYS); + result.days = normalizePositiveNumber( + argv[index + 1], + DEFAULT_SINCE_DAYS, + ); index += 1; } } @@ -217,8 +220,10 @@ async function buildGovernanceGraphReport({ const signalsSummary = summarizeSignals(nodes); const summary = buildSummary(nodes, edges, signalsSummary); const links = { - selfHtmlHref: pathToFileURL(path.join(outputDir, "governance-graph.html")).href, - selfJsonHref: pathToFileURL(path.join(outputDir, "governance-graph.json")).href, + selfHtmlHref: pathToFileURL(path.join(outputDir, "governance-graph.html")) + .href, + selfJsonHref: pathToFileURL(path.join(outputDir, "governance-graph.json")) + .href, heatmapHtmlHref: fs.existsSync(path.join(outputDir, "index.html")) ? pathToFileURL(path.join(outputDir, "index.html")).href : "", @@ -280,7 +285,9 @@ function collectGovernanceFiles(repoRoot) { const ext = path.extname(relativePath).toLowerCase(); const language = ext === ".rs" ? "rs" : ext.replace(/^\./, ""); - const layer = relativePath.startsWith("src-tauri/src/") ? "rust" : "frontend"; + const layer = relativePath.startsWith("src-tauri/src/") + ? "rust" + : "frontend"; records.push({ path: relativePath, @@ -399,7 +406,10 @@ function collectGitChurn({ gitCommand, repoRoot, sinceDays, trackedFiles }) { continue; } - fileChurn.set(normalizedGitPath, (fileChurn.get(normalizedGitPath) || 0) + churn); + fileChurn.set( + normalizedGitPath, + (fileChurn.get(normalizedGitPath) || 0) + churn, + ); } return fileChurn; @@ -469,7 +479,9 @@ async function collectFrontendDependencies(repoRoot, fileIndex) { } function collectRustDependencies(repoRoot, fileInventory, fileIndex) { - const rustFiles = fileInventory.filter((fileRecord) => fileRecord.layer === "rust"); + const rustFiles = fileInventory.filter( + (fileRecord) => fileRecord.layer === "rust", + ); const rustFileSet = new Set(rustFiles.map((fileRecord) => fileRecord.path)); const moduleIndex = buildRustModuleIndex([...rustFileSet]); const edges = []; @@ -477,7 +489,8 @@ function collectRustDependencies(repoRoot, fileInventory, fileIndex) { for (const fileRecord of rustFiles) { const absolutePath = path.join(repoRoot, fileRecord.path); const sourceCode = fs.readFileSync(absolutePath, "utf8"); - const currentModulePath = buildRustModulePathFromFile(fileRecord.path) || ""; + const currentModulePath = + buildRustModulePathFromFile(fileRecord.path) || ""; for (const moduleName of extractRustModDeclarations(sourceCode)) { const targetPath = resolveRustSubmodulePath( @@ -661,7 +674,10 @@ function buildLegacyOverlays(legacyReport) { } } - for (const result of [...legacyReport.frontendTextResults, ...legacyReport.rustTextResults]) { + for (const result of [ + ...legacyReport.frontendTextResults, + ...legacyReport.rustTextResults, + ]) { for (const callerPath of result.references) { pushOverlay(callsiteOverlays, callerPath, { source: "legacy-report", @@ -674,7 +690,9 @@ function buildLegacyOverlays(legacyReport) { } for (const result of legacyReport.rustTextCountResults) { - for (const callerPath of result.runtimeMatches.map((item) => item.relativePath)) { + for (const callerPath of result.runtimeMatches.map( + (item) => item.relativePath, + )) { pushOverlay(callsiteOverlays, callerPath, { source: "legacy-report", overlayType: "callsite", @@ -727,7 +745,8 @@ function buildFileNodes({ legacyOverlays.callsiteOverlays.get(fileRecord.path) ?? []; const explicitStatus = matchingRule?.status ?? "unclassified"; const overlayStatus = pickOverlayStatus(surfaceOverlays); - const status = explicitStatus !== "unclassified" ? explicitStatus : overlayStatus; + const status = + explicitStatus !== "unclassified" ? explicitStatus : overlayStatus; const ruleSource = matchingRule ? { reason: matchingRule.reason || "", @@ -743,12 +762,16 @@ function buildFileNodes({ signals.add("unused-file"); } - if ((knipSignals.unusedExportsByFile.get(fileRecord.path) ?? []).length > 0) { + if ( + (knipSignals.unusedExportsByFile.get(fileRecord.path) ?? []).length > 0 + ) { signals.add("unused-export"); } if ( - surfaceOverlays.some((overlay) => overlay.classification === "dead-candidate") + surfaceOverlays.some( + (overlay) => overlay.classification === "dead-candidate", + ) ) { signals.add("dead-candidate"); } @@ -801,11 +824,7 @@ function finalizeNodeSignals(nodes, edges, governanceRules) { return nodes.map((node) => { const signals = new Set(node.signals); - if ( - !node.reachable && - node.kind === "page" && - node.layer === "frontend" - ) { + if (!node.reachable && node.kind === "page" && node.layer === "frontend") { signals.add("page-unreachable"); } @@ -861,7 +880,10 @@ function pickOverlayStatus(surfaceOverlays) { } function calculateNodeSize(loc) { - return Math.max(26, Math.min(86, Math.round(18 + Math.sqrt(Math.max(loc, 1))))); + return Math.max( + 26, + Math.min(86, Math.round(18 + Math.sqrt(Math.max(loc, 1)))), + ); } function createParentDirectoryId(relativePath) { @@ -901,7 +923,9 @@ function buildDirectoryNodes(fileNodes) { } } - return [...dirMap.values()].sort((left, right) => left.path.localeCompare(right.path)); + return [...dirMap.values()].sort((left, right) => + left.path.localeCompare(right.path), + ); } function collectReachableFiles(entryPaths, edges) { @@ -966,7 +990,10 @@ function isProtectedEntryNode(node, governanceRules) { return true; } - return resolveMatchingGovernanceRule(node.path, governanceRules)?.status === "current"; + return ( + resolveMatchingGovernanceRule(node.path, governanceRules)?.status === + "current" + ); } function calculateCandidateScore(node) { @@ -1003,16 +1030,21 @@ function summarizeSignals(nodes) { return [...counts.entries()] .map(([signal, count]) => ({ signal, count })) - .sort((left, right) => right.count - left.count || left.signal.localeCompare(right.signal)); + .sort( + (left, right) => + right.count - left.count || left.signal.localeCompare(right.signal), + ); } function buildSummary(nodes, edges, signalsSummary) { const fileNodes = nodes.filter((node) => node.kind !== "dir"); const statusCounts = Object.fromEntries( - ["current", "compat", "deprecated", "dead", "unclassified"].map((status) => [ - status, - fileNodes.filter((node) => node.status === status).length, - ]), + ["current", "compat", "deprecated", "dead", "unclassified"].map( + (status) => [ + status, + fileNodes.filter((node) => node.status === status).length, + ], + ), ); const layerCounts = Object.fromEntries( ["frontend", "rust"].map((layer) => [ diff --git a/scripts/harness-eval-history-record.mjs b/scripts/harness-eval-history-record.mjs new file mode 100644 index 000000000..879d5ac18 --- /dev/null +++ b/scripts/harness-eval-history-record.mjs @@ -0,0 +1,343 @@ +#!/usr/bin/env node + +import { execFileSync } from "node:child_process"; +import fs from "node:fs"; +import path from "node:path"; +import process from "node:process"; + +const RUNNER_PATH = "scripts/harness-eval-runner.mjs"; +const TREND_PATH = "scripts/harness-eval-trend-report.mjs"; +const CLEANUP_PATH = "scripts/report-generated-slop.mjs"; + +function parseArgs(argv) { + const result = { + cleanupJson: "", + cleanupMarkdown: "", + format: "text", + help: false, + historyDir: "./artifacts/history", + outputJson: "", + retain: 30, + skipCleanup: false, + skipTrend: false, + trendJson: "", + trendMarkdown: "", + workspaceRoot: process.cwd(), + }; + + for (let index = 0; index < argv.length; index += 1) { + const arg = argv[index]; + + if (arg === "--history-dir" && argv[index + 1]) { + result.historyDir = String(argv[index + 1]).trim(); + index += 1; + continue; + } + + if (arg === "--workspace-root" && argv[index + 1]) { + result.workspaceRoot = String(argv[index + 1]).trim(); + index += 1; + continue; + } + + if (arg === "--retain" && argv[index + 1]) { + result.retain = Number.parseInt(String(argv[index + 1]).trim(), 10); + index += 1; + continue; + } + + if (arg === "--format" && argv[index + 1]) { + result.format = String(argv[index + 1]).trim(); + index += 1; + continue; + } + + if (arg === "--output-json" && argv[index + 1]) { + result.outputJson = String(argv[index + 1]).trim(); + index += 1; + continue; + } + + if (arg === "--trend-json" && argv[index + 1]) { + result.trendJson = String(argv[index + 1]).trim(); + index += 1; + continue; + } + + if (arg === "--trend-markdown" && argv[index + 1]) { + result.trendMarkdown = String(argv[index + 1]).trim(); + index += 1; + continue; + } + + if (arg === "--cleanup-json" && argv[index + 1]) { + result.cleanupJson = String(argv[index + 1]).trim(); + index += 1; + continue; + } + + if (arg === "--cleanup-markdown" && argv[index + 1]) { + result.cleanupMarkdown = String(argv[index + 1]).trim(); + index += 1; + continue; + } + + if (arg === "--skip-trend") { + result.skipTrend = true; + continue; + } + + if (arg === "--skip-cleanup") { + result.skipCleanup = true; + continue; + } + + if (arg === "--help" || arg === "-h") { + result.help = true; + } + } + + return result; +} + +function printHelp() { + console.log(` +Lime Harness Eval History Record + +用法: + node scripts/harness-eval-history-record.mjs + node scripts/harness-eval-history-record.mjs --workspace-root "/path/to/workspace" + node scripts/harness-eval-history-record.mjs --history-dir "./artifacts/history" --output-json "./tmp/harness-history-record.json" + +选项: + --history-dir PATH summary 历史目录,默认 ./artifacts/history + --workspace-root PATH 生成当前 summary 时使用的工作区根目录 + --retain N 历史窗口保留数量,默认 30 + --trend-json PATH trend JSON 输出路径 + --trend-markdown PATH trend Markdown 输出路径 + --cleanup-json PATH cleanup JSON 输出路径 + --cleanup-markdown PATH cleanup Markdown 输出路径 + --skip-trend 只记录 summary,不生成 trend + --skip-cleanup 只记录 summary / trend,不生成 cleanup + --format FMT 标准输出格式:text | json + --output-json PATH 将记录结果写入指定路径 + -h, --help 显示帮助 +`); +} + +function resolvePath(baseDir, targetPath) { + return path.resolve(baseDir, targetPath); +} + +function ensureParentDirectory(filePath) { + fs.mkdirSync(path.dirname(filePath), { recursive: true }); +} + +function readJsonFile(filePath) { + return JSON.parse(fs.readFileSync(filePath, "utf8")); +} + +function timestampForFilename() { + return new Date().toISOString().replace(/[-:]/g, "").replace(/\./g, ""); +} + +function collectHistoryFiles(historyDir) { + if (!fs.existsSync(historyDir)) { + return []; + } + + return fs + .readdirSync(historyDir) + .filter((entry) => entry.endsWith(".json")) + .map((entry) => path.join(historyDir, entry)) + .sort((left, right) => left.localeCompare(right)); +} + +function buildCurrentSummary(repoRoot, workspaceRoot) { + const runnerPath = resolvePath(repoRoot, RUNNER_PATH); + const output = execFileSync( + process.execPath, + [runnerPath, "--format", "json", "--workspace-root", workspaceRoot], + { + cwd: repoRoot, + encoding: "utf8", + stdio: ["ignore", "pipe", "inherit"], + }, + ); + return JSON.parse(output); +} + +function writeJsonFile(filePath, payload) { + ensureParentDirectory(filePath); + fs.writeFileSync(filePath, `${JSON.stringify(payload, null, 2)}\n`, "utf8"); +} + +function trimHistoryFiles(historyDir, retain) { + const files = collectHistoryFiles(historyDir).sort((left, right) => + right.localeCompare(left), + ); + const removable = files.slice(Math.max(retain, 0)); + for (const filePath of removable) { + fs.rmSync(filePath, { force: true }); + } + return removable; +} + +function buildDefaultArtifactPaths(historyDir) { + const artifactsRoot = path.dirname(historyDir); + return { + trendJson: path.join(artifactsRoot, "harness-eval-trend.json"), + trendMarkdown: path.join(artifactsRoot, "harness-eval-trend.md"), + cleanupJson: path.join(artifactsRoot, "harness-cleanup-report.json"), + cleanupMarkdown: path.join(artifactsRoot, "harness-cleanup-report.md"), + }; +} + +function renderOutput(result, format) { + if (format === "json") { + return `${JSON.stringify(result, null, 2)}\n`; + } + + const lines = [ + "[lime] harness eval history record", + `[lime] history dir: ${result.historyDir}`, + `[lime] recorded summary: ${result.recordedSummaryPath}`, + `[lime] history count: ${result.historyCount}`, + `[lime] trimmed files: ${result.trimmedPaths.length}`, + ]; + + if (result.trend) { + lines.push(`[lime] trend sample count: ${result.trend.sampleCount}`); + if (result.trend.outputJsonPath) { + lines.push(`[lime] trend json: ${result.trend.outputJsonPath}`); + } + } + + if (result.cleanup) { + lines.push( + `[lime] cleanup trend samples: ${result.cleanup.trendSampleCount}`, + ); + if (result.cleanup.outputJsonPath) { + lines.push(`[lime] cleanup json: ${result.cleanup.outputJsonPath}`); + } + } + + return `${lines.join("\n")}\n`; +} + +function runHistoryRecordCli() { + const options = parseArgs(process.argv.slice(2)); + if (options.help) { + printHelp(); + return; + } + + const repoRoot = process.cwd(); + const historyDir = resolvePath(repoRoot, options.historyDir); + fs.mkdirSync(historyDir, { recursive: true }); + + const summary = buildCurrentSummary(repoRoot, options.workspaceRoot); + const summaryFilePath = path.join( + historyDir, + `${timestampForFilename()}-harness-eval-summary.json`, + ); + writeJsonFile(summaryFilePath, summary); + const trimmedPaths = trimHistoryFiles(historyDir, options.retain); + const historyCount = collectHistoryFiles(historyDir).length; + const defaults = buildDefaultArtifactPaths(historyDir); + + const result = { + recordedAt: new Date().toISOString(), + historyDir, + recordedSummaryPath: summaryFilePath, + historyCount, + trimmedPaths, + trend: null, + cleanup: null, + }; + + if (!options.skipTrend) { + const trendJsonPath = resolvePath( + repoRoot, + options.trendJson || defaults.trendJson, + ); + const trendMarkdownPath = resolvePath( + repoRoot, + options.trendMarkdown || defaults.trendMarkdown, + ); + const trendScriptPath = resolvePath(repoRoot, TREND_PATH); + const trendOutput = execFileSync( + process.execPath, + [ + trendScriptPath, + "--format", + "json", + "--history-dir", + historyDir, + "--output-json", + trendJsonPath, + "--output-markdown", + trendMarkdownPath, + ], + { + cwd: repoRoot, + encoding: "utf8", + stdio: ["ignore", "pipe", "inherit"], + }, + ); + const trendReport = JSON.parse(trendOutput); + result.trend = { + sampleCount: trendReport.sampleCount, + outputJsonPath: trendJsonPath, + outputMarkdownPath: trendMarkdownPath, + }; + } + + if (!options.skipCleanup) { + const cleanupJsonPath = resolvePath( + repoRoot, + options.cleanupJson || defaults.cleanupJson, + ); + const cleanupMarkdownPath = resolvePath( + repoRoot, + options.cleanupMarkdown || defaults.cleanupMarkdown, + ); + const cleanupScriptPath = resolvePath(repoRoot, CLEANUP_PATH); + const cleanupOutput = execFileSync( + process.execPath, + [ + cleanupScriptPath, + "--format", + "json", + "--trend-history-dir", + historyDir, + "--output-json", + cleanupJsonPath, + "--output-markdown", + cleanupMarkdownPath, + ], + { + cwd: repoRoot, + encoding: "utf8", + stdio: ["ignore", "pipe", "inherit"], + }, + ); + const cleanupReport = JSON.parse(cleanupOutput); + result.cleanup = { + trendSampleCount: cleanupReport.summary?.trend?.sampleCount ?? 0, + outputJsonPath: cleanupJsonPath, + outputMarkdownPath: cleanupMarkdownPath, + }; + } + + const rendered = renderOutput(result, options.format); + if (options.outputJson) { + const outputPath = resolvePath(repoRoot, options.outputJson); + writeJsonFile(outputPath, result); + console.log(`[lime] harness eval history record JSON: ${outputPath}`); + } else { + process.stdout.write(rendered); + } +} + +runHistoryRecordCli(); diff --git a/scripts/harness-eval-runner.mjs b/scripts/harness-eval-runner.mjs index 59616f7c6..ceb42b2ad 100644 --- a/scripts/harness-eval-runner.mjs +++ b/scripts/harness-eval-runner.mjs @@ -5,14 +5,33 @@ import path from "node:path"; import process from "node:process"; const DEFAULT_MANIFEST_PATH = "docs/test/harness-evals.manifest.json"; +const REVIEW_DECISION_JSON_CANDIDATES = [ + "review-decision.json", + "../review/review-decision.json", +]; +const REVIEW_DECISION_STATUS_SET = new Set([ + "accepted", + "deferred", + "rejected", + "needs_more_evidence", + "pending_review", +]); +const REVIEW_DECISION_RISK_LEVEL_SET = new Set([ + "low", + "medium", + "high", + "unknown", +]); function parseArgs(argv) { const result = { format: "text", help: false, + historyRetain: 30, manifest: DEFAULT_MANIFEST_PATH, outputJson: "", outputMarkdown: "", + recordHistoryDir: "", strict: true, workspaceRoot: process.cwd(), }; @@ -50,6 +69,18 @@ function parseArgs(argv) { continue; } + if (arg === "--record-history-dir" && argv[index + 1]) { + result.recordHistoryDir = String(argv[index + 1]).trim(); + index += 1; + continue; + } + + if (arg === "--history-retain" && argv[index + 1]) { + result.historyRetain = Number.parseInt(String(argv[index + 1]), 10); + index += 1; + continue; + } + if (arg === "--no-strict") { result.strict = false; continue; @@ -77,6 +108,7 @@ Lime Harness Eval Runner node scripts/harness-eval-runner.mjs --format json node scripts/harness-eval-runner.mjs --workspace-root "/path/to/workspace" node scripts/harness-eval-runner.mjs --output-json "./tmp/harness-eval-summary.json" --output-markdown "./tmp/harness-eval-summary.md" + node scripts/harness-eval-runner.mjs --record-history-dir "./artifacts/history" 选项: --manifest PATH 指定 manifest,默认 docs/test/harness-evals.manifest.json @@ -84,6 +116,8 @@ Lime Harness Eval Runner --format FMT 控制标准输出格式:text | json | markdown --output-json PATH 将 JSON 摘要写入指定路径 --output-markdown PATH 将 Markdown 摘要写入指定路径 + --record-history-dir PATH 将当前 summary 追加写入历史目录,供 trend/nightly 复用 + --history-retain N 历史目录最多保留多少条 summary,默认 30 --strict 严格模式(默认),发现 invalid case 时返回非 0 --no-strict 非严格模式,只输出摘要,不因 invalid case 退出失败 -h, --help 显示帮助 @@ -102,6 +136,56 @@ function ensureParentDirectory(filePath) { fs.mkdirSync(path.dirname(filePath), { recursive: true }); } +function ensureDirectory(dirPath) { + fs.mkdirSync(dirPath, { recursive: true }); +} + +function toHistoryTimestamp(generatedAt) { + const parsed = new Date(generatedAt); + const normalized = Number.isNaN(parsed.getTime()) + ? new Date().toISOString() + : parsed.toISOString(); + return normalized.replace(/[-:]/g, "").replace(/\.(\d{3})Z$/, "$1Z"); +} + +function trimHistoryDirectory(historyDir, retainCount) { + const normalizedRetainCount = + Number.isInteger(retainCount) && retainCount > 0 ? retainCount : 30; + const historyFiles = fs + .readdirSync(historyDir, { withFileTypes: true }) + .filter( + (entry) => + entry.isFile() && + entry.name.endsWith("-harness-eval-summary.json"), + ) + .map((entry) => path.join(historyDir, entry.name)) + .sort((left, right) => right.localeCompare(left)); + + for (const staleFile of historyFiles.slice(normalizedRetainCount)) { + fs.rmSync(staleFile, { force: true }); + } +} + +function recordSummaryHistory(repoRoot, summary, options) { + if (!options.recordHistoryDir) { + return ""; + } + + const historyDir = resolvePath(repoRoot, options.recordHistoryDir); + ensureDirectory(historyDir); + const historyFilePath = path.join( + historyDir, + `${toHistoryTimestamp(summary.generatedAt)}-harness-eval-summary.json`, + ); + fs.writeFileSync( + historyFilePath, + `${JSON.stringify(summary, null, 2)}\n`, + "utf8", + ); + trimHistoryDirectory(historyDir, options.historyRetain); + return historyFilePath; +} + function normalizeStringList(value) { if (!Array.isArray(value)) { return []; @@ -115,6 +199,20 @@ function mergeUniqueStrings(...groups) { return [...new Set(groups.flatMap((group) => normalizeStringList(group)))]; } +function normalizeOptionalString(value) { + return typeof value === "string" && value.trim().length > 0 + ? value.trim() + : ""; +} + +function normalizeEnumString(value, allowedValues, fallback = "") { + const normalized = normalizeOptionalString(value); + if (allowedValues.has(normalized)) { + return normalized; + } + return fallback; +} + function createBreakdownEntry(name) { return { name, @@ -208,6 +306,88 @@ function listReplayDirectories(rootPath) { .sort((left, right) => left.localeCompare(right)); } +function resolveExistingReviewDecisionPath(caseDir) { + for (const relativePath of REVIEW_DECISION_JSON_CANDIDATES) { + const candidate = path.resolve(caseDir, relativePath); + if (fs.existsSync(candidate) && fs.statSync(candidate).isFile()) { + return candidate; + } + } + return ""; +} + +function normalizeReviewDecisionPayload(payload) { + const decision = + payload && typeof payload === "object" && payload.decision + ? payload.decision + : {}; + + return { + decisionStatus: normalizeEnumString( + decision.decisionStatus ?? + decision.decision_status ?? + payload?.decisionStatus ?? + payload?.decision_status, + REVIEW_DECISION_STATUS_SET, + "pending_review", + ), + riskLevel: normalizeEnumString( + decision.riskLevel ?? + decision.risk_level ?? + payload?.riskLevel ?? + payload?.risk_level, + REVIEW_DECISION_RISK_LEVEL_SET, + "unknown", + ), + humanReviewer: normalizeOptionalString( + decision.humanReviewer ?? + decision.human_reviewer ?? + payload?.humanReviewer ?? + payload?.human_reviewer, + ), + reviewedAt: normalizeOptionalString( + decision.reviewedAt ?? + decision.reviewed_at ?? + payload?.reviewedAt ?? + payload?.reviewed_at, + ), + }; +} + +function loadReviewDecisionForCase(caseDir, inlineReviewDecision) { + const reviewDecisionPath = resolveExistingReviewDecisionPath(caseDir); + + if (reviewDecisionPath) { + try { + return { + reviewDecision: normalizeReviewDecisionPayload( + readJsonFile(reviewDecisionPath), + ), + issues: [], + }; + } catch (error) { + return { + reviewDecision: null, + issues: [ + `review-decision.json 解析失败: ${String(error.message ?? error)}`, + ], + }; + } + } + + if (inlineReviewDecision && typeof inlineReviewDecision === "object") { + return { + reviewDecision: normalizeReviewDecisionPayload(inlineReviewDecision), + issues: [], + }; + } + + return { + reviewDecision: null, + issues: [], + }; +} + function validateCaseDirectory(caseDir, caseConfig, defaults, context) { const requiredArtifacts = normalizeStringList( caseConfig.requiredArtifacts ?? defaults.requiredArtifacts, @@ -237,6 +417,7 @@ function validateCaseDirectory(caseDir, caseConfig, defaults, context) { let inputPayload = null; let expectedPayload = null; let evidencePayload = null; + let reviewDecision = null; if (fs.existsSync(files["input.json"] ?? "")) { try { @@ -281,6 +462,13 @@ function validateCaseDirectory(caseDir, caseConfig, defaults, context) { } } + const reviewDecisionResult = loadReviewDecisionForCase( + resolvedCaseDir, + caseConfig.reviewDecision, + ); + reviewDecision = reviewDecisionResult.reviewDecision; + issues.push(...reviewDecisionResult.issues); + const pendingRequestCount = Array.isArray( inputPayload?.runtimeContext?.pendingRequests, ) @@ -303,6 +491,7 @@ function validateCaseDirectory(caseDir, caseConfig, defaults, context) { typeof expectedPayload?.graderSuggestion?.preferredMode === "string" ? expectedPayload.graderSuggestion.preferredMode : ""; + const reviewDecisionRecorded = reviewDecision != null; return { caseId: context.caseId, @@ -326,6 +515,11 @@ function validateCaseDirectory(caseDir, caseConfig, defaults, context) { inputPayload?.task?.goalSummary ?? expectedPayload?.goalSummary ?? "", pendingRequestCount, requiresHumanReview, + reviewDecisionRecorded, + reviewDecisionStatus: reviewDecision?.decisionStatus ?? "", + reviewRiskLevel: reviewDecision?.riskLevel ?? "", + reviewHumanReviewer: reviewDecision?.humanReviewer ?? "", + reviewReviewedAt: reviewDecision?.reviewedAt ?? "", preferredMode, status: issues.length === 0 ? "ready" : "invalid", issues, @@ -385,6 +579,11 @@ function expandSuiteCases(suiteConfig, defaults, repoRoot, workspaceRoot) { goalSummary: "", pendingRequestCount: 0, requiresHumanReview: false, + reviewDecisionRecorded: false, + reviewDecisionStatus: "", + reviewRiskLevel: "", + reviewHumanReviewer: "", + reviewReviewedAt: "", preferredMode: "", status: "invalid", issues: [ @@ -429,6 +628,11 @@ function expandSuiteCases(suiteConfig, defaults, repoRoot, workspaceRoot) { goalSummary: "", pendingRequestCount: 0, requiresHumanReview: false, + reviewDecisionRecorded: false, + reviewDecisionStatus: "", + reviewRiskLevel: "", + reviewHumanReviewer: "", + reviewReviewedAt: "", preferredMode: "", status: "invalid", issues: [`不支持的 case source: ${source || "(empty)"}`], @@ -469,6 +673,9 @@ function buildSummary(manifest, suites, options) { const pendingCases = allCases.filter( (entry) => entry.pendingRequestCount > 0, ); + const recordedReviewDecisionCases = allCases.filter( + (entry) => entry.reviewDecisionRecorded, + ); return { manifestVersion: String(manifest.manifestVersion ?? "unknown"), @@ -484,6 +691,7 @@ function buildSummary(manifest, suites, options) { invalidCount: invalidCases.length, needsHumanReviewCount: reviewCases.length, pendingRequestCaseCount: pendingCases.length, + reviewDecisionRecordedCount: recordedReviewDecisionCases.length, }, breakdowns: { suiteTags: aggregateCaseBreakdown(allCases, (entry) => entry.tags), @@ -491,6 +699,12 @@ function buildSummary(manifest, suites, options) { allCases, (entry) => entry.failureModes, ), + reviewDecisionStatuses: aggregateCaseBreakdown(allCases, (entry) => + entry.reviewDecisionStatus ? [entry.reviewDecisionStatus] : [], + ), + reviewRiskLevels: aggregateCaseBreakdown(allCases, (entry) => + entry.reviewRiskLevel ? [entry.reviewRiskLevel] : [], + ), }, suites, }; @@ -506,6 +720,7 @@ function renderText(summary) { `[harness-eval] invalid: ${summary.totals.invalidCount}`, `[harness-eval] pending-request cases: ${summary.totals.pendingRequestCaseCount}`, `[harness-eval] needs-review cases : ${summary.totals.needsHumanReviewCount}`, + `[harness-eval] recorded review decisions: ${summary.totals.reviewDecisionRecordedCount}`, ]; const topFailureModes = summary.breakdowns.failureModes.slice(0, 5); @@ -528,6 +743,16 @@ function renderText(summary) { } } + const topReviewStatuses = summary.breakdowns.reviewDecisionStatuses.slice(0, 5); + if (topReviewStatuses.length > 0) { + lines.push("[harness-eval] review decision statuses:"); + for (const entry of topReviewStatuses) { + lines.push( + ` - ${entry.name}: case=${entry.caseCount}, ready=${entry.readyCount}, invalid=${entry.invalidCount}`, + ); + } + } + for (const suite of summary.suites) { lines.push( `[harness-eval] suite ${suite.id}: ready ${suite.stats.readyCount} / ${suite.stats.caseCount}`, @@ -542,6 +767,11 @@ function renderText(summary) { if (entry.failureModes.length > 0) { lines.push(` failure_modes: ${entry.failureModes.join(", ")}`); } + if (entry.reviewDecisionStatus) { + lines.push( + ` review: ${entry.reviewDecisionStatus} / risk=${entry.reviewRiskLevel || "unknown"}`, + ); + } for (const issue of entry.issues) { lines.push(` * ${issue}`); } @@ -564,6 +794,7 @@ function renderMarkdown(summary) { `- invalid:${summary.totals.invalidCount}`, `- pending request case:${summary.totals.pendingRequestCaseCount}`, `- needs review case:${summary.totals.needsHumanReviewCount}`, + `- 已记录人工审核:${summary.totals.reviewDecisionRecordedCount}`, "", ]; @@ -595,6 +826,32 @@ function renderMarkdown(summary) { lines.push(""); } + if (summary.breakdowns.reviewDecisionStatuses.length > 0) { + lines.push("## 人工审核状态分布"); + lines.push(""); + lines.push("| 审核状态 | case | ready | invalid |"); + lines.push("| --- | --- | --- | --- |"); + for (const entry of summary.breakdowns.reviewDecisionStatuses) { + lines.push( + `| ${entry.name} | ${entry.caseCount} | ${entry.readyCount} | ${entry.invalidCount} |`, + ); + } + lines.push(""); + } + + if (summary.breakdowns.reviewRiskLevels.length > 0) { + lines.push("## 风险等级分布"); + lines.push(""); + lines.push("| 风险等级 | case | ready | invalid |"); + lines.push("| --- | --- | --- | --- |"); + for (const entry of summary.breakdowns.reviewRiskLevels) { + lines.push( + `| ${entry.name} | ${entry.caseCount} | ${entry.readyCount} | ${entry.invalidCount} |`, + ); + } + lines.push(""); + } + for (const suite of summary.suites) { lines.push(`## ${suite.title}`); lines.push(""); @@ -613,8 +870,8 @@ function renderMarkdown(summary) { `- ready / total:${suite.stats.readyCount} / ${suite.stats.caseCount}`, ); lines.push(""); - lines.push("| Case | 状态 | 来源 | 分类 | 目录 | 问题 |"); - lines.push("| --- | --- | --- | --- | --- | --- |"); + lines.push("| Case | 状态 | 来源 | 分类 | 审核 | 目录 | 问题 |"); + lines.push("| --- | --- | --- | --- | --- | --- | --- |"); for (const entry of suite.cases) { const issueText = entry.issues.length === 0 ? "无" : entry.issues.join("
"); @@ -628,8 +885,19 @@ function renderMarkdown(summary) { if (entry.primaryBlockingKind) { classificationText.push(`blocking: ${entry.primaryBlockingKind}`); } + const reviewText = entry.reviewDecisionStatus + ? [ + entry.reviewDecisionStatus, + entry.reviewRiskLevel ? `risk: ${entry.reviewRiskLevel}` : "", + entry.reviewHumanReviewer + ? `by: ${entry.reviewHumanReviewer}` + : "", + ] + .filter(Boolean) + .join("
") + : "无"; lines.push( - `| ${entry.caseId} | ${entry.status} | ${entry.source} | ${classificationText.join("
") || "无"} | \`${entry.relativeCaseDir || "."}\` | ${issueText} |`, + `| ${entry.caseId} | ${entry.status} | ${entry.source} | ${classificationText.join("
") || "无"} | ${reviewText} | \`${entry.relativeCaseDir || "."}\` | ${issueText} |`, ); } lines.push(""); @@ -670,6 +938,7 @@ function main() { const jsonOutput = `${JSON.stringify(summary, null, 2)}\n`; const markdownOutput = renderMarkdown(summary); const textOutput = renderText(summary); + const historyFilePath = recordSummaryHistory(repoRoot, summary, options); if (options.outputJson) { const outputPath = resolvePath(repoRoot, options.outputJson); @@ -689,6 +958,11 @@ function main() { process.stdout.write(markdownOutput); } else { process.stdout.write(textOutput); + if (historyFilePath) { + process.stdout.write( + `[harness-eval] history snapshot: ${historyFilePath}\n`, + ); + } } const exitCode = determineExitCode(summary, options); diff --git a/scripts/harness-eval-trend-report.mjs b/scripts/harness-eval-trend-report.mjs index 60a1c5b6d..0b4b4307f 100644 --- a/scripts/harness-eval-trend-report.mjs +++ b/scripts/harness-eval-trend-report.mjs @@ -13,6 +13,7 @@ function parseArgs(argv) { help: false, historyDir: "", inputs: [], + manifest: "", outputJson: "", outputMarkdown: "", workspaceRoot: process.cwd(), @@ -27,6 +28,12 @@ function parseArgs(argv) { continue; } + if (arg === "--manifest" && argv[index + 1]) { + result.manifest = String(argv[index + 1]).trim(); + index += 1; + continue; + } + if (arg === "--history-dir" && argv[index + 1]) { result.historyDir = String(argv[index + 1]).trim(); index += 1; @@ -71,11 +78,13 @@ Lime Harness Eval Trend Report 用法: node scripts/harness-eval-trend-report.mjs + node scripts/harness-eval-trend-report.mjs --manifest "./tmp/harness-evals.manifest.json" node scripts/harness-eval-trend-report.mjs --input "./tmp/harness-eval-summary.json" node scripts/harness-eval-trend-report.mjs --history-dir "./artifacts/history" node scripts/harness-eval-trend-report.mjs --output-json "./tmp/harness-eval-trend.json" --output-markdown "./tmp/harness-eval-trend.md" 选项: + --manifest PATH 透传给 harness eval runner,覆盖默认 manifest --input PATH 显式加入一个或多个 harness eval summary JSON --history-dir PATH 扫描目录下的历史 summary JSON --workspace-root PATH 未提供输入时,用该工作区生成当前 summary @@ -138,12 +147,16 @@ function isHarnessEvalSummary(candidate) { ); } -function buildCurrentSummary(repoRoot, workspaceRoot) { +function buildCurrentSummary(repoRoot, workspaceRoot, manifestPath) { const nodeCommand = process.execPath; const runnerPath = resolvePath(repoRoot, RUNNER_PATH); + const args = [runnerPath, "--format", "json", "--workspace-root", workspaceRoot]; + if (manifestPath) { + args.push("--manifest", manifestPath); + } const output = execFileSync( nodeCommand, - [runnerPath, "--format", "json", "--workspace-root", workspaceRoot], + args, { cwd: repoRoot, encoding: "utf8", @@ -403,6 +416,9 @@ function buildTrendReport(samples, repoRoot) { needsHumanReviewCount: normalizeNumber(latest?.totals?.needsHumanReviewCount) - normalizeNumber(baseline?.totals?.needsHumanReviewCount), + reviewDecisionRecordedCount: + normalizeNumber(latest?.totals?.reviewDecisionRecordedCount) - + normalizeNumber(baseline?.totals?.reviewDecisionRecordedCount), readyRate: readyRateDelta, }, signals: buildStatusSignals(baseline, latest, sortedSamples.length), @@ -415,6 +431,16 @@ function buildTrendReport(samples, repoRoot) { classificationDeltas: { suiteTags: buildBreakdownDeltas(baseline, latest, "suiteTags"), failureModes: buildBreakdownDeltas(baseline, latest, "failureModes"), + reviewDecisionStatuses: buildBreakdownDeltas( + baseline, + latest, + "reviewDecisionStatuses", + ), + reviewRiskLevels: buildBreakdownDeltas( + baseline, + latest, + "reviewRiskLevels", + ), }, }; } @@ -428,6 +454,7 @@ function renderText(report) { `[harness-eval-trend] delta readyCount: ${report.delta.readyCount}`, `[harness-eval-trend] delta invalidCount: ${report.delta.invalidCount}`, `[harness-eval-trend] delta pendingRequestCaseCount: ${report.delta.pendingRequestCaseCount}`, + `[harness-eval-trend] delta reviewDecisionRecordedCount: ${report.delta.reviewDecisionRecordedCount}`, `[harness-eval-trend] delta readyRate: ${(report.delta.readyRate * 100).toFixed(1)}%`, ]; @@ -448,6 +475,17 @@ function renderText(report) { } } + const topReviewDecisionDeltas = + report.classificationDeltas.reviewDecisionStatuses.slice(0, 5); + if (topReviewDecisionDeltas.length > 0) { + lines.push("[harness-eval-trend] review decision deltas:"); + for (const entry of topReviewDecisionDeltas) { + lines.push( + ` - ${entry.name}: delta_case=${entry.delta.caseCount}, delta_invalid=${entry.delta.invalidCount}`, + ); + } + } + return `${lines.join("\n")}\n`; } @@ -468,6 +506,7 @@ function renderMarkdown(report) { `- invalid 数变化:${report.delta.invalidCount}`, `- pending request case 变化:${report.delta.pendingRequestCaseCount}`, `- needs review case 变化:${report.delta.needsHumanReviewCount}`, + `- 已记录人工审核变化:${report.delta.reviewDecisionRecordedCount}`, `- ready rate 变化:${(report.delta.readyRate * 100).toFixed(1)}%`, "", "## 信号", @@ -508,6 +547,32 @@ function renderMarkdown(report) { } } + if (report.classificationDeltas.reviewDecisionStatuses.length > 0) { + lines.push(""); + lines.push("## 人工审核状态变化"); + lines.push(""); + lines.push("| 审核状态 | baseline case | latest case | delta case | delta invalid |"); + lines.push("| --- | --- | --- | --- | --- |"); + for (const entry of report.classificationDeltas.reviewDecisionStatuses) { + lines.push( + `| ${entry.name} | ${entry.baseline.caseCount} | ${entry.latest.caseCount} | ${entry.delta.caseCount} | ${entry.delta.invalidCount} |`, + ); + } + } + + if (report.classificationDeltas.reviewRiskLevels.length > 0) { + lines.push(""); + lines.push("## 风险等级变化"); + lines.push(""); + lines.push("| 风险等级 | baseline case | latest case | delta case | delta invalid |"); + lines.push("| --- | --- | --- | --- | --- |"); + for (const entry of report.classificationDeltas.reviewRiskLevels) { + lines.push( + `| ${entry.name} | ${entry.baseline.caseCount} | ${entry.latest.caseCount} | ${entry.delta.caseCount} | ${entry.delta.invalidCount} |`, + ); + } + } + lines.push(""); lines.push("## 时间线样本"); lines.push(""); @@ -587,6 +652,7 @@ function loadSamples(options, repoRoot) { const currentSummary = buildCurrentSummary( repoRoot, path.resolve(options.workspaceRoot), + options.manifest, ); sampleEntries.push({ sourcePath: "(generated-current-summary)", diff --git a/scripts/harness-replay-promote.mjs b/scripts/harness-replay-promote.mjs index 664bf0124..813f1a2f8 100644 --- a/scripts/harness-replay-promote.mjs +++ b/scripts/harness-replay-promote.mjs @@ -8,12 +8,27 @@ const DEFAULT_MANIFEST_PATH = "docs/test/harness-evals.manifest.json"; const DEFAULT_FIXTURES_ROOT = "docs/test/harness-fixtures/replay"; const DEFAULT_SUITE_ID = "repo-promoted-replays"; const DEFAULT_SANITIZED_WORKSPACE_ROOT = "/workspace/lime"; +const REVIEW_DECISION_JSON_FILE_NAME = "review-decision.json"; +const REVIEW_DECISION_MARKDOWN_FILE_NAME = "review-decision.md"; const REQUIRED_ARTIFACTS = [ "input.json", "expected.json", "grader.md", "evidence-links.json", ]; +const REVIEW_DECISION_STATUS_SET = new Set([ + "accepted", + "deferred", + "rejected", + "needs_more_evidence", + "pending_review", +]); +const REVIEW_DECISION_RISK_LEVEL_SET = new Set([ + "low", + "medium", + "high", + "unknown", +]); function parseArgs(argv) { const result = { @@ -150,6 +165,10 @@ function readJsonFile(filePath) { return JSON.parse(fs.readFileSync(filePath, "utf8")); } +function readTextFile(filePath) { + return fs.readFileSync(filePath, "utf8"); +} + function writeJsonFile(filePath, value) { fs.writeFileSync(filePath, `${JSON.stringify(value, null, 2)}\n`, "utf8"); } @@ -180,6 +199,20 @@ function mergeUniqueStrings(...groups) { return [...new Set(groups.flatMap((group) => normalizeStringList(group)))]; } +function normalizeEnumString(value, allowedValues, fallback = "") { + const normalized = typeof value === "string" ? value.trim() : ""; + if (allowedValues.has(normalized)) { + return normalized; + } + return fallback; +} + +function normalizeOptionalString(value) { + return typeof value === "string" && value.trim().length > 0 + ? value.trim() + : ""; +} + function slugify(value) { return String(value) .trim() @@ -322,6 +355,7 @@ function getRelativeIfInside(rootPath, absolutePath) { function buildPromotionMetadata({ promotedAt, replayDir, + reviewDecision, sessionId, workspaceRoot, sanitizedWorkspaceRoot, @@ -338,6 +372,10 @@ function buildPromotionMetadata({ metadata.sourceReplayDir = replayRelativeDir; } + if (reviewDecision) { + metadata.reviewDecision = reviewDecision; + } + return metadata; } @@ -382,9 +420,10 @@ function buildManifestCaseEntry({ caseId, caseTitle, inputPayload, + reviewDecision, targetCaseDirValue, }) { - return { + const entry = { id: caseId, title: caseTitle, source: "repo_fixture", @@ -394,6 +433,12 @@ function buildManifestCaseEntry({ inputPayload?.classification?.suiteTags, ), }; + + if (reviewDecision) { + entry.reviewDecision = reviewDecision; + } + + return entry; } function updateManifestCase({ @@ -448,6 +493,8 @@ function writePromotedArtifacts({ expectedPayload, graderMarkdown, inputPayload, + reviewDecisionJson, + reviewDecisionMarkdown, targetDir, }) { ensureDirectory(targetDir); @@ -455,6 +502,19 @@ function writePromotedArtifacts({ writeJsonFile(path.join(targetDir, "expected.json"), expectedPayload); writeJsonFile(path.join(targetDir, "evidence-links.json"), evidencePayload); fs.writeFileSync(path.join(targetDir, "grader.md"), graderMarkdown, "utf8"); + if (reviewDecisionJson) { + writeJsonFile( + path.join(targetDir, REVIEW_DECISION_JSON_FILE_NAME), + reviewDecisionJson, + ); + } + if (reviewDecisionMarkdown) { + fs.writeFileSync( + path.join(targetDir, REVIEW_DECISION_MARKDOWN_FILE_NAME), + reviewDecisionMarkdown, + "utf8", + ); + } } function toManifestCaseDirValue(repoRoot, targetDir) { @@ -485,9 +545,100 @@ function renderText(result) { lines.push(`[harness-replay-promote] tags: ${result.tags.join(", ")}`); } + if (result.reviewDecisionStatus) { + lines.push( + `[harness-replay-promote] review decision: ${result.reviewDecisionStatus} / risk=${result.reviewRiskLevel || "unknown"}`, + ); + } + return `${lines.join("\n")}\n`; } +function resolveSiblingReviewArtifactPath(replayDir, fileName) { + return path.resolve(replayDir, "..", "review", fileName); +} + +function normalizeReviewDecisionMetadata(reviewDecisionPayload) { + const decision = + reviewDecisionPayload && + typeof reviewDecisionPayload === "object" && + reviewDecisionPayload.decision && + typeof reviewDecisionPayload.decision === "object" + ? reviewDecisionPayload.decision + : {}; + const decisionStatus = normalizeEnumString( + decision.decisionStatus ?? decision.decision_status, + REVIEW_DECISION_STATUS_SET, + "pending_review", + ); + const riskLevel = normalizeEnumString( + decision.riskLevel ?? decision.risk_level, + REVIEW_DECISION_RISK_LEVEL_SET, + "unknown", + ); + const humanReviewer = normalizeOptionalString( + decision.humanReviewer ?? decision.human_reviewer, + ); + const reviewedAt = normalizeOptionalString( + decision.reviewedAt ?? decision.reviewed_at, + ); + + return { + decisionStatus, + riskLevel, + humanReviewer, + reviewedAt, + }; +} + +function loadOptionalReviewDecisionArtifacts({ + replayDir, + sanitizedWorkspaceRoot, + workspaceRoot, +}) { + const reviewDecisionJsonPath = resolveSiblingReviewArtifactPath( + replayDir, + REVIEW_DECISION_JSON_FILE_NAME, + ); + const reviewDecisionMarkdownPath = resolveSiblingReviewArtifactPath( + replayDir, + REVIEW_DECISION_MARKDOWN_FILE_NAME, + ); + const hasJson = fs.existsSync(reviewDecisionJsonPath); + const hasMarkdown = fs.existsSync(reviewDecisionMarkdownPath); + + if (!hasJson && !hasMarkdown) { + return { + metadata: null, + reviewDecisionJson: null, + reviewDecisionMarkdown: "", + }; + } + + const reviewDecisionJson = hasJson + ? sanitizePayload( + readJsonFile(reviewDecisionJsonPath), + workspaceRoot, + sanitizedWorkspaceRoot, + ) + : null; + const reviewDecisionMarkdown = hasMarkdown + ? replaceWorkspaceRootInString( + readTextFile(reviewDecisionMarkdownPath), + workspaceRoot, + sanitizedWorkspaceRoot, + ) + : ""; + + return { + metadata: reviewDecisionJson + ? normalizeReviewDecisionMetadata(reviewDecisionJson) + : null, + reviewDecisionJson, + reviewDecisionMarkdown, + }; +} + function main() { const options = parseArgs(process.argv.slice(2)); if (options.help) { @@ -531,6 +682,7 @@ function main() { const promotionMetadata = buildPromotionMetadata({ promotedAt, replayDir, + reviewDecision: undefined, sanitizedWorkspaceRoot: options.sanitizedWorkspaceRoot, sessionId, workspaceRoot, @@ -570,6 +722,18 @@ function main() { ), promotionMetadata, ); + const { + metadata: reviewDecisionMetadata, + reviewDecisionJson, + reviewDecisionMarkdown, + } = loadOptionalReviewDecisionArtifacts({ + replayDir, + sanitizedWorkspaceRoot: options.sanitizedWorkspaceRoot, + workspaceRoot, + }); + if (reviewDecisionMetadata) { + promotionMetadata.reviewDecision = reviewDecisionMetadata; + } const fixturesRoot = resolvePath(repoRoot, options.fixturesRoot); const targetDir = path.join(fixturesRoot, slug); @@ -584,6 +748,7 @@ function main() { caseId, caseTitle: title, inputPayload, + reviewDecision: reviewDecisionMetadata, targetCaseDirValue: manifestCaseDir, }); @@ -597,6 +762,8 @@ function main() { expectedPayload, graderMarkdown, inputPayload, + reviewDecisionJson, + reviewDecisionMarkdown, targetDir, }); const manifestUpdate = updateManifestCase({ @@ -623,6 +790,10 @@ function main() { targetCaseDir: toPortablePath(targetDir), title, }; + if (reviewDecisionMetadata) { + result.reviewDecisionStatus = reviewDecisionMetadata.decisionStatus; + result.reviewRiskLevel = reviewDecisionMetadata.riskLevel; + } if (options.format === "json") { process.stdout.write(`${JSON.stringify(result, null, 2)}\n`); diff --git a/scripts/lib/doc-freshness-core.mjs b/scripts/lib/doc-freshness-core.mjs new file mode 100644 index 000000000..b047c1466 --- /dev/null +++ b/scripts/lib/doc-freshness-core.mjs @@ -0,0 +1,380 @@ +import path from "node:path"; + +export const DOC_FRESHNESS_SPECS = [ + { + path: "docs/tech/harness/README.md", + requiredMentions: [ + "iteration-roadmap.md", + "review-decision-workflow.md", + "tooling-roadmap.md", + "entropy-governance-workflow.md", + ], + }, + { + path: "docs/tech/harness/iteration-roadmap.md", + requiredMentions: [ + "review-decision-workflow.md", + "tooling-roadmap.md", + "entropy-governance-workflow.md", + "scripts/report-generated-slop.mjs", + "scripts/check-doc-freshness.mjs", + ], + }, + { + path: "docs/tech/harness/tooling-roadmap.md", + requiredMentions: [ + "harness-evals.md", + "scripts/report-generated-slop.mjs", + "scripts/check-doc-freshness.mjs", + ], + }, + { + path: "docs/tech/harness/entropy-governance-workflow.md", + requiredMentions: [ + "iteration-roadmap.md", + "tooling-roadmap.md", + "harness-evals.md", + "scripts/report-generated-slop.mjs", + "scripts/check-doc-freshness.mjs", + ], + }, + { + path: "docs/tech/harness/review-decision-workflow.md", + requiredMentions: [ + "external-analysis-handoff.md", + "iteration-roadmap.md", + ], + }, + { + path: "docs/tech/harness/external-analysis-handoff.md", + requiredMentions: [ + "iteration-roadmap.md", + "review-decision-workflow.md", + "tooling-roadmap.md", + ], + }, + { + path: "docs/tech/harness/implementation-blueprint.md", + requiredMentions: [ + "harness-evals.md", + "scripts/report-generated-slop.mjs", + ], + }, + { + path: "docs/test/harness-evals.md", + requiredMentions: [ + "tooling-roadmap.md", + "entropy-governance-workflow.md", + "scripts/report-generated-slop.mjs", + "scripts/check-doc-freshness.mjs", + ], + }, + { + path: "docs/aiprompts/governance.md", + requiredMentions: [], + }, + { + path: "docs/aiprompts/quality-workflow.md", + requiredMentions: [], + }, +]; + +const LOCAL_LINK_PATTERN = /\[[^\]]+\]\(([^)]+)\)/g; +const REPO_PATH_TOKEN_PATTERN = + /(^|[\s("'`])((?:docs|scripts|src|src-tauri|\.github)\/[A-Za-z0-9._/-]+(?:\.[A-Za-z0-9]+)?)/gm; +const REPO_ROOT_RELATIVE_PREFIXES = [ + "docs/", + "scripts/", + "src/", + "src-tauri/", + ".github/", +]; + +function normalizePath(filePath) { + return filePath.split(path.sep).join("/"); +} + +function stripAnchor(target) { + const anchorIndex = target.indexOf("#"); + return anchorIndex >= 0 ? target.slice(0, anchorIndex) : target; +} + +function cleanLinkTarget(target) { + const trimmed = String(target ?? "").trim(); + if ( + trimmed.length === 0 || + trimmed.startsWith("http://") || + trimmed.startsWith("https://") || + trimmed.startsWith("mailto:") || + trimmed.startsWith("#") || + trimmed.startsWith("vscode:") || + trimmed.startsWith("file://") + ) { + return ""; + } + + const withoutAngleBrackets = + trimmed.startsWith("<") && trimmed.endsWith(">") + ? trimmed.slice(1, -1).trim() + : trimmed; + const withoutTitle = withoutAngleBrackets.split(/\s+"/, 1)[0]; + const withoutAnchor = stripAnchor(withoutTitle); + + return withoutAnchor.includes("<") ? "" : withoutAnchor; +} + +function resolveRepoPath({ repoRoot, documentPath, rawPath }) { + const cleanedPath = cleanLinkTarget(rawPath); + if (!cleanedPath) { + return null; + } + + const normalizedRepoRoot = path.resolve(repoRoot); + let absolutePath = ""; + + if (path.isAbsolute(cleanedPath)) { + const normalizedAbsolute = path.resolve(cleanedPath); + if (!normalizedAbsolute.startsWith(normalizedRepoRoot)) { + return null; + } + absolutePath = normalizedAbsolute; + } else if ( + REPO_ROOT_RELATIVE_PREFIXES.some((prefix) => cleanedPath.startsWith(prefix)) + ) { + absolutePath = path.resolve(normalizedRepoRoot, cleanedPath); + } else { + absolutePath = path.resolve( + normalizedRepoRoot, + path.dirname(documentPath), + cleanedPath, + ); + } + + const repoRelativePath = normalizePath( + path.relative(normalizedRepoRoot, absolutePath), + ); + if (repoRelativePath.startsWith("..")) { + return null; + } + + return { + absolutePath, + repoRelativePath, + }; +} + +function collectLocalLinks(content) { + const results = []; + for (const match of content.matchAll(LOCAL_LINK_PATTERN)) { + const rawTarget = match[1]; + if (!rawTarget) { + continue; + } + results.push(rawTarget); + } + return [...new Set(results)]; +} + +function collectRepoPathTokens(content) { + const results = []; + for (const match of content.matchAll(REPO_PATH_TOKEN_PATTERN)) { + const rawTarget = match[2]; + if (!rawTarget) { + continue; + } + results.push(rawTarget.trim()); + } + return [...new Set(results)]; +} + +function createIssue(kind, documentPath, detail) { + return { + kind, + documentPath, + detail, + }; +} + +export function buildDocFreshnessReport({ + repoRoot, + documents, + deletedSurfaceTargets = [], + pathExists, + specs = DOC_FRESHNESS_SPECS, +}) { + const documentsByPath = new Map( + (Array.isArray(documents) ? documents : []).map((entry) => [ + normalizePath(entry.path), + String(entry.content ?? ""), + ]), + ); + const issues = []; + const documentReports = []; + + for (const spec of specs) { + const documentPath = normalizePath(spec.path); + const content = documentsByPath.get(documentPath); + + if (content == null) { + issues.push(createIssue("missing-document", documentPath, documentPath)); + documentReports.push({ + path: documentPath, + exists: false, + requiredMentions: [], + localLinks: [], + codePathMentions: [], + deletedSurfaceReferences: [], + }); + continue; + } + + const requiredMentions = (spec.requiredMentions ?? []).map((needle) => ({ + needle, + found: content.includes(needle), + })); + for (const entry of requiredMentions) { + if (!entry.found) { + issues.push( + createIssue("missing-required-reference", documentPath, entry.needle), + ); + } + } + + const localLinks = collectLocalLinks(content).map((rawTarget) => { + const resolved = resolveRepoPath({ + repoRoot, + documentPath, + rawPath: rawTarget, + }); + if (!resolved) { + return { + rawTarget, + repoRelativePath: "", + exists: true, + }; + } + + const exists = pathExists(resolved.absolutePath, resolved.repoRelativePath); + if (!exists) { + issues.push( + createIssue( + "broken-markdown-link", + documentPath, + `${rawTarget} -> ${resolved.repoRelativePath}`, + ), + ); + } + + return { + rawTarget, + repoRelativePath: resolved.repoRelativePath, + exists, + }; + }); + + const codePathMentions = collectRepoPathTokens(content).map((rawTarget) => { + const resolved = resolveRepoPath({ + repoRoot, + documentPath, + rawPath: rawTarget, + }); + if (!resolved) { + return { + rawTarget, + repoRelativePath: "", + exists: true, + }; + } + + const exists = pathExists(resolved.absolutePath, resolved.repoRelativePath); + if (!exists) { + issues.push( + createIssue( + "broken-code-path-reference", + documentPath, + `${rawTarget} -> ${resolved.repoRelativePath}`, + ), + ); + } + + return { + rawTarget, + repoRelativePath: resolved.repoRelativePath, + exists, + }; + }); + + const deletedSurfaceReferences = [...new Set(deletedSurfaceTargets)] + .filter(Boolean) + .filter((target) => content.includes(target)); + for (const target of deletedSurfaceReferences) { + issues.push( + createIssue("deleted-surface-reference", documentPath, target), + ); + } + + documentReports.push({ + path: documentPath, + exists: true, + requiredMentions, + localLinks, + codePathMentions, + deletedSurfaceReferences, + }); + } + + const summary = { + monitoredDocumentCount: specs.length, + existingDocumentCount: documentReports.filter((entry) => entry.exists).length, + issueCount: issues.length, + missingDocumentCount: issues.filter((entry) => entry.kind === "missing-document") + .length, + missingRequiredReferenceCount: issues.filter( + (entry) => entry.kind === "missing-required-reference", + ).length, + brokenMarkdownLinkCount: issues.filter( + (entry) => entry.kind === "broken-markdown-link", + ).length, + brokenCodePathReferenceCount: issues.filter( + (entry) => entry.kind === "broken-code-path-reference", + ).length, + deletedSurfaceReferenceCount: issues.filter( + (entry) => entry.kind === "deleted-surface-reference", + ).length, + }; + + return { + reportVersion: "v1", + generatedAt: new Date().toISOString(), + repoRoot: path.resolve(repoRoot), + summary, + documents: documentReports, + issues, + }; +} + +export function renderDocFreshnessText(report) { + const lines = [ + "[lime] doc freshness report", + `[lime] monitored docs: ${report.summary.monitoredDocumentCount}`, + `[lime] existing docs: ${report.summary.existingDocumentCount}`, + `[lime] issues: ${report.summary.issueCount}`, + `[lime] missing docs: ${report.summary.missingDocumentCount}`, + `[lime] missing required refs: ${report.summary.missingRequiredReferenceCount}`, + `[lime] broken markdown links: ${report.summary.brokenMarkdownLinkCount}`, + `[lime] broken code path refs: ${report.summary.brokenCodePathReferenceCount}`, + `[lime] deleted surface refs: ${report.summary.deletedSurfaceReferenceCount}`, + ]; + + if (report.issues.length === 0) { + lines.push("[lime] doc freshness: clean"); + return `${lines.join("\n")}\n`; + } + + lines.push("[lime] issues:"); + for (const issue of report.issues) { + lines.push(` - [${issue.kind}] ${issue.documentPath} -> ${issue.detail}`); + } + + return `${lines.join("\n")}\n`; +} diff --git a/scripts/lib/doc-freshness-core.test.ts b/scripts/lib/doc-freshness-core.test.ts new file mode 100644 index 000000000..049335b6d --- /dev/null +++ b/scripts/lib/doc-freshness-core.test.ts @@ -0,0 +1,86 @@ +import { describe, expect, it } from "vitest"; + +import { buildDocFreshnessReport } from "./doc-freshness-core.mjs"; + +describe("doc-freshness-core", () => { + it("应在回链与路径都正常时返回 clean 报告", () => { + const report = buildDocFreshnessReport({ + repoRoot: "/tmp/lime", + specs: [ + { + path: "docs/tech/harness/entropy-governance-workflow.md", + requiredMentions: [ + "iteration-roadmap.md", + "tooling-roadmap.md", + "harness-evals.md", + "scripts/report-generated-slop.mjs", + "scripts/check-doc-freshness.mjs", + ], + }, + ], + documents: [ + { + path: "docs/tech/harness/entropy-governance-workflow.md", + content: ` +[Roadmap](iteration-roadmap.md) +[Tooling](tooling-roadmap.md) +[Evals](../../test/harness-evals.md) +\`scripts/report-generated-slop.mjs\` +\`scripts/check-doc-freshness.mjs\` +`, + }, + ], + deletedSurfaceTargets: ["src/lib/api/agentCompat.ts"], + pathExists: (_absolutePath, repoRelativePath) => + [ + "docs/tech/harness/iteration-roadmap.md", + "docs/tech/harness/tooling-roadmap.md", + "docs/test/harness-evals.md", + "scripts/report-generated-slop.mjs", + "scripts/check-doc-freshness.mjs", + ].includes(repoRelativePath), + }); + + expect(report.summary.issueCount).toBe(0); + expect(report.documents[0].requiredMentions.every((entry) => entry.found)).toBe( + true, + ); + }); + + it("应识别缺失回链、坏链接、坏路径与已删除表面引用", () => { + const report = buildDocFreshnessReport({ + repoRoot: "/tmp/lime", + specs: [ + { + path: "docs/tech/harness/review-decision-workflow.md", + requiredMentions: [ + "external-analysis-handoff.md", + "iteration-roadmap.md", + ], + }, + ], + documents: [ + { + path: "docs/tech/harness/review-decision-workflow.md", + content: ` +[Bad Link](missing-doc.md) +\`scripts/missing-tool.mjs\` +旧入口:src/lib/api/agentCompat.ts +`, + }, + ], + deletedSurfaceTargets: ["src/lib/api/agentCompat.ts"], + pathExists: () => false, + }); + + expect(report.summary.issueCount).toBe(5); + expect(report.issues.map((entry) => entry.kind)).toEqual( + expect.arrayContaining([ + "missing-required-reference", + "broken-markdown-link", + "broken-code-path-reference", + "deleted-surface-reference", + ]), + ); + }); +}); diff --git a/scripts/lib/generated-slop-report-core.mjs b/scripts/lib/generated-slop-report-core.mjs new file mode 100644 index 000000000..846cb68ee --- /dev/null +++ b/scripts/lib/generated-slop-report-core.mjs @@ -0,0 +1,1004 @@ +import { + getCommandStatus, + getImportStatus, + getTextCountStatus, + getTextStatus, +} from "./legacy-surface-report-summary.mjs"; + +const PRIORITY_RANK = { + P0: 0, + P1: 1, + P2: 2, + P3: 3, +}; + +const CLASSIFICATION_WEIGHT = { + current: 0, + compat: 25, + deprecated: 35, + "dead-candidate": 45, +}; + +const STATUS_WEIGHT = { + "违规": 200, + "受控": 40, + "零引用": 0, + "已删除": 0, +}; + +function normalizeNumber(value) { + return typeof value === "number" && Number.isFinite(value) ? value : 0; +} + +function normalizeString(value, fallback = "") { + return typeof value === "string" && value.trim().length > 0 + ? value.trim() + : fallback; +} + +function isObject(value) { + return value != null && typeof value === "object"; +} + +function flattenCommandReferenceGroups(referenceGroups) { + if (referenceGroups instanceof Map) { + return [...referenceGroups.values()].flat(); + } + + if (isObject(referenceGroups)) { + return Object.values(referenceGroups).flatMap((entry) => + Array.isArray(entry) ? entry : [], + ); + } + + return []; +} + +function toReferenceGroupMap(referenceGroups) { + if (referenceGroups instanceof Map) { + return referenceGroups; + } + + if (isObject(referenceGroups)) { + return new Map( + Object.entries(referenceGroups).map(([key, value]) => [ + key, + Array.isArray(value) ? value : [], + ]), + ); + } + + return new Map(); +} + +function getTextCountOccurrenceCount(result) { + const runtimeMatches = Array.isArray(result?.runtimeMatches) + ? result.runtimeMatches + : []; + + return runtimeMatches.reduce( + (total, entry) => + total + + (Array.isArray(entry?.counts) + ? entry.counts.reduce( + (innerTotal, item) => innerTotal + normalizeNumber(item?.count), + 0, + ) + : 0), + 0, + ); +} + +function getClassificationWeight(classification) { + return CLASSIFICATION_WEIGHT[classification] ?? 10; +} + +function getStatusWeight(status) { + return STATUS_WEIGHT[status] ?? 0; +} + +function buildGovernanceSurfaceEntry({ + sourceType, + result, + status, + referenceCount, + testReferenceCount, + occurrenceCount, +}) { + const classification = normalizeString(result?.classification, "unknown"); + const violationCount = Array.isArray(result?.violations) + ? result.violations.length + : 0; + const active = status !== "零引用" && status !== "已删除"; + const score = active + ? getStatusWeight(status) + + getClassificationWeight(classification) + + violationCount * 200 + + referenceCount * 5 + + occurrenceCount + : 0; + + return { + id: normalizeString(result?.id, `${sourceType}-surface`), + sourceType, + classification, + description: normalizeString(result?.description, "(无描述)"), + status, + active, + referenceCount, + testReferenceCount, + occurrenceCount, + violationCount, + score, + }; +} + +function buildGovernanceSurfaceEntries(governanceReport) { + const importResults = Array.isArray(governanceReport?.importResults) + ? governanceReport.importResults + : []; + const commandResults = Array.isArray(governanceReport?.commandResults) + ? governanceReport.commandResults + : []; + const frontendTextResults = Array.isArray(governanceReport?.frontendTextResults) + ? governanceReport.frontendTextResults + : []; + const rustTextResults = Array.isArray(governanceReport?.rustTextResults) + ? governanceReport.rustTextResults + : []; + const rustTextCountResults = Array.isArray( + governanceReport?.rustTextCountResults, + ) + ? governanceReport.rustTextCountResults + : []; + + const importEntries = importResults.map((result) => + buildGovernanceSurfaceEntry({ + sourceType: "import", + result, + status: getImportStatus(result), + referenceCount: Array.isArray(result?.references) ? result.references.length : 0, + testReferenceCount: Array.isArray(result?.testReferences) + ? result.testReferences.length + : 0, + occurrenceCount: 0, + }), + ); + + const commandEntries = commandResults.map((result) => + buildGovernanceSurfaceEntry({ + sourceType: "command", + result, + status: getCommandStatus({ + ...result, + referencesByCommand: toReferenceGroupMap(result?.referencesByCommand), + }), + referenceCount: [ + ...new Set(flattenCommandReferenceGroups(result?.referencesByCommand)), + ].length, + testReferenceCount: [ + ...new Set(flattenCommandReferenceGroups(result?.testReferencesByCommand)), + ].length, + occurrenceCount: 0, + }), + ); + + const frontendTextEntries = frontendTextResults.map((result) => + buildGovernanceSurfaceEntry({ + sourceType: "frontend-text", + result, + status: getTextStatus(result), + referenceCount: Array.isArray(result?.references) ? result.references.length : 0, + testReferenceCount: Array.isArray(result?.testReferences) + ? result.testReferences.length + : 0, + occurrenceCount: 0, + }), + ); + + const rustTextEntries = rustTextResults.map((result) => + buildGovernanceSurfaceEntry({ + sourceType: "rust-text", + result, + status: getTextStatus(result), + referenceCount: Array.isArray(result?.references) ? result.references.length : 0, + testReferenceCount: Array.isArray(result?.testReferences) + ? result.testReferences.length + : 0, + occurrenceCount: 0, + }), + ); + + const rustTextCountEntries = rustTextCountResults.map((result) => + buildGovernanceSurfaceEntry({ + sourceType: "rust-text-count", + result, + status: getTextCountStatus(result), + referenceCount: Array.isArray(result?.runtimeMatches) + ? result.runtimeMatches.length + : 0, + testReferenceCount: Array.isArray(result?.testMatches) + ? result.testMatches.length + : 0, + occurrenceCount: getTextCountOccurrenceCount(result), + }), + ); + + return [ + ...importEntries, + ...commandEntries, + ...frontendTextEntries, + ...rustTextEntries, + ...rustTextCountEntries, + ].sort((left, right) => { + if (right.score !== left.score) { + return right.score - left.score; + } + + if (right.referenceCount !== left.referenceCount) { + return right.referenceCount - left.referenceCount; + } + + return left.id.localeCompare(right.id); + }); +} + +function buildTrendFocusEntries(entries, sampleCount) { + const normalizedEntries = Array.isArray(entries) ? entries : []; + + return normalizedEntries + .map((entry) => { + const latest = isObject(entry?.latest) ? entry.latest : {}; + const delta = isObject(entry?.delta) ? entry.delta : {}; + const baseline = isObject(entry?.baseline) ? entry.baseline : {}; + const positiveDeltaInvalid = Math.max(0, normalizeNumber(delta.invalidCount)); + const positiveDeltaPending = Math.max( + 0, + normalizeNumber(delta.pendingRequestCaseCount), + ); + const positiveDeltaReview = Math.max( + 0, + normalizeNumber(delta.needsHumanReviewCount), + ); + const latestInvalid = normalizeNumber(latest.invalidCount); + const latestPending = normalizeNumber(latest.pendingRequestCaseCount); + const latestReview = normalizeNumber(latest.needsHumanReviewCount); + const latestCase = normalizeNumber(latest.caseCount); + const latestReady = normalizeNumber(latest.readyCount); + const score = + positiveDeltaInvalid * 120 + + positiveDeltaPending * 80 + + positiveDeltaReview * 50 + + latestInvalid * 40 + + latestPending * 25 + + latestReview * 15 + + latestCase; + + let state = "stable"; + if (sampleCount < 2 && score > 0) { + state = "seed-risk"; + } else if ( + positiveDeltaInvalid > 0 || + positiveDeltaPending > 0 || + positiveDeltaReview > 0 + ) { + state = "regressing"; + } else if (latestInvalid > 0 || latestPending > 0 || latestReview > 0) { + state = "present"; + } + + return { + name: normalizeString(entry?.name, "(unknown)"), + baseline: { + caseCount: normalizeNumber(baseline.caseCount), + readyCount: normalizeNumber(baseline.readyCount), + invalidCount: normalizeNumber(baseline.invalidCount), + pendingRequestCaseCount: normalizeNumber( + baseline.pendingRequestCaseCount, + ), + needsHumanReviewCount: normalizeNumber( + baseline.needsHumanReviewCount, + ), + }, + latest: { + caseCount: latestCase, + readyCount: latestReady, + invalidCount: latestInvalid, + pendingRequestCaseCount: latestPending, + needsHumanReviewCount: latestReview, + }, + delta: { + caseCount: normalizeNumber(delta.caseCount), + readyCount: normalizeNumber(delta.readyCount), + invalidCount: normalizeNumber(delta.invalidCount), + pendingRequestCaseCount: normalizeNumber(delta.pendingRequestCaseCount), + needsHumanReviewCount: normalizeNumber(delta.needsHumanReviewCount), + }, + state, + score, + }; + }) + .filter((entry) => entry.score > 0 || entry.latest.caseCount > 0) + .sort((left, right) => { + if (right.score !== left.score) { + return right.score - left.score; + } + return left.name.localeCompare(right.name); + }); +} + +function buildGovernanceSummary(governanceReport, surfaces) { + const summary = isObject(governanceReport?.summary) ? governanceReport.summary : {}; + const activeByClassification = { + current: 0, + compat: 0, + deprecated: 0, + "dead-candidate": 0, + other: 0, + }; + + for (const surface of surfaces) { + if (!surface.active) { + continue; + } + + if (surface.classification in activeByClassification) { + activeByClassification[surface.classification] += 1; + } else { + activeByClassification.other += 1; + } + } + + return { + monitorCount: surfaces.length, + activeSurfaceCount: surfaces.filter((surface) => surface.active).length, + activeByClassification, + violationCount: Array.isArray(summary.violations) ? summary.violations.length : 0, + classificationDriftCount: Array.isArray(summary.classificationDriftCandidates) + ? summary.classificationDriftCandidates.length + : 0, + zeroReferenceCandidateCount: Array.isArray(summary.zeroReferenceCandidates) + ? summary.zeroReferenceCandidates.length + : 0, + }; +} + +function buildTrendSummary(trendReport) { + const delta = isObject(trendReport?.delta) ? trendReport.delta : {}; + const latestTotals = isObject(trendReport?.latest?.totals) + ? trendReport.latest.totals + : {}; + const baselineTotals = isObject(trendReport?.baseline?.totals) + ? trendReport.baseline.totals + : {}; + const latestNeedsReviewCount = normalizeNumber( + latestTotals.needsHumanReviewCount, + ); + const latestReviewDecisionRecordedCount = normalizeNumber( + latestTotals.reviewDecisionRecordedCount, + ); + return { + sampleCount: normalizeNumber(trendReport?.sampleCount), + isSeed: normalizeNumber(trendReport?.sampleCount) < 2, + invalidDelta: normalizeNumber(delta.invalidCount), + pendingDelta: normalizeNumber(delta.pendingRequestCaseCount), + needsReviewDelta: normalizeNumber(delta.needsHumanReviewCount), + reviewDecisionRecordedDelta: + delta.reviewDecisionRecordedCount != null + ? normalizeNumber(delta.reviewDecisionRecordedCount) + : latestReviewDecisionRecordedCount - + normalizeNumber(baselineTotals.reviewDecisionRecordedCount), + latestNeedsReviewCount, + latestReviewDecisionRecordedCount, + reviewDecisionBacklogCount: Math.max( + 0, + latestNeedsReviewCount - latestReviewDecisionRecordedCount, + ), + readyRateDelta: normalizeNumber(delta.readyRate), + signals: Array.isArray(trendReport?.signals) ? trendReport.signals : [], + }; +} + +function buildDocFreshnessSummary(docFreshnessReport) { + const summary = isObject(docFreshnessReport?.summary) + ? docFreshnessReport.summary + : {}; + + return { + monitoredDocumentCount: normalizeNumber(summary.monitoredDocumentCount), + existingDocumentCount: normalizeNumber(summary.existingDocumentCount), + issueCount: normalizeNumber(summary.issueCount), + missingDocumentCount: normalizeNumber(summary.missingDocumentCount), + missingRequiredReferenceCount: normalizeNumber( + summary.missingRequiredReferenceCount, + ), + brokenMarkdownLinkCount: normalizeNumber(summary.brokenMarkdownLinkCount), + brokenCodePathReferenceCount: normalizeNumber( + summary.brokenCodePathReferenceCount, + ), + deletedSurfaceReferenceCount: normalizeNumber( + summary.deletedSurfaceReferenceCount, + ), + }; +} + +function buildDocFreshnessFocus(docFreshnessReport) { + const issues = Array.isArray(docFreshnessReport?.issues) + ? docFreshnessReport.issues + : []; + + const issueCounts = new Map(); + const documentCounts = new Map(); + + for (const issue of issues) { + const kind = normalizeString(issue?.kind, "unknown"); + const documentPath = normalizeString(issue?.documentPath, "(unknown)"); + issueCounts.set(kind, normalizeNumber(issueCounts.get(kind)) + 1); + documentCounts.set( + documentPath, + normalizeNumber(documentCounts.get(documentPath)) + 1, + ); + } + + return { + issueKinds: [...issueCounts.entries()] + .map(([kind, count]) => ({ kind, count })) + .sort((left, right) => right.count - left.count || left.kind.localeCompare(right.kind)) + .slice(0, 5), + documents: [...documentCounts.entries()] + .map(([documentPath, issueCount]) => ({ documentPath, issueCount })) + .sort( + (left, right) => + right.issueCount - left.issueCount || + left.documentPath.localeCompare(right.documentPath), + ) + .slice(0, 5), + }; +} + +function maybePushRecommendation(recommendations, action) { + if (recommendations.some((entry) => entry.id === action.id)) { + return; + } + recommendations.push(action); +} + +function buildRecommendations({ + trendSummary, + focusFailureModes, + focusSuiteTags, + focusReviewDecisionStatuses, + focusReviewRiskLevels, + docFreshnessSummary, + docFreshnessFocus, + governanceSummary, + governanceSurfaces, +}) { + const recommendations = []; + const topFailureModes = focusFailureModes + .slice(0, 3) + .map((entry) => entry.name); + const topSuiteTags = focusSuiteTags.slice(0, 3).map((entry) => entry.name); + const topReviewDecisionStatuses = focusReviewDecisionStatuses + .slice(0, 3) + .map((entry) => entry.name); + const topReviewRiskLevels = focusReviewRiskLevels + .slice(0, 3) + .map((entry) => entry.name); + const topGovernanceSurfaceIds = governanceSurfaces + .filter((entry) => entry.active && entry.classification !== "current") + .slice(0, 3) + .map((entry) => entry.id); + + if (governanceSummary.violationCount > 0) { + maybePushRecommendation(recommendations, { + id: "contracts-and-boundary-guards", + priority: "P0", + title: "先修命令/治理边界违规,再继续扩主线", + rationale: [ + `当前 governance report 发现 ${governanceSummary.violationCount} 条边界违规。`, + "这类问题会让 current / compat / deprecated 边界重新漂移,必须先封老路。", + ], + commands: ["npm run governance:legacy-report", "npm run test:contracts"], + backlogTools: [], + focusFailureModes: topFailureModes, + focusSuiteTags: topSuiteTags, + focusReviewDecisionStatuses: topReviewDecisionStatuses, + focusReviewRiskLevels: topReviewRiskLevels, + focusSurfaceIds: topGovernanceSurfaceIds, + }); + } + + if (trendSummary.isSeed) { + maybePushRecommendation(recommendations, { + id: "promote-high-value-replay", + priority: "P1", + title: "提升高价值 Replay 样本,结束 trend seed 状态", + rationale: [ + `当前 trend 样本数只有 ${trendSummary.sampleCount},还不能判断长期退化。`, + "先把最近一次高价值失败提升为 repo current 样本,再谈趋势治理。", + ], + commands: [ + "npm run harness:eval:promote -- --session-id \"\" --slug \"\"", + "npm run harness:eval:trend", + ], + backlogTools: [], + focusFailureModes: topFailureModes, + focusSuiteTags: topSuiteTags, + focusReviewDecisionStatuses: topReviewDecisionStatuses, + focusReviewRiskLevels: topReviewRiskLevels, + focusSurfaceIds: [], + }); + } + + const hasTrendPressure = + trendSummary.invalidDelta > 0 || + trendSummary.pendingDelta > 0 || + trendSummary.needsReviewDelta > 0 || + focusFailureModes.some((entry) => entry.state !== "stable"); + + if (hasTrendPressure) { + maybePushRecommendation(recommendations, { + id: "replay-and-smoke-follow-up", + priority: "P1", + title: "把高风险 failure mode 回挂到 replay / smoke 验证", + rationale: [ + `当前 failure mode 焦点:${topFailureModes.join("、") || "暂无"}。`, + "先用 replay / eval 固化失败,再按受影响主路径补最小 smoke,而不是直接凭印象清理。", + ], + commands: ["npm run harness:eval", "npm run harness:eval:trend"], + backlogTools: ["按受影响主路径追加 `npm run verify:gui-smoke` 或专项 smoke"], + focusFailureModes: topFailureModes, + focusSuiteTags: topSuiteTags, + focusReviewDecisionStatuses: topReviewDecisionStatuses, + focusReviewRiskLevels: topReviewRiskLevels, + focusSurfaceIds: [], + }); + } + + const hasReviewDecisionPressure = + trendSummary.reviewDecisionBacklogCount > 0 || + focusReviewDecisionStatuses.some( + (entry) => + entry.latest.caseCount > 0 && + (entry.name === "pending_review" || + entry.name === "needs_more_evidence" || + entry.state !== "stable"), + ) || + focusReviewRiskLevels.some( + (entry) => + entry.latest.caseCount > 0 && + (entry.name === "high" || entry.name === "medium"), + ); + + if (hasReviewDecisionPressure) { + maybePushRecommendation(recommendations, { + id: "review-decision-follow-up", + priority: + trendSummary.reviewDecisionBacklogCount > 0 || + topReviewRiskLevels.includes("high") + ? "P1" + : "P2", + title: "把人工审核状态与风险等级回挂到回归动作", + rationale: [ + trendSummary.reviewDecisionBacklogCount > 0 + ? `当前仍有 ${trendSummary.reviewDecisionBacklogCount} 个需要人工审核的 case 尚未留下最终 decision。` + : `当前人工审核状态焦点:${topReviewDecisionStatuses.join("、") || "暂无"}。`, + `当前风险等级焦点:${topReviewRiskLevels.join("、") || "暂无"}。高风险或补证据状态应优先回挂到 replay / contracts / smoke 主链。`, + ], + commands: [ + "npm run harness:eval", + "npm run harness:eval:trend", + "npm run harness:cleanup-report", + ], + backlogTools: + topReviewDecisionStatuses.includes("needs_more_evidence") || + trendSummary.reviewDecisionBacklogCount > 0 + ? [ + "补 evidence pack / analysis handoff / replay 证据字段,缩短 review-decision 从 pending_review 或 needs_more_evidence 回到可执行回归的路径", + ] + : [], + focusFailureModes: topFailureModes, + focusSuiteTags: topSuiteTags, + focusReviewDecisionStatuses: topReviewDecisionStatuses, + focusReviewRiskLevels: topReviewRiskLevels, + focusSurfaceIds: [], + }); + } + + const hasGovernancePressure = + governanceSummary.activeByClassification.compat > 0 || + governanceSummary.activeByClassification.deprecated > 0 || + governanceSummary.activeByClassification["dead-candidate"] > 0 || + governanceSummary.classificationDriftCount > 0 || + governanceSummary.zeroReferenceCandidateCount > 0; + + if (hasGovernancePressure) { + maybePushRecommendation(recommendations, { + id: "governance-cleanup-priority", + priority: + governanceSummary.activeByClassification["dead-candidate"] > 0 || + governanceSummary.activeByClassification.deprecated > 0 + ? "P1" + : "P2", + title: "按治理分类清理 compat / deprecated / dead-candidate 表面", + rationale: [ + `当前活跃 legacy surface:compat ${governanceSummary.activeByClassification.compat}、deprecated ${governanceSummary.activeByClassification.deprecated}、dead-candidate ${governanceSummary.activeByClassification["dead-candidate"]}。`, + "这一步是证据驱动的人类治理,不是 Lime 内部自动清理。", + ], + commands: ["npm run governance:legacy-report"], + backlogTools: [], + focusFailureModes: topFailureModes, + focusSuiteTags: topSuiteTags, + focusReviewDecisionStatuses: topReviewDecisionStatuses, + focusReviewRiskLevels: topReviewRiskLevels, + focusSurfaceIds: topGovernanceSurfaceIds, + }); + } + + const shouldReviewDocs = + docFreshnessSummary.issueCount > 0 || + trendSummary.isSeed || + governanceSummary.classificationDriftCount > 0 || + governanceSummary.activeByClassification.compat > 0 || + governanceSummary.activeByClassification.deprecated > 0 || + trendSummary.signals.some((signal) => signal.includes("字段漂移")); + + if (shouldReviewDocs) { + const topDocPaths = docFreshnessFocus.documents + .slice(0, 3) + .map((entry) => entry.documentPath); + maybePushRecommendation(recommendations, { + id: "doc-freshness-review", + priority: docFreshnessSummary.issueCount > 0 ? "P1" : "P2", + title: "回看 Harness 文档与事实源是否过期", + rationale: [ + "当 replay/trend 与治理边界同时在演进时,文档漂移会把分析和修复重新推回聊天窗口。", + docFreshnessSummary.issueCount > 0 + ? `当前 doc freshness 发现 ${docFreshnessSummary.issueCount} 个问题,应先修回链和失效引用。` + : "当前仓库已经有自动化 doc freshness 检查,可先扫描高频 Harness 文档的回链和已删除表面引用。", + ], + commands: ["npm run harness:doc-freshness"], + backlogTools: [], + focusFailureModes: topFailureModes, + focusSuiteTags: topSuiteTags, + focusReviewDecisionStatuses: topReviewDecisionStatuses, + focusReviewRiskLevels: topReviewRiskLevels, + focusSurfaceIds: + topDocPaths.length > 0 ? topDocPaths : topGovernanceSurfaceIds, + }); + } + + return recommendations.sort((left, right) => { + const priorityDiff = + (PRIORITY_RANK[left.priority] ?? 99) - + (PRIORITY_RANK[right.priority] ?? 99); + if (priorityDiff !== 0) { + return priorityDiff; + } + + return left.id.localeCompare(right.id); + }); +} + +export function buildGeneratedSlopReport({ + repoRoot, + trendReport, + docFreshnessReport, + governanceReport, + sources = {}, +}) { + const trendSummary = buildTrendSummary(trendReport); + const docFreshnessSummary = buildDocFreshnessSummary(docFreshnessReport); + const docFreshnessFocus = buildDocFreshnessFocus(docFreshnessReport); + const focusFailureModes = buildTrendFocusEntries( + trendReport?.classificationDeltas?.failureModes, + trendSummary.sampleCount, + ); + const focusSuiteTags = buildTrendFocusEntries( + trendReport?.classificationDeltas?.suiteTags, + trendSummary.sampleCount, + ); + const focusReviewDecisionStatuses = buildTrendFocusEntries( + trendReport?.classificationDeltas?.reviewDecisionStatuses, + trendSummary.sampleCount, + ); + const focusReviewRiskLevels = buildTrendFocusEntries( + trendReport?.classificationDeltas?.reviewRiskLevels, + trendSummary.sampleCount, + ); + const governanceSurfaces = buildGovernanceSurfaceEntries(governanceReport); + const governanceSummary = buildGovernanceSummary( + governanceReport, + governanceSurfaces, + ); + const recommendations = buildRecommendations({ + trendSummary, + focusFailureModes, + focusSuiteTags, + focusReviewDecisionStatuses, + focusReviewRiskLevels, + docFreshnessSummary, + docFreshnessFocus, + governanceSummary, + governanceSurfaces, + }); + + return { + reportVersion: "v1", + generatedAt: new Date().toISOString(), + repoRoot: normalizeString(repoRoot), + sources: { + trend: { + kind: normalizeString(sources?.trend?.kind, "generated"), + path: normalizeString(sources?.trend?.path), + }, + docFreshness: { + kind: normalizeString(sources?.docFreshness?.kind, "generated"), + path: normalizeString(sources?.docFreshness?.path), + }, + governance: { + kind: normalizeString(sources?.governance?.kind, "generated"), + path: normalizeString(sources?.governance?.path), + }, + }, + summary: { + trend: trendSummary, + docFreshness: docFreshnessSummary, + governance: governanceSummary, + }, + signals: [ + ...trendSummary.signals, + docFreshnessSummary.issueCount > 0 + ? `doc freshness 发现 ${docFreshnessSummary.issueCount} 个问题。` + : "当前高频 Harness 文档回链与路径引用保持新鲜。", + governanceSummary.violationCount > 0 + ? `governance 边界违规 ${governanceSummary.violationCount} 条。` + : "当前没有新的 governance 边界违规。", + governanceSummary.activeByClassification.compat > 0 + ? `compat 活跃 surface ${governanceSummary.activeByClassification.compat} 个。` + : "当前没有活跃 compat surface。", + governanceSummary.activeByClassification.deprecated > 0 + ? `deprecated 活跃 surface ${governanceSummary.activeByClassification.deprecated} 个。` + : "当前没有活跃 deprecated surface。", + trendSummary.reviewDecisionBacklogCount > 0 + ? `仍有 ${trendSummary.reviewDecisionBacklogCount} 个需要人工审核的 case 尚未记录最终 decision。` + : "当前没有待补录的人工审核 backlog。", + ], + focus: { + failureModes: focusFailureModes.slice(0, 5), + suiteTags: focusSuiteTags.slice(0, 5), + reviewDecisionStatuses: focusReviewDecisionStatuses.slice(0, 5), + reviewRiskLevels: focusReviewRiskLevels.slice(0, 5), + docFreshness: docFreshnessFocus, + governanceSurfaces: governanceSurfaces + .filter((entry) => entry.active && entry.classification !== "current") + .slice(0, 5), + }, + recommendations, + }; +} + +export function renderGeneratedSlopText(report) { + const lines = [ + "[harness-cleanup] generated slop report", + `[harness-cleanup] trend source: ${report.sources.trend.kind}${report.sources.trend.path ? ` (${report.sources.trend.path})` : ""}`, + `[harness-cleanup] doc freshness source: ${report.sources.docFreshness.kind}${report.sources.docFreshness.path ? ` (${report.sources.docFreshness.path})` : ""}`, + `[harness-cleanup] governance source: ${report.sources.governance.kind}${report.sources.governance.path ? ` (${report.sources.governance.path})` : ""}`, + `[harness-cleanup] trend samples: ${report.summary.trend.sampleCount}`, + `[harness-cleanup] trend seed: ${report.summary.trend.isSeed ? "yes" : "no"}`, + `[harness-cleanup] delta invalid: ${report.summary.trend.invalidDelta}`, + `[harness-cleanup] delta pending_request: ${report.summary.trend.pendingDelta}`, + `[harness-cleanup] delta review_decision_recorded: ${report.summary.trend.reviewDecisionRecordedDelta}`, + `[harness-cleanup] review backlog: ${report.summary.trend.reviewDecisionBacklogCount}`, + `[harness-cleanup] doc freshness issues: ${report.summary.docFreshness.issueCount}`, + `[harness-cleanup] active compat/deprecated/dead-candidate: ${report.summary.governance.activeByClassification.compat}/${report.summary.governance.activeByClassification.deprecated}/${report.summary.governance.activeByClassification["dead-candidate"]}`, + `[harness-cleanup] governance violations: ${report.summary.governance.violationCount}`, + ]; + + for (const signal of report.signals) { + lines.push(`[harness-cleanup] signal: ${signal}`); + } + + if (report.focus.failureModes.length > 0) { + lines.push("[harness-cleanup] top failure modes:"); + for (const entry of report.focus.failureModes) { + lines.push( + ` - ${entry.name}: state=${entry.state}, latest_pending=${entry.latest.pendingRequestCaseCount}, latest_invalid=${entry.latest.invalidCount}, score=${entry.score}`, + ); + } + } + + if (report.focus.governanceSurfaces.length > 0) { + lines.push("[harness-cleanup] top governance surfaces:"); + for (const entry of report.focus.governanceSurfaces) { + lines.push( + ` - ${entry.id}: ${entry.classification}/${entry.status}, refs=${entry.referenceCount}, violations=${entry.violationCount}`, + ); + } + } + + if (report.focus.reviewDecisionStatuses.length > 0) { + lines.push("[harness-cleanup] top review decision statuses:"); + for (const entry of report.focus.reviewDecisionStatuses) { + lines.push( + ` - ${entry.name}: state=${entry.state}, latest_case=${entry.latest.caseCount}, delta_case=${entry.delta.caseCount}, score=${entry.score}`, + ); + } + } + + if (report.focus.reviewRiskLevels.length > 0) { + lines.push("[harness-cleanup] top review risk levels:"); + for (const entry of report.focus.reviewRiskLevels) { + lines.push( + ` - ${entry.name}: state=${entry.state}, latest_case=${entry.latest.caseCount}, delta_case=${entry.delta.caseCount}, score=${entry.score}`, + ); + } + } + + if (report.focus.docFreshness.issueKinds.length > 0) { + lines.push("[harness-cleanup] doc freshness issues:"); + for (const entry of report.focus.docFreshness.issueKinds) { + lines.push(` - ${entry.kind}: count=${entry.count}`); + } + } + + if (report.recommendations.length > 0) { + lines.push("[harness-cleanup] recommendations:"); + for (const action of report.recommendations) { + lines.push(` - [${action.priority}] ${action.title}`); + } + } + + return `${lines.join("\n")}\n`; +} + +export function renderGeneratedSlopMarkdown(report) { + const lines = [ + "# Lime Harness Cleanup / Slop Report", + "", + `- 生成时间:${report.generatedAt}`, + `- trend 来源:${report.sources.trend.kind}${report.sources.trend.path ? `(\`${report.sources.trend.path}\`)` : ""}`, + `- doc freshness 来源:${report.sources.docFreshness.kind}${report.sources.docFreshness.path ? `(\`${report.sources.docFreshness.path}\`)` : ""}`, + `- governance 来源:${report.sources.governance.kind}${report.sources.governance.path ? `(\`${report.sources.governance.path}\`)` : ""}`, + "", + "## 摘要", + "", + `- trend 样本数:${report.summary.trend.sampleCount}`, + `- trend 是否仍为 seed:${report.summary.trend.isSeed ? "是" : "否"}`, + `- invalid delta:${report.summary.trend.invalidDelta}`, + `- pending request delta:${report.summary.trend.pendingDelta}`, + `- 已记录人工审核 delta:${report.summary.trend.reviewDecisionRecordedDelta}`, + `- 待补录人工审核 backlog:${report.summary.trend.reviewDecisionBacklogCount}`, + `- doc freshness 问题数:${report.summary.docFreshness.issueCount}`, + `- governance 违规数:${report.summary.governance.violationCount}`, + `- compat 活跃 surface:${report.summary.governance.activeByClassification.compat}`, + `- deprecated 活跃 surface:${report.summary.governance.activeByClassification.deprecated}`, + `- dead-candidate 活跃 surface:${report.summary.governance.activeByClassification["dead-candidate"]}`, + "", + "## 信号", + "", + ]; + + for (const signal of report.signals) { + lines.push(`- ${signal}`); + } + + if (report.focus.failureModes.length > 0) { + lines.push(""); + lines.push("## Failure Mode 焦点"); + lines.push(""); + lines.push( + "| Failure Mode | 状态 | latest case | latest invalid | latest pending_request | delta invalid | score |", + ); + lines.push("| --- | --- | --- | --- | --- | --- | --- |"); + for (const entry of report.focus.failureModes) { + lines.push( + `| ${entry.name} | ${entry.state} | ${entry.latest.caseCount} | ${entry.latest.invalidCount} | ${entry.latest.pendingRequestCaseCount} | ${entry.delta.invalidCount} | ${entry.score} |`, + ); + } + } + + if (report.focus.docFreshness.issueKinds.length > 0) { + lines.push(""); + lines.push("## Doc Freshness 焦点"); + lines.push(""); + lines.push("| Issue Kind | Count |"); + lines.push("| --- | --- |"); + for (const entry of report.focus.docFreshness.issueKinds) { + lines.push(`| ${entry.kind} | ${entry.count} |`); + } + } + + if (report.focus.governanceSurfaces.length > 0) { + lines.push(""); + lines.push("## Governance 焦点"); + lines.push(""); + lines.push( + "| Surface | 分类 | 状态 | 引用数 | 违规数 | 来源类型 |", + ); + lines.push("| --- | --- | --- | --- | --- | --- |"); + for (const entry of report.focus.governanceSurfaces) { + lines.push( + `| ${entry.id} | ${entry.classification} | ${entry.status} | ${entry.referenceCount} | ${entry.violationCount} | ${entry.sourceType} |`, + ); + } + } + + if (report.focus.reviewDecisionStatuses.length > 0) { + lines.push(""); + lines.push("## 人工审核状态焦点"); + lines.push(""); + lines.push( + "| 审核状态 | 状态 | latest case | latest invalid | delta case | score |", + ); + lines.push("| --- | --- | --- | --- | --- | --- |"); + for (const entry of report.focus.reviewDecisionStatuses) { + lines.push( + `| ${entry.name} | ${entry.state} | ${entry.latest.caseCount} | ${entry.latest.invalidCount} | ${entry.delta.caseCount} | ${entry.score} |`, + ); + } + } + + if (report.focus.reviewRiskLevels.length > 0) { + lines.push(""); + lines.push("## 风险等级焦点"); + lines.push(""); + lines.push( + "| 风险等级 | 状态 | latest case | latest invalid | delta case | score |", + ); + lines.push("| --- | --- | --- | --- | --- | --- |"); + for (const entry of report.focus.reviewRiskLevels) { + lines.push( + `| ${entry.name} | ${entry.state} | ${entry.latest.caseCount} | ${entry.latest.invalidCount} | ${entry.delta.caseCount} | ${entry.score} |`, + ); + } + } + + if (report.recommendations.length > 0) { + lines.push(""); + lines.push("## 推荐动作"); + lines.push(""); + for (const action of report.recommendations) { + lines.push(`### ${action.priority} · ${action.title}`); + lines.push(""); + for (const rationale of action.rationale) { + lines.push(`- ${rationale}`); + } + if (action.focusFailureModes.length > 0) { + lines.push(`- 关注 failure mode:${action.focusFailureModes.join("、")}`); + } + if (action.focusSuiteTags.length > 0) { + lines.push(`- 关注 suite tag:${action.focusSuiteTags.join("、")}`); + } + if (action.focusReviewDecisionStatuses.length > 0) { + lines.push( + `- 关注人工审核状态:${action.focusReviewDecisionStatuses.join("、")}`, + ); + } + if (action.focusReviewRiskLevels.length > 0) { + lines.push( + `- 关注风险等级:${action.focusReviewRiskLevels.join("、")}`, + ); + } + if (action.focusSurfaceIds.length > 0) { + lines.push(`- 关注 surface:${action.focusSurfaceIds.join("、")}`); + } + if (action.commands.length > 0) { + lines.push("- 可直接执行的命令:"); + for (const command of action.commands) { + lines.push(` - \`${command}\``); + } + } + if (action.backlogTools.length > 0) { + lines.push("- 待补工具 / 条件动作:"); + for (const item of action.backlogTools) { + lines.push(` - ${item}`); + } + } + lines.push(""); + } + } + + return `${lines.join("\n")}\n`; +} diff --git a/scripts/lib/generated-slop-report-core.test.ts b/scripts/lib/generated-slop-report-core.test.ts new file mode 100644 index 000000000..01ef68ad1 --- /dev/null +++ b/scripts/lib/generated-slop-report-core.test.ts @@ -0,0 +1,324 @@ +import { describe, expect, it } from "vitest"; + +import { buildGeneratedSlopReport } from "./generated-slop-report-core.mjs"; + +describe("generated-slop-report-core", () => { + it("应把 trend seed 与活跃 legacy surface 转成治理建议", () => { + const report = buildGeneratedSlopReport({ + repoRoot: "/tmp/lime", + trendReport: { + sampleCount: 1, + delta: { + invalidCount: 0, + pendingRequestCaseCount: 0, + needsHumanReviewCount: 0, + reviewDecisionRecordedCount: 0, + readyRate: 0, + }, + signals: ["样本数不足 2,当前仅形成 trend seed,还不能判断长期退化。"], + baseline: { + totals: { + needsHumanReviewCount: 0, + reviewDecisionRecordedCount: 0, + }, + }, + latest: { + totals: { + needsHumanReviewCount: 1, + reviewDecisionRecordedCount: 0, + }, + }, + classificationDeltas: { + failureModes: [ + { + name: "pending_request", + baseline: { + caseCount: 1, + readyCount: 1, + invalidCount: 0, + pendingRequestCaseCount: 1, + needsHumanReviewCount: 0, + }, + latest: { + caseCount: 1, + readyCount: 1, + invalidCount: 0, + pendingRequestCaseCount: 1, + needsHumanReviewCount: 0, + }, + delta: { + caseCount: 0, + readyCount: 0, + invalidCount: 0, + pendingRequestCaseCount: 0, + needsHumanReviewCount: 0, + }, + }, + ], + suiteTags: [ + { + name: "conversation-runtime", + baseline: { + caseCount: 1, + readyCount: 1, + invalidCount: 0, + pendingRequestCaseCount: 1, + needsHumanReviewCount: 0, + }, + latest: { + caseCount: 1, + readyCount: 1, + invalidCount: 0, + pendingRequestCaseCount: 1, + needsHumanReviewCount: 0, + }, + delta: { + caseCount: 0, + readyCount: 0, + invalidCount: 0, + pendingRequestCaseCount: 0, + needsHumanReviewCount: 0, + }, + }, + ], + reviewDecisionStatuses: [ + { + name: "pending_review", + baseline: { + caseCount: 0, + readyCount: 0, + invalidCount: 0, + pendingRequestCaseCount: 0, + needsHumanReviewCount: 0, + }, + latest: { + caseCount: 1, + readyCount: 0, + invalidCount: 1, + pendingRequestCaseCount: 0, + needsHumanReviewCount: 1, + }, + delta: { + caseCount: 1, + readyCount: 0, + invalidCount: 1, + pendingRequestCaseCount: 0, + needsHumanReviewCount: 1, + }, + }, + ], + reviewRiskLevels: [ + { + name: "high", + baseline: { + caseCount: 0, + readyCount: 0, + invalidCount: 0, + pendingRequestCaseCount: 0, + needsHumanReviewCount: 0, + }, + latest: { + caseCount: 1, + readyCount: 0, + invalidCount: 1, + pendingRequestCaseCount: 0, + needsHumanReviewCount: 1, + }, + delta: { + caseCount: 1, + readyCount: 0, + invalidCount: 1, + pendingRequestCaseCount: 0, + needsHumanReviewCount: 1, + }, + }, + ], + }, + }, + governanceReport: { + summary: { + zeroReferenceCandidates: [], + classificationDriftCandidates: [], + violations: [], + }, + importResults: [ + { + id: "team-subagent-scheduler-hook", + classification: "compat", + description: "compat hook", + existingTargets: ["src/hooks/useSubAgentScheduler.ts"], + references: [ + "src/components/agent/chat/hooks/useCompatSubagentRuntime.ts", + ], + testReferences: ["src/hooks/useSubAgentScheduler.test.tsx"], + violations: [], + }, + ], + commandResults: [], + frontendTextResults: [ + { + id: "migration-setting-key-leak", + classification: "deprecated", + description: "migration keys", + references: ["src-tauri/crates/core/src/database/migration/foo.rs"], + testReferences: [], + violations: [], + }, + ], + rustTextResults: [], + rustTextCountResults: [], + }, + docFreshnessReport: { + summary: { + monitoredDocumentCount: 10, + existingDocumentCount: 10, + issueCount: 0, + missingDocumentCount: 0, + missingRequiredReferenceCount: 0, + brokenMarkdownLinkCount: 0, + brokenCodePathReferenceCount: 0, + deletedSurfaceReferenceCount: 0, + }, + issues: [], + }, + sources: { + trend: { kind: "generated-current" }, + docFreshness: { kind: "live-scan" }, + governance: { kind: "live-scan" }, + }, + }); + + expect(report.summary.trend.isSeed).toBe(true); + expect(report.summary.trend.reviewDecisionBacklogCount).toBe(1); + expect(report.summary.docFreshness.issueCount).toBe(0); + expect(report.focus.failureModes[0].name).toBe("pending_request"); + expect(report.focus.reviewDecisionStatuses[0].name).toBe("pending_review"); + expect(report.focus.reviewRiskLevels[0].name).toBe("high"); + expect(report.focus.governanceSurfaces[0].id).toBe( + "migration-setting-key-leak", + ); + expect(report.recommendations.map((entry) => entry.id)).toEqual( + expect.arrayContaining([ + "promote-high-value-replay", + "replay-and-smoke-follow-up", + "review-decision-follow-up", + "governance-cleanup-priority", + "doc-freshness-review", + ]), + ); + expect( + report.recommendations.find( + (entry) => entry.id === "review-decision-follow-up", + )?.focusReviewRiskLevels, + ).toContain("high"); + expect( + report.recommendations.find((entry) => entry.id === "doc-freshness-review") + ?.commands, + ).toContain("npm run harness:doc-freshness"); + }); + + it("应把 governance 违规提升为 P0 守卫动作", () => { + const report = buildGeneratedSlopReport({ + repoRoot: "/tmp/lime", + trendReport: { + sampleCount: 2, + delta: { + invalidCount: 1, + pendingRequestCaseCount: 0, + needsHumanReviewCount: 0, + reviewDecisionRecordedCount: 0, + readyRate: -0.5, + }, + signals: ["invalid case 增加 1,存在回归候选。"], + baseline: { + totals: { + needsHumanReviewCount: 0, + reviewDecisionRecordedCount: 0, + }, + }, + latest: { + totals: { + needsHumanReviewCount: 0, + reviewDecisionRecordedCount: 0, + }, + }, + classificationDeltas: { + failureModes: [], + suiteTags: [], + reviewDecisionStatuses: [], + reviewRiskLevels: [], + }, + }, + governanceReport: { + summary: { + zeroReferenceCandidates: [], + classificationDriftCandidates: [], + violations: ["chat-compat -> src/foo.ts"], + }, + importResults: [], + commandResults: [ + { + id: "chat-compat-command", + classification: "deprecated", + description: "legacy command", + referencesByCommand: { + chat_create_session: ["src/lib/api/legacy.ts"], + }, + testReferencesByCommand: {}, + violations: ["chat_create_session -> src/lib/api/legacy.ts"], + }, + ], + frontendTextResults: [], + rustTextResults: [], + rustTextCountResults: [], + }, + docFreshnessReport: { + summary: { + monitoredDocumentCount: 10, + existingDocumentCount: 9, + issueCount: 2, + missingDocumentCount: 0, + missingRequiredReferenceCount: 1, + brokenMarkdownLinkCount: 0, + brokenCodePathReferenceCount: 1, + deletedSurfaceReferenceCount: 0, + }, + issues: [ + { + kind: "missing-required-reference", + documentPath: "docs/tech/harness/entropy-governance-workflow.md", + detail: "harness-evals.md", + }, + { + kind: "broken-code-path-reference", + documentPath: "docs/tech/harness/tooling-roadmap.md", + detail: "scripts/missing-tool.mjs", + }, + ], + }, + sources: { + trend: { kind: "input-json" }, + docFreshness: { kind: "input-json" }, + governance: { kind: "input-json" }, + }, + }); + + expect(report.recommendations[0].id).toBe("contracts-and-boundary-guards"); + expect(report.recommendations[0].priority).toBe("P0"); + expect(report.signals).toContain("doc freshness 发现 2 个问题。"); + expect(report.focus.docFreshness.issueKinds[0]).toEqual({ + kind: "broken-code-path-reference", + count: 1, + }); + expect(report.recommendations[0].commands).toEqual( + expect.arrayContaining([ + "npm run governance:legacy-report", + "npm run test:contracts", + ]), + ); + expect( + report.recommendations.find((entry) => entry.id === "doc-freshness-review") + ?.priority, + ).toBe("P1"); + }); +}); diff --git a/scripts/lib/harness-eval-history-record.test.ts b/scripts/lib/harness-eval-history-record.test.ts new file mode 100644 index 000000000..0e9c2de1c --- /dev/null +++ b/scripts/lib/harness-eval-history-record.test.ts @@ -0,0 +1,87 @@ +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import { execFileSync } from "node:child_process"; +import { afterEach, describe, expect, it } from "vitest"; + +const repoRoot = process.cwd(); +const tempRoots: string[] = []; + +function createTempRoot() { + const tempRoot = fs.mkdtempSync( + path.join(os.tmpdir(), "lime-harness-history-record-"), + ); + tempRoots.push(tempRoot); + return tempRoot; +} + +function runNodeScript(scriptRelativePath: string, args: string[]) { + const output = execFileSync( + process.execPath, + [path.join(repoRoot, scriptRelativePath), ...args], + { + cwd: repoRoot, + encoding: "utf8", + stdio: ["ignore", "pipe", "pipe"], + }, + ); + return JSON.parse(output); +} + +afterEach(() => { + while (tempRoots.length > 0) { + const tempRoot = tempRoots.pop(); + if (tempRoot) { + fs.rmSync(tempRoot, { recursive: true, force: true }); + } + } +}); + +describe("harness-eval-history-record", () => { + it("应记录本地 summary 历史,并生成非 seed 的 trend / cleanup", () => { + const tempRoot = createTempRoot(); + const historyDir = path.join(tempRoot, "artifacts", "history"); + const workspaceRoot = path.join(tempRoot, "workspace"); + + const firstResult = runNodeScript("scripts/harness-eval-history-record.mjs", [ + "--format", + "json", + "--history-dir", + historyDir, + "--workspace-root", + workspaceRoot, + "--output-json", + path.join(tempRoot, "first.json"), + ]); + + const secondResult = runNodeScript( + "scripts/harness-eval-history-record.mjs", + [ + "--format", + "json", + "--history-dir", + historyDir, + "--workspace-root", + workspaceRoot, + "--output-json", + path.join(tempRoot, "second.json"), + ], + ); + + expect(firstResult.historyCount).toBe(1); + expect(firstResult.trend.sampleCount).toBe(1); + + expect(secondResult.historyCount).toBe(2); + expect(secondResult.trend.sampleCount).toBe(2); + expect(secondResult.cleanup.trendSampleCount).toBe(2); + expect(fs.existsSync(secondResult.recordedSummaryPath)).toBe(true); + expect( + fs.existsSync(path.join(tempRoot, "artifacts", "harness-eval-trend.json")), + ).toBe(true); + expect( + fs.existsSync( + path.join(tempRoot, "artifacts", "harness-cleanup-report.json"), + ), + ).toBe(true); + }); +}); diff --git a/scripts/lib/harness-eval-history-window.test.ts b/scripts/lib/harness-eval-history-window.test.ts new file mode 100644 index 000000000..f377f90da --- /dev/null +++ b/scripts/lib/harness-eval-history-window.test.ts @@ -0,0 +1,226 @@ +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import { execFileSync } from "node:child_process"; +import { afterEach, describe, expect, it } from "vitest"; + +const repoRoot = process.cwd(); +const tempRoots: string[] = []; + +function createTempRoot() { + const tempRoot = fs.mkdtempSync( + path.join(os.tmpdir(), "lime-harness-history-window-"), + ); + tempRoots.push(tempRoot); + return tempRoot; +} + +function ensureDir(dirPath: string) { + fs.mkdirSync(dirPath, { recursive: true }); +} + +function writeJson(filePath: string, payload: unknown) { + ensureDir(path.dirname(filePath)); + fs.writeFileSync(filePath, `${JSON.stringify(payload, null, 2)}\n`, "utf8"); +} + +function writeText(filePath: string, value: string) { + ensureDir(path.dirname(filePath)); + fs.writeFileSync(filePath, value, "utf8"); +} + +function sleepMs(durationMs: number) { + Atomics.wait(new Int32Array(new SharedArrayBuffer(4)), 0, 0, durationMs); +} + +function createHarnessManifest(caseDir: string) { + return { + manifestVersion: "v1", + title: "Harness History Window Test Manifest", + defaults: { + requiredArtifacts: [ + "input.json", + "expected.json", + "grader.md", + "evidence-links.json", + ], + requiredInputFields: [ + "session.sessionId", + "session.threadId", + "task.goalSummary", + "classification.suiteTags", + "classification.failureModes", + "linkedArtifacts.handoffBundle.relativeRoot", + "linkedArtifacts.evidencePack.relativeRoot", + ], + requiredExpectedFields: [ + "successCriteria", + "blockingChecks", + "artifactChecks", + "graderSuggestion.preferredMode", + ], + requiredEvidenceFields: [ + "handoffBundle.relativeRoot", + "evidencePack.relativeRoot", + ], + }, + suites: [ + { + id: "repo-fixtures", + title: "仓库固定 Replay 样本", + cases: [ + { + id: "fixture-history-window", + title: "Harness 历史窗口样本", + source: "repo_fixture", + caseDir, + }, + ], + }, + ], + }; +} + +function createReplayFixture(tempRoot: string) { + const caseDir = path.join(tempRoot, "fixture-case"); + const handoffRoot = path.join(tempRoot, "handoff"); + const evidenceRoot = path.join(tempRoot, "evidence"); + + writeJson(path.join(caseDir, "input.json"), { + session: { + sessionId: "fixture-history-session", + threadId: "fixture-history-thread", + }, + task: { + goalSummary: "验证 harness eval history window", + }, + classification: { + suiteTags: ["conversation-runtime", "replay"], + failureModes: ["pending_request"], + }, + linkedArtifacts: { + handoffBundle: { + relativeRoot: ".lime/handoff", + absoluteRoot: handoffRoot, + }, + evidencePack: { + relativeRoot: ".lime/evidence", + absoluteRoot: evidenceRoot, + }, + }, + }); + writeJson(path.join(caseDir, "expected.json"), { + sessionId: "fixture-history-session", + threadId: "fixture-history-thread", + goalSummary: "验证 harness eval history window", + successCriteria: ["history snapshot 可以被 trend 使用"], + blockingChecks: ["summary history 目录会保留最近窗口"], + artifactChecks: ["history dir 中存在 harness-eval-summary.json"], + graderSuggestion: { + preferredMode: "summary_only", + requiresHumanReview: false, + }, + }); + writeJson(path.join(caseDir, "evidence-links.json"), { + handoffBundle: { + relativeRoot: ".lime/handoff", + absoluteRoot: handoffRoot, + }, + evidencePack: { + relativeRoot: ".lime/evidence", + absoluteRoot: evidenceRoot, + }, + }); + writeText(path.join(caseDir, "grader.md"), "# Grader\n"); + + return caseDir; +} + +function runNodeScript(scriptRelativePath: string, args: string[]) { + const output = execFileSync( + process.execPath, + [path.join(repoRoot, scriptRelativePath), ...args], + { + cwd: repoRoot, + encoding: "utf8", + stdio: ["ignore", "pipe", "pipe"], + }, + ); + return JSON.parse(output); +} + +afterEach(() => { + while (tempRoots.length > 0) { + const tempRoot = tempRoots.pop(); + if (tempRoot) { + fs.rmSync(tempRoot, { recursive: true, force: true }); + } + } +}); + +describe("Harness eval history window", () => { + it("runner 应记录并裁剪历史窗口,trend 应复用该目录", () => { + const tempRoot = createTempRoot(); + const caseDir = createReplayFixture(tempRoot); + const manifestPath = path.join(tempRoot, "manifest.json"); + const historyDir = path.join(tempRoot, "history"); + + writeJson(manifestPath, createHarnessManifest(caseDir)); + + runNodeScript("scripts/harness-eval-runner.mjs", [ + "--format", + "json", + "--manifest", + manifestPath, + "--record-history-dir", + historyDir, + "--history-retain", + "2", + "--no-strict", + ]); + sleepMs(10); + runNodeScript("scripts/harness-eval-runner.mjs", [ + "--format", + "json", + "--manifest", + manifestPath, + "--record-history-dir", + historyDir, + "--history-retain", + "2", + "--no-strict", + ]); + sleepMs(10); + runNodeScript("scripts/harness-eval-runner.mjs", [ + "--format", + "json", + "--manifest", + manifestPath, + "--record-history-dir", + historyDir, + "--history-retain", + "2", + "--no-strict", + ]); + + const historyFiles = fs + .readdirSync(historyDir) + .filter((entry) => entry.endsWith("-harness-eval-summary.json")) + .sort(); + expect(historyFiles).toHaveLength(2); + + const report = runNodeScript("scripts/harness-eval-trend-report.mjs", [ + "--format", + "json", + "--manifest", + manifestPath, + "--history-dir", + historyDir, + ]); + + expect(report.sampleCount).toBe(2); + expect(report.signals).not.toContain( + "样本数不足 2,当前仅形成 trend seed,还不能判断长期退化。", + ); + }); +}); diff --git a/scripts/lib/harness-review-decision-evals.test.ts b/scripts/lib/harness-review-decision-evals.test.ts new file mode 100644 index 000000000..c1146e280 --- /dev/null +++ b/scripts/lib/harness-review-decision-evals.test.ts @@ -0,0 +1,447 @@ +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import { execFileSync } from "node:child_process"; +import { afterEach, describe, expect, it } from "vitest"; + +const repoRoot = process.cwd(); +const tempRoots: string[] = []; + +function createTempRoot() { + const tempRoot = fs.mkdtempSync( + path.join(os.tmpdir(), "lime-harness-review-evals-"), + ); + tempRoots.push(tempRoot); + return tempRoot; +} + +function ensureDir(dirPath: string) { + fs.mkdirSync(dirPath, { recursive: true }); +} + +function writeJson(filePath: string, payload: unknown) { + ensureDir(path.dirname(filePath)); + fs.writeFileSync(filePath, `${JSON.stringify(payload, null, 2)}\n`, "utf8"); +} + +function writeText(filePath: string, value: string) { + ensureDir(path.dirname(filePath)); + fs.writeFileSync(filePath, value, "utf8"); +} + +function createHarnessManifest(overrides?: Record) { + return { + manifestVersion: "v1", + title: "Test Harness Eval Manifest", + defaults: { + requiredArtifacts: [ + "input.json", + "expected.json", + "grader.md", + "evidence-links.json", + ], + requiredInputFields: [ + "session.sessionId", + "session.threadId", + "task.goalSummary", + "classification.suiteTags", + "classification.failureModes", + "linkedArtifacts.handoffBundle.relativeRoot", + "linkedArtifacts.evidencePack.relativeRoot", + ], + requiredExpectedFields: [ + "successCriteria", + "blockingChecks", + "artifactChecks", + "graderSuggestion.preferredMode", + ], + requiredEvidenceFields: [ + "handoffBundle.relativeRoot", + "evidencePack.relativeRoot", + ], + }, + suites: [], + ...overrides, + }; +} + +function createWorkspaceSessionArtifacts(tempRoot: string, sessionId: string) { + const workspaceRoot = path.join(tempRoot, "workspace"); + const sessionRoot = path.join( + workspaceRoot, + ".lime", + "harness", + "sessions", + sessionId, + ); + const replayDir = path.join(sessionRoot, "replay"); + const reviewDir = path.join(sessionRoot, "review"); + ensureDir(replayDir); + ensureDir(reviewDir); + + const handoffRoot = path.join(sessionRoot, "handoff"); + const evidenceRoot = path.join(sessionRoot, "evidence"); + const inputPayload = { + session: { + sessionId: sessionId, + threadId: `${sessionId}-thread`, + }, + task: { + goalSummary: "验证 review decision 会进入 replay/eval 资产", + }, + classification: { + suiteTags: ["conversation-runtime", "replay"], + failureModes: ["pending_request"], + primaryBlockingKind: "pending_request", + }, + runtimeContext: { + pendingRequests: [ + { + id: "req-1", + }, + ], + }, + linkedArtifacts: { + handoffBundle: { + relativeRoot: `.lime/harness/sessions/${sessionId}/handoff`, + absoluteRoot: handoffRoot, + }, + evidencePack: { + relativeRoot: `.lime/harness/sessions/${sessionId}/evidence`, + absoluteRoot: evidenceRoot, + }, + }, + }; + const expectedPayload = { + sessionId: sessionId, + threadId: `${sessionId}-thread`, + goalSummary: "验证 review decision 会进入 replay/eval 资产", + successCriteria: ["人工审核状态进入 eval 摘要"], + blockingChecks: ["review decision 不再停留在工作区孤岛"], + artifactChecks: ["review-decision.json 被识别"], + graderSuggestion: { + preferredMode: "manual_review", + requiresHumanReview: true, + }, + }; + const evidencePayload = { + handoffBundle: { + relativeRoot: `.lime/harness/sessions/${sessionId}/handoff`, + absoluteRoot: handoffRoot, + }, + evidencePack: { + relativeRoot: `.lime/harness/sessions/${sessionId}/evidence`, + absoluteRoot: evidenceRoot, + }, + }; + const reviewDecisionPayload = { + schemaVersion: "v1", + contractShape: "lime_review_decision_template", + decision: { + decisionStatus: "accepted", + decisionSummary: "确认该失败应沉淀为长期回归样本。", + chosenFixStrategy: "继续把人工审核结果挂到 replay/eval 主链。", + riskLevel: "high", + riskTags: ["runtime", "eval"], + humanReviewer: "Lime Maintainer", + reviewedAt: "2026-03-27T12:00:00Z", + followupActions: ["补 trend 汇总"], + regressionRequirements: ["npm run harness:eval"], + notes: `工作区目录 ${workspaceRoot} 已参与人工审核。`, + }, + }; + + writeJson(path.join(replayDir, "input.json"), inputPayload); + writeJson(path.join(replayDir, "expected.json"), expectedPayload); + writeJson(path.join(replayDir, "evidence-links.json"), evidencePayload); + writeText( + path.join(replayDir, "grader.md"), + `# Grader\n\n- 工作区:\`${workspaceRoot}\`\n`, + ); + writeJson( + path.join(reviewDir, "review-decision.json"), + reviewDecisionPayload, + ); + writeText( + path.join(reviewDir, "review-decision.md"), + `# Review\n\n- 工作区:\`${workspaceRoot}\`\n`, + ); + + return { + workspaceRoot, + replayDir, + reviewDir, + }; +} + +function runNodeScript(scriptRelativePath: string, args: string[]) { + const output = execFileSync( + process.execPath, + [path.join(repoRoot, scriptRelativePath), ...args], + { + cwd: repoRoot, + encoding: "utf8", + stdio: ["ignore", "pipe", "pipe"], + }, + ); + return JSON.parse(output); +} + +afterEach(() => { + while (tempRoots.length > 0) { + const tempRoot = tempRoots.pop(); + if (tempRoot) { + fs.rmSync(tempRoot, { recursive: true, force: true }); + } + } +}); + +describe("Harness review decision / eval integration", () => { + it("harness-replay-promote 应复制并脱敏 review decision 制品", () => { + const tempRoot = createTempRoot(); + const { workspaceRoot } = createWorkspaceSessionArtifacts( + tempRoot, + "session-promote-1", + ); + const manifestPath = path.join(tempRoot, "manifest.json"); + const fixturesRoot = path.join(tempRoot, "fixtures"); + + writeJson( + manifestPath, + createHarnessManifest({ + suites: [ + { + id: "repo-promoted-replays", + title: "仓库沉淀 Replay 样本", + cases: [], + }, + ], + }), + ); + + const result = runNodeScript("scripts/harness-replay-promote.mjs", [ + "--workspace-root", + workspaceRoot, + "--session-id", + "session-promote-1", + "--slug", + "review-promoted-case", + "--manifest", + manifestPath, + "--fixtures-root", + fixturesRoot, + "--format", + "json", + ]); + + const promotedCaseDir = path.join(fixturesRoot, "review-promoted-case"); + const promotedReviewJson = JSON.parse( + fs.readFileSync( + path.join(promotedCaseDir, "review-decision.json"), + "utf8", + ), + ); + const promotedReviewMarkdown = fs.readFileSync( + path.join(promotedCaseDir, "review-decision.md"), + "utf8", + ); + const manifest = JSON.parse(fs.readFileSync(manifestPath, "utf8")); + const caseEntry = manifest.suites[0].cases[0]; + + expect(result.reviewDecisionStatus).toBe("accepted"); + expect(promotedReviewJson.decision.decisionStatus).toBe("accepted"); + expect(promotedReviewJson.decision.notes).toContain("/workspace/lime"); + expect(promotedReviewMarkdown).toContain("/workspace/lime"); + expect(caseEntry.reviewDecision).toMatchObject({ + decisionStatus: "accepted", + riskLevel: "high", + humanReviewer: "Lime Maintainer", + }); + }); + + it("harness-eval-runner 应在 workspace replay discovery 中读取 sibling review decision", () => { + const tempRoot = createTempRoot(); + const { workspaceRoot } = createWorkspaceSessionArtifacts( + tempRoot, + "session-runner-1", + ); + const manifestPath = path.join(tempRoot, "manifest.json"); + + writeJson( + manifestPath, + createHarnessManifest({ + suites: [ + { + id: "workspace-replay-discovery", + title: "工作区 Replay 自动发现", + cases: [ + { + id: "workspace-session-replays", + title: "工作区会话 Replay 样本", + source: "workspace_replay_discovery", + root: ".lime/harness/sessions", + allowZeroMatches: false, + tags: ["workspace", "replay"], + }, + ], + }, + ], + }), + ); + + const summary = runNodeScript("scripts/harness-eval-runner.mjs", [ + "--format", + "json", + "--manifest", + manifestPath, + "--workspace-root", + workspaceRoot, + "--no-strict", + ]); + + expect(summary.totals.reviewDecisionRecordedCount).toBe(1); + expect(summary.breakdowns.reviewDecisionStatuses).toEqual( + expect.arrayContaining([ + expect.objectContaining({ + name: "accepted", + caseCount: 1, + }), + ]), + ); + expect(summary.breakdowns.reviewRiskLevels).toEqual( + expect.arrayContaining([ + expect.objectContaining({ + name: "high", + caseCount: 1, + }), + ]), + ); + expect(summary.suites[0].cases[0]).toMatchObject({ + reviewDecisionStatus: "accepted", + reviewRiskLevel: "high", + reviewHumanReviewer: "Lime Maintainer", + }); + }); + + it("harness-eval-trend-report 应聚合人工审核状态与风险等级变化", () => { + const tempRoot = createTempRoot(); + const baselinePath = path.join(tempRoot, "baseline.json"); + const latestPath = path.join(tempRoot, "latest.json"); + + writeJson(baselinePath, { + generatedAt: "2026-03-27T10:00:00Z", + totals: { + suiteCount: 1, + caseCount: 1, + readyCount: 1, + invalidCount: 0, + pendingRequestCaseCount: 0, + needsHumanReviewCount: 1, + }, + breakdowns: { + suiteTags: [], + failureModes: [], + reviewDecisionStatuses: [ + { + name: "needs_more_evidence", + caseCount: 1, + readyCount: 1, + invalidCount: 0, + pendingRequestCaseCount: 0, + needsHumanReviewCount: 1, + }, + ], + reviewRiskLevels: [ + { + name: "high", + caseCount: 1, + readyCount: 1, + invalidCount: 0, + pendingRequestCaseCount: 0, + needsHumanReviewCount: 1, + }, + ], + }, + suites: [], + }); + + writeJson(latestPath, { + generatedAt: "2026-03-27T12:00:00Z", + totals: { + suiteCount: 1, + caseCount: 1, + readyCount: 1, + invalidCount: 0, + pendingRequestCaseCount: 0, + needsHumanReviewCount: 0, + }, + breakdowns: { + suiteTags: [], + failureModes: [], + reviewDecisionStatuses: [ + { + name: "accepted", + caseCount: 1, + readyCount: 1, + invalidCount: 0, + pendingRequestCaseCount: 0, + needsHumanReviewCount: 0, + }, + ], + reviewRiskLevels: [ + { + name: "medium", + caseCount: 1, + readyCount: 1, + invalidCount: 0, + pendingRequestCaseCount: 0, + needsHumanReviewCount: 0, + }, + ], + }, + suites: [], + }); + + const report = runNodeScript("scripts/harness-eval-trend-report.mjs", [ + "--format", + "json", + "--input", + baselinePath, + "--input", + latestPath, + ]); + + expect(report.classificationDeltas.reviewDecisionStatuses).toEqual( + expect.arrayContaining([ + expect.objectContaining({ + name: "accepted", + delta: expect.objectContaining({ + caseCount: 1, + }), + }), + expect.objectContaining({ + name: "needs_more_evidence", + delta: expect.objectContaining({ + caseCount: -1, + }), + }), + ]), + ); + expect(report.classificationDeltas.reviewRiskLevels).toEqual( + expect.arrayContaining([ + expect.objectContaining({ + name: "high", + delta: expect.objectContaining({ + caseCount: -1, + }), + }), + expect.objectContaining({ + name: "medium", + delta: expect.objectContaining({ + caseCount: 1, + }), + }), + ]), + ); + }); +}); diff --git a/scripts/lib/legacy-surface-report-core.mjs b/scripts/lib/legacy-surface-report-core.mjs new file mode 100644 index 000000000..9432cedc6 --- /dev/null +++ b/scripts/lib/legacy-surface-report-core.mjs @@ -0,0 +1,688 @@ +import fs from "node:fs"; +import path from "node:path"; +import process from "node:process"; +import legacySurfaceCatalog from "../../src/lib/governance/legacySurfaceCatalog.json" with { type: "json" }; +import { + buildLegacySurfaceSummary, + getCommandStatus, + getImportStatus, + getTextCountStatus, + getTextStatus, + serializeMapEntries, +} from "./legacy-surface-report-summary.mjs"; + +const repoRoot = path.resolve(process.cwd()); +const sourceRoots = ["src"]; +const sourceExtensions = new Set([ + ".ts", + ".tsx", + ".js", + ".jsx", + ".mjs", + ".cjs", +]); +const rustSourceRoots = ["src-tauri/src", "src-tauri/crates"]; +const rustSourceExtensions = new Set([".rs"]); +const ignoredDirs = new Set([ + "node_modules", + "dist", + "build", + "coverage", + "target", + ".git", + ".turbo", + ".next", +]); + +const { + imports: importSurfaceMonitors, + commands: commandSurfaceMonitors, + frontendText: frontendTextSurfaceMonitors, + rustText: rustTextSurfaceMonitors, + rustTextCounts: rustTextCountMonitors, +} = legacySurfaceCatalog; + +function normalizePath(filePath) { + return filePath.split(path.sep).join("/"); +} + +function resolveExistingSourcePath(absolutePath) { + if (fs.existsSync(absolutePath)) { + const stats = fs.statSync(absolutePath); + if (stats.isFile()) { + return absolutePath; + } + } + + if (!path.extname(absolutePath)) { + for (const extension of sourceExtensions) { + const fileCandidate = `${absolutePath}${extension}`; + if (fs.existsSync(fileCandidate) && fs.statSync(fileCandidate).isFile()) { + return fileCandidate; + } + } + } + + for (const extension of sourceExtensions) { + const indexCandidate = path.join(absolutePath, `index${extension}`); + if (fs.existsSync(indexCandidate) && fs.statSync(indexCandidate).isFile()) { + return indexCandidate; + } + } + + return null; +} + +function resolveImportPath(importerRelativePath, specifier) { + let absoluteCandidate = null; + + if (specifier.startsWith("@/")) { + absoluteCandidate = path.join(repoRoot, "src", specifier.slice(2)); + } else if (specifier.startsWith(".")) { + absoluteCandidate = path.resolve( + path.dirname(path.join(repoRoot, importerRelativePath)), + specifier, + ); + } + + if (!absoluteCandidate) { + return null; + } + + const resolvedPath = resolveExistingSourcePath(absoluteCandidate); + if (!resolvedPath) { + return null; + } + + return normalizePath(path.relative(repoRoot, resolvedPath)); +} + +function isTestFile(relativePath) { + return ( + /(^|\/)tests(\/|$)/.test(relativePath) || + /(^|\/)(__tests__|__mocks__)(\/|$)/.test(relativePath) || + /\.(test|spec)\.[^/.]+$/.test(relativePath) + ); +} + +function walkDirectory(directoryPath, extensions) { + const files = []; + + for (const entry of fs.readdirSync(directoryPath, { withFileTypes: true })) { + if (ignoredDirs.has(entry.name)) { + continue; + } + + const fullPath = path.join(directoryPath, entry.name); + if (entry.isDirectory()) { + files.push(...walkDirectory(fullPath, extensions)); + continue; + } + + if (!extensions.has(path.extname(entry.name))) { + continue; + } + + files.push(fullPath); + } + + return files; +} + +function extractImportSpecifiers(sourceCode) { + const specifiers = new Set(); + const patterns = [ + /\bimport\s+(?:type\s+)?(?:[\s\S]*?\s+from\s+)?["'`]([^"'`]+)["'`]/g, + /\bexport\s+(?:type\s+)?[\s\S]*?\s+from\s+["'`]([^"'`]+)["'`]/g, + /\bimport\s*\(\s*["'`]([^"'`]+)["'`]\s*\)/g, + /\brequire\s*\(\s*["'`]([^"'`]+)["'`]\s*\)/g, + /\b(?:vi|jest)\.mock\s*\(\s*["'`]([^"'`]+)["'`]/g, + ]; + + for (const pattern of patterns) { + for (const match of sourceCode.matchAll(pattern)) { + specifiers.add(match[1]); + } + } + + return specifiers; +} + +function extractInvokeCommands(sourceCode) { + const commands = new Set(); + const patterns = [ + /\bsafeInvoke(?:<[^>]+>)?\s*\(\s*["'`]([^"'`]+)["'`]/g, + /\binvoke(?:<[^>]+>)?\s*\(\s*["'`]([^"'`]+)["'`]/g, + ]; + + for (const pattern of patterns) { + for (const match of sourceCode.matchAll(pattern)) { + commands.add(match[1]); + } + } + + return commands; +} + +function stripRustTestModules(sourceCode) { + return sourceCode.replace( + /(?:^|\n)\s*#\s*\[\s*cfg\s*\(\s*test\s*\)\s*\]\s*(?:pub\s+)?mod\s+\w+\s*(?:\{[\s\S]*$|;)/m, + "\n", + ); +} + +function collectSources() { + const runtimeSources = []; + const testSources = []; + + for (const root of sourceRoots) { + const absoluteRoot = path.join(repoRoot, root); + if (!fs.existsSync(absoluteRoot)) { + continue; + } + + for (const filePath of walkDirectory(absoluteRoot, sourceExtensions)) { + const relativePath = normalizePath(path.relative(repoRoot, filePath)); + const sourceCode = fs.readFileSync(filePath, "utf8"); + const imports = extractImportSpecifiers(sourceCode); + const collectedSource = { + relativePath, + imports, + resolvedImports: new Set( + [...imports] + .map((specifier) => resolveImportPath(relativePath, specifier)) + .filter(Boolean), + ), + commands: extractInvokeCommands(sourceCode), + }; + + if (isTestFile(relativePath)) { + testSources.push(collectedSource); + continue; + } + + runtimeSources.push(collectedSource); + } + } + + return { + runtimeSources, + testSources, + }; +} + +function collectTextSources(roots, extensions) { + const runtimeSources = []; + const testSources = []; + + for (const root of roots) { + const absoluteRoot = path.join(repoRoot, root); + if (!fs.existsSync(absoluteRoot)) { + continue; + } + + for (const filePath of walkDirectory(absoluteRoot, extensions)) { + const relativePath = normalizePath(path.relative(repoRoot, filePath)); + const sourceCode = fs.readFileSync(filePath, "utf8"); + const collectedSource = { + relativePath, + sourceCode: + path.extname(relativePath) === ".rs" + ? stripRustTestModules(sourceCode) + : sourceCode, + rawSourceCode: sourceCode, + }; + + if (isTestFile(relativePath)) { + testSources.push(collectedSource); + continue; + } + + runtimeSources.push(collectedSource); + } + } + + return { + runtimeSources, + testSources, + }; +} + +function formatPaths(paths) { + if (paths.length === 0) { + return "无"; + } + + return paths.map((item) => ` - ${item}`).join("\n"); +} + +function evaluateImportMonitor(monitor, runtimeSources, testSources) { + const existingTargets = monitor.targets.filter((target) => + fs.existsSync(path.join(repoRoot, target)), + ); + const missingTargets = monitor.targets.filter( + (target) => !fs.existsSync(path.join(repoRoot, target)), + ); + const references = runtimeSources + .filter((file) => + [...file.resolvedImports].some((resolvedPath) => + monitor.targets.includes(resolvedPath), + ), + ) + .map((file) => file.relativePath) + .sort(); + const testReferences = testSources + .filter((file) => + [...file.resolvedImports].some((resolvedPath) => + monitor.targets.includes(resolvedPath), + ), + ) + .map((file) => file.relativePath) + .sort(); + + const violations = references.filter( + (relativePath) => !monitor.allowedPaths.includes(relativePath), + ); + + return { + ...monitor, + existingTargets, + missingTargets, + references, + testReferences, + violations, + }; +} + +function evaluateCommandMonitor(monitor, runtimeSources, testSources) { + const referencesByCommand = new Map(); + const testReferencesByCommand = new Map(); + + for (const command of monitor.commands) { + referencesByCommand.set( + command, + runtimeSources + .filter((file) => file.commands.has(command)) + .map((file) => file.relativePath) + .sort(), + ); + testReferencesByCommand.set( + command, + testSources + .filter((file) => file.commands.has(command)) + .map((file) => file.relativePath) + .sort(), + ); + } + + const violations = []; + for (const [command, references] of referencesByCommand.entries()) { + for (const relativePath of references) { + if (!monitor.allowedPaths.includes(relativePath)) { + violations.push(`${command} -> ${relativePath}`); + } + } + } + + return { + ...monitor, + referencesByCommand, + testReferencesByCommand, + violations, + }; +} + +function evaluateTextMonitor(monitor, runtimeSources, testSources) { + const filteredRuntimeSources = monitor.includePathPrefixes + ? runtimeSources.filter((file) => + monitor.includePathPrefixes.some((prefix) => + file.relativePath.startsWith(prefix), + ), + ) + : runtimeSources; + const filteredTestSources = monitor.includePathPrefixes + ? testSources.filter((file) => + monitor.includePathPrefixes.some((prefix) => + file.relativePath.startsWith(prefix), + ), + ) + : testSources; + const matchesPattern = (sourceCode) => + monitor.patterns.some((pattern) => sourceCode.includes(pattern)) || + (monitor.regexPatterns ?? []).some((pattern) => + new RegExp(pattern, "m").test(sourceCode), + ); + + const references = filteredRuntimeSources + .filter((file) => matchesPattern(file.sourceCode)) + .map((file) => file.relativePath) + .sort(); + const testReferences = filteredTestSources + .filter((file) => matchesPattern(file.rawSourceCode ?? file.sourceCode)) + .map((file) => file.relativePath) + .sort(); + const violations = references.filter( + (relativePath) => !monitor.allowedPaths.includes(relativePath), + ); + + return { + ...monitor, + references, + testReferences, + violations, + }; +} + +function countOccurrences(sourceCode, pattern) { + if (!pattern) { + return 0; + } + + let count = 0; + let startIndex = 0; + + while (true) { + const matchIndex = sourceCode.indexOf(pattern, startIndex); + if (matchIndex === -1) { + return count; + } + count += 1; + startIndex = matchIndex + pattern.length; + } +} + +function evaluateTextCountMonitor(monitor, runtimeSources, testSources) { + const filteredRuntimeSources = monitor.includePathPrefixes + ? runtimeSources.filter((file) => + monitor.includePathPrefixes.some((prefix) => + file.relativePath.startsWith(prefix), + ), + ) + : runtimeSources; + const filteredTestSources = monitor.includePathPrefixes + ? testSources.filter((file) => + monitor.includePathPrefixes.some((prefix) => + file.relativePath.startsWith(prefix), + ), + ) + : testSources; + const runtimeMatches = []; + const testMatches = []; + const violations = []; + + for (const file of filteredRuntimeSources) { + const counts = monitor.occurrences + .map((rule) => ({ + ...rule, + count: countOccurrences(file.sourceCode, rule.pattern), + })) + .filter((rule) => rule.count > 0); + + if (counts.length === 0) { + continue; + } + + runtimeMatches.push({ + relativePath: file.relativePath, + counts, + }); + + for (const rule of counts) { + if (rule.count > rule.maxCount) { + violations.push( + `${file.relativePath} -> ${rule.pattern} (${rule.count} > ${rule.maxCount})`, + ); + } + } + } + + for (const file of filteredTestSources) { + const counts = monitor.occurrences + .map((rule) => ({ + ...rule, + count: countOccurrences( + file.rawSourceCode ?? file.sourceCode, + rule.pattern, + ), + })) + .filter((rule) => rule.count > 0); + + if (counts.length === 0) { + continue; + } + + testMatches.push({ + relativePath: file.relativePath, + counts, + }); + } + + return { + ...monitor, + runtimeMatches, + testMatches, + violations, + }; +} + +function printImportReport(result) { + const status = getImportStatus(result); + + console.log( + `- [${status}] ${result.id} (${result.classification}):${result.description}`, + ); + console.log(` 目标文件:${result.targets.join(", ")}`); + console.log(` 允许引用:${result.allowedPaths.join(", ") || "无"}`); + if (result.missingTargets.length > 0) { + console.log(` 已删除目标:\n${formatPaths(result.missingTargets)}`); + } + console.log(` 实际引用:\n${formatPaths(result.references)}`); + console.log(` 测试引用:\n${formatPaths(result.testReferences)}`); + + if (result.violations.length > 0) { + console.log(` 违规引用:\n${formatPaths(result.violations)}`); + } +} + +function printCommandReport(result) { + const status = getCommandStatus(result); + + console.log( + `- [${status}] ${result.id} (${result.classification}):${result.description}`, + ); + console.log(` 命令:${result.commands.join(", ")}`); + console.log(` 允许引用:${result.allowedPaths.join(", ") || "无"}`); + + for (const command of result.commands) { + const references = result.referencesByCommand.get(command) ?? []; + const testReferences = result.testReferencesByCommand.get(command) ?? []; + console.log(` ${command}:\n${formatPaths(references)}`); + console.log(` ${command}(测试):\n${formatPaths(testReferences)}`); + } + + if (result.violations.length > 0) { + console.log(` 违规引用:\n${formatPaths(result.violations)}`); + } +} + +function printTextReport(result) { + const status = getTextStatus(result); + + console.log( + `- [${status}] ${result.id} (${result.classification}):${result.description}`, + ); + const keywords = [ + ...result.patterns, + ...(result.regexPatterns ?? []).map((pattern) => `regex:${pattern}`), + ]; + console.log(` 关键字:${keywords.join(", ")}`); + console.log(` 允许引用:${result.allowedPaths.join(", ") || "无"}`); + console.log(` 实际引用:\n${formatPaths(result.references)}`); + console.log(` 测试引用:\n${formatPaths(result.testReferences)}`); + + if (result.violations.length > 0) { + console.log(` 违规引用:\n${formatPaths(result.violations)}`); + } +} + +function printTextCountReport(result) { + const status = getTextCountStatus(result); + + console.log( + `- [${status}] ${result.id} (${result.classification}):${result.description}`, + ); + console.log( + ` 次数规则:${result.occurrences + .map((rule) => `${rule.pattern} <= ${rule.maxCount}`) + .join(";")}`, + ); + console.log( + ` 实际命中:\n${formatPaths( + result.runtimeMatches.map( + (item) => + `${item.relativePath} -> ${item.counts + .map((rule) => `${rule.pattern} (${rule.count})`) + .join(";")}`, + ), + )}`, + ); + console.log( + ` 测试命中:\n${formatPaths( + result.testMatches.map( + (item) => + `${item.relativePath} -> ${item.counts + .map((rule) => `${rule.pattern} (${rule.count})`) + .join(";")}`, + ), + )}`, + ); + + if (result.violations.length > 0) { + console.log(` 违规引用:\n${formatPaths(result.violations)}`); + } +} + +export function buildLegacySurfaceReport() { + const { runtimeSources, testSources } = collectSources(); + const { + runtimeSources: frontendRuntimeTextSources, + testSources: frontendTestTextSources, + } = collectTextSources(sourceRoots, sourceExtensions); + const { runtimeSources: rustRuntimeSources, testSources: rustTestSources } = + collectTextSources(rustSourceRoots, rustSourceExtensions); + const importResults = importSurfaceMonitors.map((monitor) => + evaluateImportMonitor(monitor, runtimeSources, testSources), + ); + const commandResults = commandSurfaceMonitors.map((monitor) => + evaluateCommandMonitor(monitor, runtimeSources, testSources), + ); + const frontendTextResults = frontendTextSurfaceMonitors.map((monitor) => + evaluateTextMonitor( + monitor, + frontendRuntimeTextSources, + frontendTestTextSources, + ), + ); + const rustTextResults = rustTextSurfaceMonitors.map((monitor) => + evaluateTextMonitor(monitor, rustRuntimeSources, rustTestSources), + ); + const rustTextCountResults = rustTextCountMonitors.map((monitor) => + evaluateTextCountMonitor(monitor, rustRuntimeSources, rustTestSources), + ); + const summary = buildLegacySurfaceSummary({ + importResults, + commandResults, + frontendTextResults, + rustTextResults, + rustTextCountResults, + }); + + return { + repoRoot, + runtimeSources, + testSources, + rustRuntimeSources, + rustTestSources, + importResults, + commandResults, + frontendTextResults, + rustTextResults, + rustTextCountResults, + ...summary, + }; +} + +export function toSerializableLegacySurfaceReport(report) { + return { + repoRoot: report.repoRoot, + summary: { + runtimeSourceCount: report.runtimeSources.length, + testSourceCount: report.testSources.length, + rustRuntimeSourceCount: report.rustRuntimeSources.length, + rustTestSourceCount: report.rustTestSources.length, + zeroReferenceCandidates: report.zeroReferenceCandidates, + classificationDriftCandidates: report.classificationDriftCandidates, + violations: report.violations, + }, + importResults: report.importResults, + commandResults: report.commandResults.map((result) => ({ + ...result, + referencesByCommand: serializeMapEntries(result.referencesByCommand), + testReferencesByCommand: serializeMapEntries( + result.testReferencesByCommand, + ), + })), + frontendTextResults: report.frontendTextResults, + rustTextResults: report.rustTextResults, + rustTextCountResults: report.rustTextCountResults, + }; +} + +export function printLegacySurfaceReport(report) { + console.log("[lime] legacy surface report"); + console.log(""); + console.log("## 入口引用"); + for (const result of report.importResults) { + printImportReport(result); + } + + console.log(""); + console.log("## 命令边界"); + for (const result of report.commandResults) { + printCommandReport(result); + } + + console.log(""); + console.log("## 前端护栏"); + for (const result of report.frontendTextResults) { + printTextReport(result); + } + + console.log(""); + console.log("## Rust 护栏"); + for (const result of report.rustTextResults) { + printTextReport(result); + } + for (const result of report.rustTextCountResults) { + printTextCountReport(result); + } + + console.log(""); + console.log("## 摘要"); + console.log(`- 扫描文件数:${report.runtimeSources.length}`); + console.log(`- 测试文件数:${report.testSources.length}`); + console.log(`- Rust 扫描文件数:${report.rustRuntimeSources.length}`); + console.log(`- Rust 测试文件数:${report.rustTestSources.length}`); + console.log(`- 零引用候选:${report.zeroReferenceCandidates.length}`); + for (const candidate of report.zeroReferenceCandidates) { + console.log(` - ${candidate}`); + } + console.log(`- 分类漂移候选:${report.classificationDriftCandidates.length}`); + for (const candidate of report.classificationDriftCandidates) { + console.log(` - ${candidate}`); + } + console.log(`- 边界违规:${report.violations.length}`); + for (const violation of report.violations) { + console.log(` - ${violation}`); + } +} diff --git a/scripts/lib/legacy-surface-report-core.test.ts b/scripts/lib/legacy-surface-report-core.test.ts new file mode 100644 index 000000000..9e7422155 --- /dev/null +++ b/scripts/lib/legacy-surface-report-core.test.ts @@ -0,0 +1,46 @@ +import { describe, expect, it } from "vitest"; + +import { toSerializableLegacySurfaceReport } from "./legacy-surface-report-core.mjs"; + +describe("legacy-surface-report-core", () => { + it("应把命令引用 Map 序列化成普通对象", () => { + const serialized = toSerializableLegacySurfaceReport({ + repoRoot: "/tmp/lime", + runtimeSources: [{ relativePath: "src/foo.ts" }], + testSources: [{ relativePath: "src/foo.test.ts" }], + rustRuntimeSources: [{ relativePath: "src-tauri/src/lib.rs" }], + rustTestSources: [], + importResults: [], + commandResults: [ + { + id: "agent-runtime", + classification: "current", + description: "runtime command", + commands: ["agent_runtime_submit_turn"], + allowedPaths: ["src/lib/api/agentRuntime.ts"], + referencesByCommand: new Map([ + ["agent_runtime_submit_turn", ["src/lib/api/agentRuntime.ts"]], + ]), + testReferencesByCommand: new Map([ + ["agent_runtime_submit_turn", ["src/lib/api/agentRuntime.test.ts"]], + ]), + violations: [], + }, + ], + frontendTextResults: [], + rustTextResults: [], + rustTextCountResults: [], + zeroReferenceCandidates: [], + classificationDriftCandidates: [], + violations: [], + }); + + expect(serialized.summary.runtimeSourceCount).toBe(1); + expect(serialized.commandResults[0].referencesByCommand).toEqual({ + agent_runtime_submit_turn: ["src/lib/api/agentRuntime.ts"], + }); + expect(serialized.commandResults[0].testReferencesByCommand).toEqual({ + agent_runtime_submit_turn: ["src/lib/api/agentRuntime.test.ts"], + }); + }); +}); diff --git a/scripts/lib/legacy-surface-report-summary.mjs b/scripts/lib/legacy-surface-report-summary.mjs new file mode 100644 index 000000000..3b5d3e731 --- /dev/null +++ b/scripts/lib/legacy-surface-report-summary.mjs @@ -0,0 +1,176 @@ +function flattenCommandReferences(result) { + return [...result.referencesByCommand.values()].flat(); +} + +export function getImportStatus(result) { + return result.violations.length > 0 + ? "违规" + : result.references.length === 0 && result.existingTargets.length === 0 + ? "已删除" + : result.references.length === 0 + ? "零引用" + : "受控"; +} + +export function getCommandStatus(result) { + const uniqueReferences = [ + ...new Set(flattenCommandReferences(result)), + ].sort(); + return result.violations.length > 0 + ? "违规" + : uniqueReferences.length === 0 + ? "零引用" + : "受控"; +} + +export function getTextStatus(result) { + return result.violations.length > 0 + ? "违规" + : result.references.length === 0 + ? "零引用" + : "受控"; +} + +export function getTextCountStatus(result) { + return result.violations.length > 0 + ? "违规" + : result.runtimeMatches.length === 0 + ? "零引用" + : "受控"; +} + +export function isStatusClassificationDrift(status, classification) { + return ( + (status === "已删除" || status === "零引用") && + classification !== "dead-candidate" + ); +} + +function collectClassificationDriftCandidates({ + importResults, + commandResults, + frontendTextResults, + rustTextResults, + rustTextCountResults, +}) { + return [ + ...importResults + .filter((result) => + isStatusClassificationDrift( + getImportStatus(result), + result.classification, + ), + ) + .map( + (result) => + `${result.id} -> ${result.classification} / ${getImportStatus(result)}`, + ), + ...commandResults + .filter((result) => + isStatusClassificationDrift( + getCommandStatus(result), + result.classification, + ), + ) + .map( + (result) => + `${result.id} -> ${result.classification} / ${getCommandStatus(result)}`, + ), + ...frontendTextResults + .filter((result) => + isStatusClassificationDrift( + getTextStatus(result), + result.classification, + ), + ) + .map( + (result) => + `${result.id} -> ${result.classification} / ${getTextStatus(result)}`, + ), + ...rustTextResults + .filter((result) => + isStatusClassificationDrift( + getTextStatus(result), + result.classification, + ), + ) + .map( + (result) => + `${result.id} -> ${result.classification} / ${getTextStatus(result)}`, + ), + ...rustTextCountResults + .filter((result) => + isStatusClassificationDrift( + getTextCountStatus(result), + result.classification, + ), + ) + .map( + (result) => + `${result.id} -> ${result.classification} / ${getTextCountStatus(result)}`, + ), + ]; +} + +function collectViolations({ + importResults, + commandResults, + frontendTextResults, + rustTextResults, + rustTextCountResults, +}) { + return [ + ...importResults.flatMap((result) => + result.violations.map((item) => `${result.id} -> ${item}`), + ), + ...commandResults.flatMap((result) => + result.violations.map((item) => `${result.id} -> ${item}`), + ), + ...frontendTextResults.flatMap((result) => + result.violations.map((item) => `${result.id} -> ${item}`), + ), + ...rustTextResults.flatMap((result) => + result.violations.map((item) => `${result.id} -> ${item}`), + ), + ...rustTextCountResults.flatMap((result) => + result.violations.map((item) => `${result.id} -> ${item}`), + ), + ]; +} + +export function buildLegacySurfaceSummary({ + importResults, + commandResults, + frontendTextResults, + rustTextResults, + rustTextCountResults, +}) { + const zeroReferenceCandidates = importResults + .filter( + (result) => + result.references.length === 0 && result.existingTargets.length > 0, + ) + .map((result) => `${result.id} (${result.description})`); + + return { + zeroReferenceCandidates, + classificationDriftCandidates: collectClassificationDriftCandidates({ + importResults, + commandResults, + frontendTextResults, + rustTextResults, + rustTextCountResults, + }), + violations: collectViolations({ + importResults, + commandResults, + frontendTextResults, + rustTextResults, + rustTextCountResults, + }), + }; +} + +export function serializeMapEntries(map) { + return Object.fromEntries(map.entries()); +} diff --git a/scripts/lib/legacy-surface-report-summary.test.ts b/scripts/lib/legacy-surface-report-summary.test.ts new file mode 100644 index 000000000..99e4a7d1e --- /dev/null +++ b/scripts/lib/legacy-surface-report-summary.test.ts @@ -0,0 +1,86 @@ +import { describe, expect, it } from "vitest"; + +import { + buildLegacySurfaceSummary, + getCommandStatus, + getImportStatus, + getTextCountStatus, + getTextStatus, +} from "./legacy-surface-report-summary.mjs"; + +describe("legacy-surface-report-summary", () => { + it("应正确判定各类 surface 状态", () => { + expect( + getImportStatus({ + violations: [], + references: [], + existingTargets: [], + }), + ).toBe("已删除"); + + expect( + getCommandStatus({ + violations: [], + referencesByCommand: new Map([["cmd", []]]), + }), + ).toBe("零引用"); + + expect( + getTextStatus({ + violations: [], + references: ["src/foo.ts"], + }), + ).toBe("受控"); + + expect( + getTextCountStatus({ + violations: [], + runtimeMatches: [], + }), + ).toBe("零引用"); + }); + + it("应汇总零引用候选、分类漂移与违规", () => { + const summary = buildLegacySurfaceSummary({ + importResults: [ + { + id: "import-monitor", + classification: "compat", + description: "import drift", + references: [], + existingTargets: ["src/legacy.ts"], + violations: [], + }, + ], + commandResults: [ + { + id: "command-monitor", + classification: "current", + referencesByCommand: new Map([["legacy_cmd", ["src/foo.ts"]]]), + violations: ["legacy_cmd -> src/foo.ts"], + }, + ], + frontendTextResults: [ + { + id: "frontend-monitor", + classification: "deprecated", + references: [], + violations: [], + }, + ], + rustTextResults: [], + rustTextCountResults: [], + }); + + expect(summary.zeroReferenceCandidates).toEqual([ + "import-monitor (import drift)", + ]); + expect(summary.classificationDriftCandidates).toEqual([ + "import-monitor -> compat / 零引用", + "frontend-monitor -> deprecated / 零引用", + ]); + expect(summary.violations).toEqual([ + "command-monitor -> legacy_cmd -> src/foo.ts", + ]); + }); +}); diff --git a/scripts/report-generated-slop.mjs b/scripts/report-generated-slop.mjs new file mode 100644 index 000000000..535723b95 --- /dev/null +++ b/scripts/report-generated-slop.mjs @@ -0,0 +1,293 @@ +#!/usr/bin/env node + +import { execFileSync } from "node:child_process"; +import fs from "node:fs"; +import path from "node:path"; +import process from "node:process"; +import { fileURLToPath } from "node:url"; + +import { + buildGeneratedSlopReport, + renderGeneratedSlopMarkdown, + renderGeneratedSlopText, +} from "./lib/generated-slop-report-core.mjs"; +import { buildLiveDocFreshnessReport } from "./check-doc-freshness.mjs"; +import { + buildLegacySurfaceReport, + toSerializableLegacySurfaceReport, +} from "./report-legacy-surfaces.mjs"; + +const TREND_REPORT_PATH = "scripts/harness-eval-trend-report.mjs"; + +export { + buildGeneratedSlopReport, + renderGeneratedSlopMarkdown, + renderGeneratedSlopText, +} from "./lib/generated-slop-report-core.mjs"; + +function parseArgs(argv) { + const result = { + format: "text", + help: false, + docFreshnessInput: "", + legacyInput: "", + outputJson: "", + outputMarkdown: "", + trendHistoryDir: "", + trendInput: "", + workspaceRoot: process.cwd(), + }; + + for (let index = 0; index < argv.length; index += 1) { + const arg = argv[index]; + + if (arg === "--trend-input" && argv[index + 1]) { + result.trendInput = String(argv[index + 1]).trim(); + index += 1; + continue; + } + + if (arg === "--doc-freshness-input" && argv[index + 1]) { + result.docFreshnessInput = String(argv[index + 1]).trim(); + index += 1; + continue; + } + + if (arg === "--legacy-input" && argv[index + 1]) { + result.legacyInput = String(argv[index + 1]).trim(); + index += 1; + continue; + } + + if (arg === "--trend-history-dir" && argv[index + 1]) { + result.trendHistoryDir = String(argv[index + 1]).trim(); + index += 1; + continue; + } + + if (arg === "--workspace-root" && argv[index + 1]) { + result.workspaceRoot = String(argv[index + 1]).trim(); + index += 1; + continue; + } + + if (arg === "--format" && argv[index + 1]) { + result.format = String(argv[index + 1]).trim(); + index += 1; + continue; + } + + if (arg === "--output-json" && argv[index + 1]) { + result.outputJson = String(argv[index + 1]).trim(); + index += 1; + continue; + } + + if (arg === "--output-markdown" && argv[index + 1]) { + result.outputMarkdown = String(argv[index + 1]).trim(); + index += 1; + continue; + } + + if (arg === "--help" || arg === "-h") { + result.help = true; + } + } + + return result; +} + +function printHelp() { + console.log(` +Lime Harness Cleanup / Slop Report + +用法: + node scripts/report-generated-slop.mjs + node scripts/report-generated-slop.mjs --trend-input "./tmp/harness-eval-trend.json" + node scripts/report-generated-slop.mjs --doc-freshness-input "./tmp/doc-freshness.json" + node scripts/report-generated-slop.mjs --legacy-input "./tmp/legacy-surface-report.json" + node scripts/report-generated-slop.mjs --trend-history-dir "./artifacts/history" + node scripts/report-generated-slop.mjs --output-json "./tmp/harness-cleanup-report.json" --output-markdown "./tmp/harness-cleanup-report.md" + +选项: + --trend-input PATH 显式指定 harness eval trend JSON + --doc-freshness-input PATH 显式指定 doc freshness JSON + --legacy-input PATH 显式指定 governance / legacy surface report JSON + --trend-history-dir PATH 透传给 trend report,用历史 summary 生成趋势 + --workspace-root PATH 未提供 trend-input 时,用该工作区生成当前 trend 报告 + --format FMT 标准输出格式:text | json | markdown + --output-json PATH 将 cleanup/slop 报告 JSON 写入指定路径 + --output-markdown PATH 将 cleanup/slop 报告 Markdown 写入指定路径 + -h, --help 显示帮助 +`); +} + +function resolvePath(baseDir, targetPath) { + return path.resolve(baseDir, targetPath); +} + +function ensureParentDirectory(filePath) { + fs.mkdirSync(path.dirname(filePath), { recursive: true }); +} + +function readJsonFile(filePath) { + return JSON.parse(fs.readFileSync(filePath, "utf8")); +} + +function detectDefaultTrendHistoryDir(repoRoot) { + const candidate = resolvePath(repoRoot, "./artifacts/history"); + if (!fs.existsSync(candidate)) { + return ""; + } + + const hasJson = fs + .readdirSync(candidate, { withFileTypes: true }) + .some((entry) => entry.isFile() && entry.name.endsWith(".json")); + + return hasJson ? candidate : ""; +} + +function buildTrendReport(repoRoot, options) { + if (options.trendInput) { + const resolvedPath = resolvePath(repoRoot, options.trendInput); + return { + report: readJsonFile(resolvedPath), + source: { + kind: "input-json", + path: resolvedPath, + }, + }; + } + + const nodeCommand = process.execPath; + const trendScriptPath = resolvePath(repoRoot, TREND_REPORT_PATH); + const args = [trendScriptPath, "--format", "json"]; + const effectiveHistoryDir = + options.trendHistoryDir || detectDefaultTrendHistoryDir(repoRoot); + + if (effectiveHistoryDir) { + args.push("--history-dir", effectiveHistoryDir); + } + + if (options.workspaceRoot) { + args.push("--workspace-root", options.workspaceRoot); + } + + const output = execFileSync(nodeCommand, args, { + cwd: repoRoot, + encoding: "utf8", + stdio: ["ignore", "pipe", "inherit"], + }); + + return { + report: JSON.parse(output), + source: { + kind: effectiveHistoryDir ? "generated-history" : "generated-current", + path: effectiveHistoryDir + ? resolvePath(repoRoot, effectiveHistoryDir) + : "", + }, + }; +} + +function buildGovernanceReport(repoRoot, options) { + if (options.legacyInput) { + const resolvedPath = resolvePath(repoRoot, options.legacyInput); + return { + report: readJsonFile(resolvedPath), + source: { + kind: "input-json", + path: resolvedPath, + }, + }; + } + + return { + report: toSerializableLegacySurfaceReport(buildLegacySurfaceReport()), + source: { + kind: "live-scan", + path: "", + }, + }; +} + +function buildDocFreshnessReport(repoRoot, options) { + if (options.docFreshnessInput) { + const resolvedPath = resolvePath(repoRoot, options.docFreshnessInput); + return { + report: readJsonFile(resolvedPath), + source: { + kind: "input-json", + path: resolvedPath, + }, + }; + } + + return { + report: buildLiveDocFreshnessReport(repoRoot), + source: { + kind: "live-scan", + path: "", + }, + }; +} + +function renderOutput(report, format) { + if (format === "json") { + return `${JSON.stringify(report, null, 2)}\n`; + } + + if (format === "markdown") { + return renderGeneratedSlopMarkdown(report); + } + + return renderGeneratedSlopText(report); +} + +function runGeneratedSlopReportCli() { + const options = parseArgs(process.argv.slice(2)); + if (options.help) { + printHelp(); + return; + } + + const repoRoot = process.cwd(); + const trendResult = buildTrendReport(repoRoot, options); + const docFreshnessResult = buildDocFreshnessReport(repoRoot, options); + const governanceResult = buildGovernanceReport(repoRoot, options); + const report = buildGeneratedSlopReport({ + repoRoot, + trendReport: trendResult.report, + docFreshnessReport: docFreshnessResult.report, + governanceReport: governanceResult.report, + sources: { + trend: trendResult.source, + docFreshness: docFreshnessResult.source, + governance: governanceResult.source, + }, + }); + + if (options.outputJson) { + const targetPath = resolvePath(repoRoot, options.outputJson); + ensureParentDirectory(targetPath); + fs.writeFileSync(targetPath, `${JSON.stringify(report, null, 2)}\n`, "utf8"); + console.log(`[lime] cleanup/slop report JSON: ${targetPath}`); + } + + if (options.outputMarkdown) { + const targetPath = resolvePath(repoRoot, options.outputMarkdown); + ensureParentDirectory(targetPath); + fs.writeFileSync(targetPath, renderGeneratedSlopMarkdown(report), "utf8"); + console.log(`[lime] cleanup/slop report Markdown: ${targetPath}`); + } + + process.stdout.write(renderOutput(report, options.format)); +} + +const isMainModule = + process.argv[1] && + path.resolve(process.argv[1]) === fileURLToPath(import.meta.url); + +if (isMainModule) { + runGeneratedSlopReportCli(); +} diff --git a/scripts/report-legacy-surfaces.mjs b/scripts/report-legacy-surfaces.mjs index f207da2c2..d41efa6cd 100644 --- a/scripts/report-legacy-surfaces.mjs +++ b/scripts/report-legacy-surfaces.mjs @@ -4,2114 +4,17 @@ import fs from "node:fs"; import path from "node:path"; import process from "node:process"; import { fileURLToPath } from "node:url"; -import agentCommandCatalog from "../src/lib/governance/agentCommandCatalog.json" with { type: "json" }; - -const repoRoot = path.resolve(process.cwd()); -const sourceRoots = ["src"]; -const sourceExtensions = new Set([ - ".ts", - ".tsx", - ".js", - ".jsx", - ".mjs", - ".cjs", -]); -const rustSourceRoots = ["src-tauri/src", "src-tauri/crates"]; -const rustSourceExtensions = new Set([".rs"]); -const ignoredDirs = new Set([ - "node_modules", - "dist", - "build", - "coverage", - "target", - ".git", - ".turbo", - ".next", -]); - -const agentLegacyCommandSurfaceMonitors = - agentCommandCatalog.legacyCommandSurfaceMonitors; - -const agentLegacyHelperSurfaceMonitors = ( - agentCommandCatalog.legacyHelperSurfaceMonitors ?? [] -).map(({ helpers, ...monitor }) => ({ - ...monitor, - patterns: helpers.map((helper) => `${helper}(`), -})); - -const importSurfaceMonitors = [ - { - id: "general-chat-root-entry", - classification: "dead-candidate", - description: "已删除的旧 general-chat 根导出入口", - targets: ["src/components/general-chat/index.ts"], - allowedPaths: [], - }, - { - id: "general-chat-page-entry", - classification: "dead-candidate", - description: "已删除的旧 general-chat 页面实现入口", - targets: ["src/components/general-chat/GeneralChatPage.tsx"], - allowedPaths: [], - }, - { - id: "general-chat-legacy-session-hook", - classification: "dead-candidate", - description: "旧 general-chat 会话兼容 Hook", - targets: ["src/components/general-chat/hooks/useSession.ts"], - allowedPaths: [], - }, - { - id: "general-chat-legacy-streaming-hook", - classification: "dead-candidate", - description: "已删除的旧 general-chat 流式兼容 Hook", - targets: ["src/components/general-chat/hooks/useStreaming.ts"], - allowedPaths: [], - }, - { - id: "general-chat-compat-gateway", - classification: "dead-candidate", - description: "general-chat compat API 网关", - targets: ["src/lib/api/generalChatCompat.ts"], - allowedPaths: [], - }, - { - id: "agent-compat-gateway", - classification: "dead-candidate", - description: "已删除的 Agent / Aster compat API 网关", - targets: ["src/lib/api/agentCompat.ts"], - allowedPaths: [], - }, - { - id: "agent-api-facade-entry", - classification: "dead-candidate", - description: "已删除 Agent API 门面入口", - targets: ["src/lib/api/agent.ts"], - allowedPaths: [], - }, - { - id: "agent-legacy-hook-entry", - classification: "dead-candidate", - description: "旧 Agent Chat Hook 入口", - targets: ["src/components/agent/chat/hooks/useAgentChat.ts"], - allowedPaths: [], - }, - { - id: "heartbeat-api-gateway", - classification: "dead-candidate", - description: "已删除的旧 heartbeat 前端 API 入口", - targets: ["src/lib/api/heartbeat.ts"], - allowedPaths: [], - }, - { - id: "heartbeat-settings-page-entry", - classification: "dead-candidate", - description: "已删除的旧 heartbeat 设置页入口", - targets: ["src/components/settings-v2/system/heartbeat/index.tsx"], - allowedPaths: [], - }, - { - id: "assistant-settings-page-entry", - classification: "dead-candidate", - description: "已删除的旧助理服务设置页入口", - targets: ["src/components/settings-v2/agent/assistant/index.tsx"], - allowedPaths: [], - }, - { - id: "stores-root-barrel-entry", - classification: "dead-candidate", - description: "旧 stores 根 barrel 入口", - targets: ["src/stores/index.ts"], - allowedPaths: [], - }, - { - id: "agent-legacy-store-entry", - classification: "dead-candidate", - description: "旧 Agent Zustand store 入口", - targets: ["src/stores/agentStore.ts"], - allowedPaths: [], - }, - { - id: "use-unified-chat-compat-hook", - classification: "dead-candidate", - description: "useUnifiedChat compat Hook 入口", - targets: ["src/hooks/useUnifiedChat.ts"], - allowedPaths: [], - }, - { - id: "unified-chat-compat-gateway", - classification: "dead-candidate", - description: "unified-chat compat API 网关", - targets: ["src/lib/api/unified-chat.ts"], - allowedPaths: [], - }, - { - id: "three-stage-workflow-hook-entry", - classification: "dead-candidate", - description: "旧 three-stage workflow React Hook 入口", - targets: ["src/hooks/useThreeStageWorkflow.ts"], - allowedPaths: [], - }, - { - id: "three-stage-workflow-manager-entry", - classification: "dead-candidate", - description: "旧 three-stage workflow 管理器入口", - targets: ["src/lib/workflow/threeStageWorkflow.ts"], - allowedPaths: [], - }, - { - id: "tool-hooks-api-gateway", - classification: "dead-candidate", - description: "旧 tool hooks 前端 API 网关", - targets: ["src/lib/api/toolHooks.ts"], - allowedPaths: [], - }, - { - id: "context-memory-legacy-api-gateway", - classification: "dead-candidate", - description: "旧 context memory 前端 API 网关", - targets: ["src/lib/api/contextMemory.ts"], - allowedPaths: [], - }, - { - id: "team-subagent-scheduler-hook", - classification: "compat", - description: "旧 SubAgent scheduler Hook 只允许停留在 compat 展示层", - targets: ["src/hooks/useSubAgentScheduler.ts"], - allowedPaths: [ - "src/components/agent/chat/hooks/useCompatSubagentRuntime.ts", - "src/components/subagent/SubAgentProgress.tsx", - "src/components/subagent/index.ts", - ], - }, - { - id: "team-subagent-scheduler-api", - classification: "compat", - description: "旧 SubAgent scheduler API 只允许被 compat Hook 与降级展示层引用", - targets: ["src/lib/api/subAgentScheduler.ts"], - allowedPaths: [ - "src/hooks/useSubAgentScheduler.ts", - "src/components/agent/chat/utils/compatSubagentRuntime.ts", - ], - }, -]; - -const commandSurfaceMonitors = [ - ...agentLegacyCommandSurfaceMonitors, - { - id: "general-chat-compat-commands", - classification: "dead-candidate", - description: "已零引用的 general_chat compat 命令前端边界", - commands: [ - "general_chat_get_session", - "general_chat_list_sessions", - "general_chat_create_session", - "general_chat_delete_session", - "general_chat_rename_session", - "general_chat_get_messages", - ], - allowedPaths: [], - }, - { - id: "conversation-memory-legacy-commands", - classification: "dead-candidate", - description: "已零引用的旧 conversation memory 命令前端边界", - commands: [ - "get_conversation_memory_overview", - "get_conversation_memory_stats", - "request_conversation_memory_analysis", - "cleanup_conversation_memory", - ], - allowedPaths: [], - }, - { - id: "context-memory-legacy-commands", - classification: "dead-candidate", - description: "旧 context memory 命令前端边界", - commands: [ - "save_memory_entry", - "get_session_memories", - "get_memory_context", - "record_error", - "should_avoid_operation", - "mark_error_resolved", - "get_memory_stats", - "cleanup_expired_memories", - ], - allowedPaths: [], - }, - { - id: "chat-compat-commands", - classification: "dead-candidate", - description: "chat_* compat 命令前端边界", - commands: [ - "chat_create_session", - "chat_list_sessions", - "chat_get_session", - "chat_delete_session", - "chat_rename_session", - "chat_get_messages", - "chat_send_message", - "chat_stop_generation", - "chat_configure_provider", - ], - allowedPaths: [], - }, - { - id: "prompt-switch-legacy-command", - classification: "dead-candidate", - description: "已零引用的旧 prompt 切换命令前端边界", - commands: ["switch_prompt"], - allowedPaths: [], - }, - { - id: "api-key-legacy-migration-commands", - classification: "dead-candidate", - description: "已零引用的旧 API Key 迁移命令前端边界", - commands: [ - "get_legacy_api_key_credentials", - "migrate_legacy_api_key_credentials", - "delete_legacy_api_key_credential", - ], - allowedPaths: [], - }, - { - id: "heartbeat-legacy-commands", - classification: "dead-candidate", - description: "已零引用的旧 heartbeat 命令前端边界", - commands: [ - "get_heartbeat_config", - "update_heartbeat_config", - "get_heartbeat_status", - "get_heartbeat_tasks", - "add_heartbeat_task", - "delete_heartbeat_task", - "update_heartbeat_task", - "get_heartbeat_history", - "get_heartbeat_execution_detail", - "get_heartbeat_task_health", - "deliver_heartbeat_task_health_alerts", - "trigger_heartbeat_now", - "get_task_templates", - "apply_task_template", - "generate_content_creator_tasks", - "preview_heartbeat_schedule", - "validate_heartbeat_schedule", - ], - allowedPaths: [], - }, - { - id: "tool-hooks-legacy-commands", - classification: "dead-candidate", - description: "旧 tool hooks 命令前端边界", - commands: [ - "execute_hooks", - "add_hook_rule", - "remove_hook_rule", - "toggle_hook_rule", - "get_hook_rules", - "get_hook_execution_stats", - "clear_hook_execution_stats", - ], - allowedPaths: [], - }, - { - id: "team-subagent-scheduler-commands", - classification: "compat", - description: "旧 execute_subagent_tasks/cancel_subagent_tasks 只允许通过 compat API 网关暴露", - commands: ["execute_subagent_tasks", "cancel_subagent_tasks"], - allowedPaths: ["src/lib/api/subAgentScheduler.ts"], - }, -]; - -const frontendTextSurfaceMonitors = [ - ...agentLegacyHelperSurfaceMonitors, - { - id: "frontend-subagent-scheduler-event-bus", - classification: "compat", - description: "旧 subagent scheduler 事件名只允许 compat Hook 持有", - patterns: ["subagent-scheduler-event"], - allowedPaths: ["src/hooks/useSubAgentScheduler.ts"], - }, - { - id: "frontend-assistant-settings-surfaces", - classification: "dead-candidate", - description: "已零引用的前端助理服务设置页与配置面回流", - patterns: [ - "SettingsTabs.Assistant", - "settings.tab.assistant", - "AssistantSettings", - "AssistantConfig", - "default_assistant_id", - "custom_assistants", - "show_suggestions", - "auto_select", - ], - allowedPaths: [], - }, - { - id: "stores-root-barrel-imports", - classification: "dead-candidate", - description: "已零引用的 @/stores 根 barrel 回流", - patterns: ['from "@/stores"', "from '@/stores'"], - allowedPaths: [], - }, -]; - -const rustTextSurfaceMonitors = [ - { - id: "rust-subagent-scheduler-event-bus", - classification: "compat", - description: "旧 subagent scheduler 事件名只允许 compat Rust emitter 持有", - patterns: ["subagent-scheduler-event"], - allowedPaths: ["src-tauri/src/agent/subagent_scheduler.rs"], - }, - { - id: "rust-general-chat-dao", - classification: "dead-candidate", - description: "已零引用的 Rust 业务层 direct GeneralChatDao 依赖", - patterns: ["GeneralChatDao", "database::dao::general_chat"], - allowedPaths: [], - }, - { - id: "rust-legacy-general-tables", - classification: "compat", - description: "Rust runtime direct legacy general 表访问", - patterns: ["general_chat_sessions", "general_chat_messages"], - allowedPaths: [ - "src-tauri/crates/core/src/app_paths.rs", - "src-tauri/crates/core/src/database/migration/general_chat_migration.rs", - "src-tauri/crates/core/src/database/migration.rs", - "src-tauri/crates/core/src/database/schema.rs", - ], - }, - { - id: "rust-legacy-general-helper-usage", - classification: "dead-candidate", - description: "Rust runtime pending general raw helper 回流", - patterns: [ - "load_pending_general_session_messages_raw", - "load_pending_general_messages_raw", - "count_pending_general_sessions_raw", - "count_pending_general_messages_raw", - "sum_pending_general_message_chars_raw", - "load_legacy_general_session_messages", - "load_unmigrated_legacy_general_messages", - "count_unmigrated_legacy_general_sessions", - "count_unmigrated_legacy_general_messages", - "sum_unmigrated_legacy_general_message_chars", - ], - allowedPaths: [], - }, - { - id: "rust-pending-general-wrapper-usage", - classification: "dead-candidate", - description: "Rust 业务层 pending general 兼容 wrapper 回流", - patterns: [ - "load_pending_general_messages(", - "load_pending_general_session_messages(", - "count_pending_general_sessions(", - "count_pending_general_messages(", - "sum_pending_general_message_chars(", - "summarize_pending_general(", - ], - includePathPrefixes: ["src-tauri/src", "src-tauri/crates"], - allowedPaths: [], - }, - { - id: "rust-legacy-general-module-imports", - classification: "dead-candidate", - description: - "已零引用的 Rust 外部模块 direct pending/legacy general 子模块", - patterns: [ - "crate::database::legacy_general_chat::", - "lime_core::database::legacy_general_chat::", - "crate::database::pending_general_chat::", - "lime_core::database::pending_general_chat::", - ], - allowedPaths: [], - }, - { - id: "rust-general-migration-flag-runtime-leak", - classification: "deprecated", - description: "Rust 业务层重新直接判断 general 迁移完成标记", - patterns: [ - "migration::is_general_chat_migration_completed", - "is_general_chat_migration_completed(", - ], - allowedPaths: [ - "src-tauri/crates/core/src/database/migration/general_chat_migration.rs", - ], - }, - { - id: "rust-services-crate-general-chat-compat", - classification: "dead-candidate", - description: "已零引用的 services crate general_chat 兼容壳回流", - patterns: [ - "use crate::general_chat::", - "use crate::general_chat::{", - "crate::general_chat::SessionService", - ], - includePathPrefixes: ["src-tauri/crates/services/src"], - allowedPaths: [], - }, - { - id: "rust-cross-crate-general-chat-compat", - classification: "dead-candidate", - description: "已零引用的跨 crate lime_services::general_chat 兼容壳回流", - patterns: ["lime_services::general_chat::"], - allowedPaths: [], - }, - { - id: "rust-provider-pool-legacy-selector", - classification: "dead-candidate", - description: "已零引用的 provider pool legacy 凭证选择兼容方法", - patterns: ["select_credential_with_fallback_legacy"], - allowedPaths: [], - }, - { - id: "rust-memory-legacy-command-shells", - classification: "dead-candidate", - description: "已零引用的旧 conversation memory Rust 命令壳回流", - patterns: [ - "get_conversation_memory_stats", - "get_conversation_memory_overview", - "request_conversation_memory_analysis", - "cleanup_conversation_memory", - ], - includePathPrefixes: ["src-tauri/src"], - allowedPaths: [], - }, - { - id: "rust-memory-profile-prompt-helper-leak", - classification: "deprecated", - description: - "低层 build_memory_profile_prompt helper 泄漏到统一装配边界之外", - patterns: ["build_memory_profile_prompt("], - includePathPrefixes: ["src-tauri/src"], - allowedPaths: ["src-tauri/src/services/memory_profile_prompt_service.rs"], - }, - { - id: "rust-memory-sources-prompt-helper-leak", - classification: "deprecated", - description: - "低层 build_memory_sources_prompt helper 泄漏到统一装配边界之外", - patterns: ["build_memory_sources_prompt("], - includePathPrefixes: ["src-tauri/src"], - allowedPaths: [ - "src-tauri/src/services/memory_profile_prompt_service.rs", - "src-tauri/src/services/memory_source_resolver_service.rs", - ], - }, - { - id: "rust-project-session-config-helper-leak", - classification: "dead-candidate", - description: "旧 create_session_config_with_project helper 回流", - patterns: ["create_session_config_with_project("], - allowedPaths: [], - }, - { - id: "rust-unified-chat-command-module-leak", - classification: "dead-candidate", - description: "Rust unified_chat compat 命令模块重新回到 commands 编译图", - patterns: ["pub mod unified_chat_cmd;"], - includePathPrefixes: ["src-tauri/src/commands"], - allowedPaths: [], - }, - { - id: "rust-tool-hooks-command-module-leak", - classification: "dead-candidate", - description: "Rust tool_hooks 旧命令模块重新回到 commands 编译图", - patterns: ["pub mod tool_hooks;"], - includePathPrefixes: ["src-tauri/src/commands"], - allowedPaths: [], - }, - { - id: "rust-tool-hooks-service-module-leak", - classification: "dead-candidate", - description: "services crate 旧 tool_hooks_service 模块重新回到编译图", - patterns: ["pub mod tool_hooks_service;"], - includePathPrefixes: ["src-tauri/crates/services/src"], - allowedPaths: [], - }, - { - id: "rust-three-stage-workflow-tool-leak", - classification: "dead-candidate", - description: "legacy three_stage_workflow 工具名重新回到 Lime Rust 编译图", - patterns: ['"three_stage_workflow"', "three_stage_workflow"], - includePathPrefixes: ["src-tauri/src", "src-tauri/crates"], - allowedPaths: [], - }, - { - id: "rust-skill-workflow-tauri-orchestration-leak", - classification: "dead-candidate", - description: "已零引用的 skill workflow 执行主链回流到 Tauri skills 模块", - patterns: [ - "SessionConfigBuilder::new(", - "convert_agent_event(", - "WriteArtifactEventEmitter::new(", - ".reply(", - ], - includePathPrefixes: ["src-tauri/src/skills"], - allowedPaths: [], - }, - { - id: "rust-skill-prompt-command-orchestration-leak", - classification: "dead-candidate", - description: "已零引用的 skill prompt 执行主链回流到 skill_exec_cmd 命令层", - patterns: [ - "SessionConfigBuilder::new(", - "convert_agent_event(", - "WriteArtifactEventEmitter::new(", - ".reply(", - ], - includePathPrefixes: ["src-tauri/src/commands/skill_exec_cmd.rs"], - allowedPaths: [], - }, - { - id: "rust-skill-runtime-command-bootstrap-leak", - classification: "dead-candidate", - description: - "已零引用的 skill runtime 准备与 provider fallback 回流到 skill_exec_cmd 命令层", - patterns: [ - "ensure_browser_mcp_tools_registered(", - "ensure_social_image_tool_registered(", - "ensure_creation_task_tools_registered(", - "build_memory_profile_prompt(", - ".configure_provider_from_pool(", - "TauriExecutionCallback::new(", - ], - includePathPrefixes: ["src-tauri/src/commands/skill_exec_cmd.rs"], - allowedPaths: [], - }, - { - id: "rust-skill-catalog-command-leak", - classification: "dead-candidate", - description: - "已零引用的 skill catalog 枚举与详情装配回流到 skill_exec_cmd 命令层", - patterns: [ - "get_skill_roots(", - "load_skills_from_directory(", - "find_skill_by_name(", - "invalid_skill_message(", - "load_skill_from_file(", - "parse_skill_frontmatter(", - "parse_allowed_tools(", - "parse_boolean(", - ], - includePathPrefixes: ["src-tauri/src/commands/skill_exec_cmd.rs"], - allowedPaths: [], - }, - { - id: "rust-skill-mode-branch-command-leak", - classification: "dead-candidate", - description: - "已零引用的 skill execution_mode 分支回流到 skill_exec_cmd 命令层", - patterns: [ - 'skill.execution_mode == "workflow"', - "!skill.workflow_steps.is_empty()", - "execute_skill_workflow(", - "execute_skill_prompt(", - ], - includePathPrefixes: ["src-tauri/src/commands/skill_exec_cmd.rs"], - allowedPaths: [], - }, - { - id: "rust-request-tool-policy-compat-service", - classification: "dead-candidate", - description: "已零引用的 request_tool_policy 旧服务壳回流", - patterns: [ - "crate::services::request_tool_policy_prompt_service::", - "lime_lib::services::request_tool_policy_prompt_service::", - "services::request_tool_policy_prompt_service::", - ], - allowedPaths: [], - }, - { - id: "rust-service-agent-table-query-leak", - classification: "dead-candidate", - description: - "已零引用的 Tauri service 层 direct agent_sessions/agent_messages 查询回流", - patterns: [ - "FROM agent_sessions s", - "FROM agent_messages m", - "JOIN agent_sessions s ON s.id = m.session_id", - ], - includePathPrefixes: ["src-tauri/src/services"], - allowedPaths: [], - }, - { - id: "rust-service-model-usage-table-query-leak", - classification: "dead-candidate", - description: - "已零引用的 Tauri service 层 direct model_usage_stats 查询回流", - patterns: [ - "FROM model_usage_stats", - "SELECT COUNT(*) FROM model_usage_stats", - ], - includePathPrefixes: ["src-tauri/src/services"], - allowedPaths: [], - }, - { - id: "rust-dev-bridge-unified-memory-sql-leak", - classification: "dead-candidate", - description: "已零引用的 DevBridge unified_memory SQL 绕路回流", - patterns: [ - "FROM unified_memory", - "INSERT INTO unified_memory", - "UPDATE unified_memory", - "DELETE FROM unified_memory", - ], - includePathPrefixes: ["src-tauri/src/dev_bridge/dispatcher/memory.rs"], - allowedPaths: [], - }, - { - id: "rust-agent-runtime-legacy-queue-table-leak", - classification: "deprecated", - description: "legacy runtime queue 表名从数据库迁移边界向外扩散", - patterns: ["agent_runtime_queued_turns"], - includePathPrefixes: ["src-tauri/src", "src-tauri/crates"], - allowedPaths: [ - "src-tauri/crates/core/src/database/agent_runtime_queue_repository.rs", - ], - }, - { - id: "rust-agent-runtime-legacy-queue-migration-leak", - classification: "deprecated", - description: - "legacy runtime queue 启动迁移 helper 只允许停留在 aster runtime support 边界", - patterns: ["migrate_legacy_runtime_queue_to_aster_store("], - includePathPrefixes: ["src-tauri/src", "src-tauri/crates"], - allowedPaths: ["src-tauri/crates/agent/src/aster_runtime_support.rs"], - }, - { - id: "rust-agent-session-legacy-todo-state-leak", - classification: "dead-candidate", - description: "已零引用的 Lime 业务层 direct legacy TodoState 读取回流", - patterns: ["TodoState::from_extension_data(", "TodoState::new("], - includePathPrefixes: ["src-tauri/src", "src-tauri/crates"], - allowedPaths: [], - }, - { - id: "rust-agent-session-structured-todo-helper-bypass", - classification: "dead-candidate", - description: - "已零引用的 Lime 业务层绕过 unified todo helper 直接读取 TodoListState", - patterns: [ - "TodoListState::from_extension_data(", - "TodoListState::from_markdown(", - ], - includePathPrefixes: ["src-tauri/src", "src-tauri/crates"], - allowedPaths: [], - }, - { - id: "rust-services-default-workspace-query-leak", - classification: "dead-candidate", - description: - "已零引用的 services crate direct 默认 workspace root 查询回流", - patterns: ["SELECT root_path FROM workspaces WHERE is_default = 1 LIMIT 1"], - includePathPrefixes: ["src-tauri/crates/services/src"], - allowedPaths: [], - }, - { - id: "rust-agent-session-direct-record-access", - classification: "deprecated", - description: "Rust 上层模块 direct Agent session 记录与消息读写回流", - patterns: [ - "AgentDao::get_session(", - "AgentDao::get_session_with_messages(", - "AgentDao::list_sessions(", - "AgentDao::list_session_overviews(", - "AgentDao::get_message_count(", - "AgentDao::get_messages(", - "AgentDao::get_session_overview(", - "AgentDao::session_exists(", - "AgentDao::update_title(", - "AgentDao::update_session_time(", - "AgentDao::rename_session(", - "AgentDao::update_working_dir(", - "AgentDao::update_execution_strategy(", - ], - allowedPaths: [ - "src-tauri/crates/core/src/database/agent_session_repository.rs", - ], - }, - { - id: "rust-agent-session-direct-delete", - classification: "dead-candidate", - description: "已零引用的 Rust 业务层 direct AgentDao::delete_session 回流", - patterns: ["AgentDao::delete_session("], - allowedPaths: [], - }, - { - id: "rust-agent-session-direct-create", - classification: "deprecated", - description: "Rust 业务层 direct AgentDao::create_session 回流", - patterns: ["AgentDao::create_session("], - allowedPaths: [ - "src-tauri/crates/core/src/database/agent_session_repository.rs", - ], - }, - { - id: "rust-agent-dao-row-type-leak", - classification: "dead-candidate", - description: "agent crate 重新泄漏 core DAO row 类型", - patterns: ["AgentSessionOverviewRow"], - includePathPrefixes: ["src-tauri/crates/agent/src"], - allowedPaths: [], - }, - { - id: "rust-agent-workspace-query-leak", - classification: "dead-candidate", - description: "agent runtime direct workspaces 绑定查询回流", - patterns: ["SELECT id FROM workspaces WHERE root_path = ? LIMIT 1"], - includePathPrefixes: ["src-tauri/crates/agent/src"], - allowedPaths: [], - }, - { - id: "rust-agent-session-record-repository-module-leak", - classification: "dead-candidate", - description: "lime-agent 本地 session_record_repository 壳重新回到编译图", - patterns: ["mod session_record_repository;"], - includePathPrefixes: ["src-tauri/crates/agent/src/lib.rs"], - allowedPaths: [], - }, - { - id: "rust-agent-session-compat-surface", - classification: "dead-candidate", - description: "agent crate 旧 compat session API 回流", - patterns: [ - "CompatSessionInfo", - "list_compat_sessions_sync(", - "get_compat_session_sync(", - ], - includePathPrefixes: ["src-tauri/crates/agent/src"], - allowedPaths: [], - }, - { - id: "rust-agent-session-store-public-module-leak", - classification: "dead-candidate", - description: "lime-agent 重新对 crate 外暴露 session_store 模块", - patterns: ["pub mod session_store;"], - includePathPrefixes: ["src-tauri/crates/agent/src/lib.rs"], - allowedPaths: [], - }, - { - id: "rust-agent-session-store-direct-module-usage", - classification: "dead-candidate", - description: "上层模块重新 direct 依赖 lime_agent::session_store 模块路径", - patterns: ["lime_agent::session_store::"], - includePathPrefixes: ["src-tauri/src"], - allowedPaths: [], - }, - { - id: "rust-agent-session-get-direct-read", - classification: "deprecated", - description: "session get_session 直读只允许统一 session_query helper 持有", - patterns: ["SessionManager::get_session("], - includePathPrefixes: ["src-tauri/src", "src-tauri/crates"], - allowedPaths: ["src-tauri/crates/agent/src/session_query.rs"], - }, - { - id: "rust-agent-subagent-child-session-direct-read", - classification: "deprecated", - description: "subagent child session 列表直读只允许统一 session_query helper 持有", - patterns: ["list_subagent_child_sessions("], - includePathPrefixes: ["src-tauri/src", "src-tauri/crates"], - allowedPaths: ["src-tauri/crates/agent/src/session_query.rs"], - }, - { - id: "rust-agent-subagent-session-list-direct-read", - classification: "deprecated", - description: "subagent 全量 session 列表直读只允许统一 session_query helper 持有", - patterns: ["list_subagent_sessions_with_metadata("], - includePathPrefixes: ["src-tauri/src", "src-tauri/crates"], - allowedPaths: ["src-tauri/crates/agent/src/session_query.rs"], - }, - { - id: "rust-agent-subagent-metadata-direct-read", - classification: "deprecated", - description: "subagent metadata 直读只允许 query 与 session_store 投影边界持有", - patterns: ["resolve_subagent_session_metadata("], - includePathPrefixes: ["src-tauri/src", "src-tauri/crates"], - allowedPaths: [ - "src-tauri/crates/agent/src/session_query.rs", - "src-tauri/crates/agent/src/session_store.rs", - ], - }, - { - id: "rust-agent-session-extension-data-direct-update", - classification: "deprecated", - description: - "session extension_data 直写 builder 只允许统一 session_update helper 持有", - patterns: [], - regexPatterns: [ - String.raw`SessionManager::update_session\([\s\S]*?\)\s*[\s\S]*?\.extension_data\([\s\S]*?\)\s*[\s\S]*?\.apply\(\)`, - ], - includePathPrefixes: ["src-tauri/src", "src-tauri/crates"], - allowedPaths: ["src-tauri/crates/agent/src/session_update.rs"], - }, - { - id: "rust-agent-session-update-direct-call", - classification: "deprecated", - description: "session update_session 直写只允许统一 session_update helper 持有", - patterns: ["SessionManager::update_session("], - includePathPrefixes: ["src-tauri/src", "src-tauri/crates"], - allowedPaths: ["src-tauri/crates/agent/src/session_update.rs"], - }, - { - id: "rust-agent-session-compaction-metrics-direct-update", - classification: "deprecated", - description: - "session compaction token 指标直写 builder 只允许统一 session_update helper 持有", - patterns: [], - regexPatterns: [ - String.raw`SessionManager::update_session\([\s\S]*?\)\s*[\s\S]*?\.schedule_id\([\s\S]*?\)\s*[\s\S]*?\.total_tokens\([\s\S]*?\)\s*[\s\S]*?\.input_tokens\([\s\S]*?\)\s*[\s\S]*?\.output_tokens\([\s\S]*?\)\s*[\s\S]*?\.accumulated_total_tokens\([\s\S]*?\)\s*[\s\S]*?\.accumulated_input_tokens\([\s\S]*?\)\s*[\s\S]*?\.accumulated_output_tokens\([\s\S]*?\)\s*[\s\S]*?\.apply\(\)`, - ], - includePathPrefixes: ["src-tauri/src", "src-tauri/crates"], - allowedPaths: ["src-tauri/crates/agent/src/session_update.rs"], - }, - { - id: "rust-agent-session-replace-conversation-direct-update", - classification: "deprecated", - description: - "session conversation 直写 replace_conversation 只允许统一 session_update helper 持有", - patterns: ["SessionManager::replace_conversation("], - includePathPrefixes: ["src-tauri/src", "src-tauri/crates"], - allowedPaths: ["src-tauri/crates/agent/src/session_update.rs"], - }, - { - id: "rust-agent-session-create-direct-update", - classification: "deprecated", - description: - "session create_session 直写只允许统一 session_update helper 持有", - patterns: ["SessionManager::create_session("], - includePathPrefixes: ["src-tauri/src", "src-tauri/crates"], - allowedPaths: ["src-tauri/crates/agent/src/session_update.rs"], - }, - { - id: "rust-agent-session-delete-direct-update", - classification: "dead-candidate", - description: "已零引用的 session delete_session 直写回流", - patterns: ["SessionManager::delete_session("], - includePathPrefixes: ["src-tauri/src", "src-tauri/crates"], - allowedPaths: [], - }, - { - id: "rust-agent-session-record-create-api-leak", - classification: "dead-candidate", - description: "lime-agent crate 根重新暴露内部 session record 创建 API", - patterns: ["create_session_record_sync,", "CreateSessionRecordInput,"], - includePathPrefixes: ["src-tauri/crates/agent/src/lib.rs"], - allowedPaths: [], - }, - { - id: "rust-agent-command-subagent-metadata-direct-read", - classification: "dead-candidate", - description: "命令层不再 direct 解析 subagent metadata,统一向 lime_agent 读取边界收敛", - patterns: ["resolve_subagent_session_metadata("], - includePathPrefixes: ["src-tauri/src/commands/aster_agent_cmd"], - allowedPaths: [], - }, - { - id: "rust-agent-tool-permission-public-module-leak", - classification: "dead-candidate", - description: "lime-agent 重新对 crate 外暴露旧 tool_permissions 模块", - patterns: ["pub mod tool_permissions;"], - includePathPrefixes: ["src-tauri/crates/agent/src/lib.rs"], - allowedPaths: [], - }, - { - id: "rust-agent-shell-security-public-module-leak", - classification: "dead-candidate", - description: "lime-agent 重新对 crate 外暴露旧 shell_security 模块", - patterns: ["pub mod shell_security;"], - includePathPrefixes: ["src-tauri/crates/agent/src/lib.rs"], - allowedPaths: [], - }, - { - id: "rust-agent-tool-permission-module-compiled-leak", - classification: "dead-candidate", - description: "旧 tool_permissions 模块重新回到 lime-agent lib.rs 编译图", - patterns: ["mod tool_permissions;"], - includePathPrefixes: ["src-tauri/crates/agent/src/lib.rs"], - allowedPaths: [], - }, - { - id: "rust-agent-shell-security-module-compiled-leak", - classification: "dead-candidate", - description: "旧 shell_security 模块重新回到 lime-agent lib.rs 编译图", - patterns: ["mod shell_security;"], - includePathPrefixes: ["src-tauri/crates/agent/src/lib.rs"], - allowedPaths: [], - }, - { - id: "rust-agent-tool-permission-root-export-leak", - classification: "dead-candidate", - description: "lime-agent crate 根重新暴露旧 tool_permissions 类型出口", - patterns: ["pub use tool_permissions::"], - includePathPrefixes: ["src-tauri/crates/agent/src/lib.rs"], - allowedPaths: [], - }, - { - id: "rust-agent-shell-security-root-export-leak", - classification: "dead-candidate", - description: "lime-agent crate 根重新暴露旧 shell_security 类型出口", - patterns: ["pub use shell_security::"], - includePathPrefixes: ["src-tauri/crates/agent/src/lib.rs"], - allowedPaths: [], - }, - { - id: "rust-agent-tool-permission-direct-module-usage", - classification: "dead-candidate", - description: - "上层模块重新 direct 依赖 lime_agent::tool_permissions 模块路径", - patterns: ["lime_agent::tool_permissions::"], - includePathPrefixes: ["src-tauri/src", "src-tauri/crates"], - allowedPaths: [], - }, - { - id: "rust-agent-shell-security-direct-module-usage", - classification: "dead-candidate", - description: "上层模块重新 direct 依赖 lime_agent::shell_security 模块路径", - patterns: ["lime_agent::shell_security::"], - includePathPrefixes: ["src-tauri/src", "src-tauri/crates"], - allowedPaths: [], - }, - { - id: "rust-agent-tool-permission-root-type-usage", - classification: "dead-candidate", - description: "上层模块重新 direct 依赖 lime_agent 根导出的旧权限类型", - patterns: [ - "lime_agent::DynamicPermissionCheck", - "lime_agent::PermissionBehavior", - "lime_agent::ShellSecurityChecker", - ], - includePathPrefixes: ["src-tauri/src", "src-tauri/crates"], - allowedPaths: [], - }, - { - id: "rust-agent-tool-permission-internal-module-usage", - classification: "dead-candidate", - description: - "lime-agent 内部除兼容壳外重新扩散 crate::tool_permissions 模块依赖", - patterns: ["crate::tool_permissions::"], - includePathPrefixes: ["src-tauri/crates/agent/src"], - allowedPaths: ["src-tauri/crates/agent/src/shell_security.rs"], - }, - { - id: "rust-agent-integration-public-module-leak", - classification: "dead-candidate", - description: "agent 集成模块重新对外暴露 integration 模块路径", - patterns: ["pub mod integration;"], - includePathPrefixes: ["src-tauri/src/agent/mod.rs"], - allowedPaths: [], - }, - { - id: "rust-agent-integration-module-compiled-leak", - classification: "dead-candidate", - description: "已退出编译图的 agent integration 模块重新回到 agent 根模块", - patterns: ["mod integration;"], - includePathPrefixes: ["src-tauri/src/agent/mod.rs"], - allowedPaths: [], - }, - { - id: "rust-agent-integration-direct-module-usage", - classification: "dead-candidate", - description: "应用层重新 direct 依赖 crate::agent::integration 模块路径", - patterns: ["crate::agent::integration::"], - includePathPrefixes: ["src-tauri/src"], - allowedPaths: [], - }, - { - id: "rust-agent-subagent-direct-module-usage", - classification: "dead-candidate", - description: - "应用层重新 direct 依赖 crate::agent::subagent_scheduler 模块路径", - patterns: ["crate::agent::subagent_scheduler::"], - includePathPrefixes: ["src-tauri/src"], - allowedPaths: [], - }, - { - id: "rust-aster-runtime-snapshot-helper-leak", - classification: "dead-candidate", - description: - "已零引用的 Lime 业务层 direct Aster runtime snapshot helper 回流", - patterns: ["load_session_runtime_snapshot("], - includePathPrefixes: ["src-tauri/src", "src-tauri/crates"], - allowedPaths: [], - }, - { - id: "rust-aster-runtime-store-leak", - classification: "dead-candidate", - description: - "已零引用的 Lime 业务层 direct Aster shared runtime store 回流", - patterns: [ - "shared_thread_runtime_store(", - "initialize_shared_thread_runtime_store(", - "initialize_shared_sqlite_thread_runtime_store(", - ], - includePathPrefixes: ["src-tauri/src", "src-tauri/crates"], - allowedPaths: [], - }, - { - id: "rust-aster-runtime-store-public-require-api-leak", - classification: "dead-candidate", - description: - "Aster runtime support 重新对 crate 外暴露 require_aster_thread_runtime_store", - patterns: ["pub fn require_aster_thread_runtime_store("], - includePathPrefixes: [ - "src-tauri/crates/agent/src/aster_runtime_support.rs", - ], - allowedPaths: [], - }, - { - id: "rust-aster-runtime-init-return-store-api-leak", - classification: "dead-candidate", - description: - "Aster runtime 启动初始化 API 重新向 crate 外返回 runtime store", - patterns: [ - "pub fn initialize_aster_thread_runtime_store() -> Result, String>", - ], - includePathPrefixes: [ - "src-tauri/crates/agent/src/aster_runtime_support.rs", - ], - allowedPaths: [], - }, - { - id: "rust-aster-runtime-public-legacy-init-helper-leak", - classification: "dead-candidate", - description: - "Aster runtime support 重新对 crate 外暴露旧 initialize_aster_thread_runtime_store helper", - patterns: ["pub fn initialize_aster_thread_runtime_store("], - includePathPrefixes: [ - "src-tauri/crates/agent/src/aster_runtime_support.rs", - ], - allowedPaths: [], - }, - { - id: "rust-aster-runtime-snapshot-root-export-leak", - classification: "dead-candidate", - description: - "lime-agent crate 根重新暴露 load_aster_runtime_snapshot helper", - patterns: ["load_aster_runtime_snapshot"], - includePathPrefixes: ["src-tauri/crates/agent/src/lib.rs"], - allowedPaths: [], - }, - { - id: "rust-aster-runtime-queue-service-leak", - classification: "dead-candidate", - description: - "已零引用的 Lime 业务层 direct Aster shared runtime queue service 回流", - patterns: [], - regexPatterns: [ - String.raw`(?]+>)?\s*\(\s*["'`]([^"'`]+)["'`]/g, - /\binvoke(?:<[^>]+>)?\s*\(\s*["'`]([^"'`]+)["'`]/g, - ]; - - for (const pattern of patterns) { - for (const match of sourceCode.matchAll(pattern)) { - commands.add(match[1]); - } - } - - return commands; -} - -function stripRustTestModules(sourceCode) { - return sourceCode.replace( - /(?:^|\n)\s*#\s*\[\s*cfg\s*\(\s*test\s*\)\s*\]\s*(?:pub\s+)?mod\s+\w+\s*(?:\{[\s\S]*$|;)/m, - "\n", - ); -} - -function collectSources() { - const runtimeSources = []; - const testSources = []; - - for (const root of sourceRoots) { - const absoluteRoot = path.join(repoRoot, root); - if (!fs.existsSync(absoluteRoot)) { - continue; - } - - for (const filePath of walkDirectory(absoluteRoot, sourceExtensions)) { - const relativePath = normalizePath(path.relative(repoRoot, filePath)); - const sourceCode = fs.readFileSync(filePath, "utf8"); - const imports = extractImportSpecifiers(sourceCode); - const collectedSource = { - relativePath, - imports, - resolvedImports: new Set( - [...imports] - .map((specifier) => resolveImportPath(relativePath, specifier)) - .filter(Boolean), - ), - commands: extractInvokeCommands(sourceCode), - }; - - if (isTestFile(relativePath)) { - testSources.push(collectedSource); - continue; - } - - runtimeSources.push(collectedSource); - } - } - - return { - runtimeSources, - testSources, - }; -} - -function collectTextSources(roots, extensions) { - const runtimeSources = []; - const testSources = []; - - for (const root of roots) { - const absoluteRoot = path.join(repoRoot, root); - if (!fs.existsSync(absoluteRoot)) { - continue; - } - - for (const filePath of walkDirectory(absoluteRoot, extensions)) { - const relativePath = normalizePath(path.relative(repoRoot, filePath)); - const sourceCode = fs.readFileSync(filePath, "utf8"); - const collectedSource = { - relativePath, - sourceCode: - path.extname(relativePath) === ".rs" - ? stripRustTestModules(sourceCode) - : sourceCode, - rawSourceCode: sourceCode, - }; - - if (isTestFile(relativePath)) { - testSources.push(collectedSource); - continue; - } - - runtimeSources.push(collectedSource); - } - } - - return { - runtimeSources, - testSources, - }; -} - -function formatPaths(paths) { - if (paths.length === 0) { - return "无"; - } - - return paths.map((item) => ` - ${item}`).join("\n"); -} - -function evaluateImportMonitor(monitor, runtimeSources, testSources) { - const existingTargets = monitor.targets.filter((target) => - fs.existsSync(path.join(repoRoot, target)), - ); - const missingTargets = monitor.targets.filter( - (target) => !fs.existsSync(path.join(repoRoot, target)), - ); - const references = runtimeSources - .filter((file) => - [...file.resolvedImports].some((resolvedPath) => - monitor.targets.includes(resolvedPath), - ), - ) - .map((file) => file.relativePath) - .sort(); - const testReferences = testSources - .filter((file) => - [...file.resolvedImports].some((resolvedPath) => - monitor.targets.includes(resolvedPath), - ), - ) - .map((file) => file.relativePath) - .sort(); - - const violations = references.filter( - (relativePath) => !monitor.allowedPaths.includes(relativePath), - ); - - return { - ...monitor, - existingTargets, - missingTargets, - references, - testReferences, - violations, - }; -} - -function evaluateCommandMonitor(monitor, runtimeSources, testSources) { - const referencesByCommand = new Map(); - const testReferencesByCommand = new Map(); - - for (const command of monitor.commands) { - referencesByCommand.set( - command, - runtimeSources - .filter((file) => file.commands.has(command)) - .map((file) => file.relativePath) - .sort(), - ); - testReferencesByCommand.set( - command, - testSources - .filter((file) => file.commands.has(command)) - .map((file) => file.relativePath) - .sort(), - ); - } - - const violations = []; - for (const [command, references] of referencesByCommand.entries()) { - for (const relativePath of references) { - if (!monitor.allowedPaths.includes(relativePath)) { - violations.push(`${command} -> ${relativePath}`); - } - } - } - - return { - ...monitor, - referencesByCommand, - testReferencesByCommand, - violations, - }; -} - -function evaluateTextMonitor(monitor, runtimeSources, testSources) { - const filteredRuntimeSources = monitor.includePathPrefixes - ? runtimeSources.filter((file) => - monitor.includePathPrefixes.some((prefix) => - file.relativePath.startsWith(prefix), - ), - ) - : runtimeSources; - const filteredTestSources = monitor.includePathPrefixes - ? testSources.filter((file) => - monitor.includePathPrefixes.some((prefix) => - file.relativePath.startsWith(prefix), - ), - ) - : testSources; - const matchesPattern = (sourceCode) => - monitor.patterns.some((pattern) => sourceCode.includes(pattern)) || - (monitor.regexPatterns ?? []).some((pattern) => - new RegExp(pattern, "m").test(sourceCode), - ); - - const references = filteredRuntimeSources - .filter((file) => matchesPattern(file.sourceCode)) - .map((file) => file.relativePath) - .sort(); - const testReferences = filteredTestSources - .filter((file) => matchesPattern(file.rawSourceCode ?? file.sourceCode)) - .map((file) => file.relativePath) - .sort(); - const violations = references.filter( - (relativePath) => !monitor.allowedPaths.includes(relativePath), - ); - - return { - ...monitor, - references, - testReferences, - violations, - }; -} - -function countOccurrences(sourceCode, pattern) { - if (!pattern) { - return 0; - } - - let count = 0; - let startIndex = 0; - - while (true) { - const matchIndex = sourceCode.indexOf(pattern, startIndex); - if (matchIndex === -1) { - return count; - } - count += 1; - startIndex = matchIndex + pattern.length; - } -} - -function evaluateTextCountMonitor(monitor, runtimeSources, testSources) { - const filteredRuntimeSources = monitor.includePathPrefixes - ? runtimeSources.filter((file) => - monitor.includePathPrefixes.some((prefix) => - file.relativePath.startsWith(prefix), - ), - ) - : runtimeSources; - const filteredTestSources = monitor.includePathPrefixes - ? testSources.filter((file) => - monitor.includePathPrefixes.some((prefix) => - file.relativePath.startsWith(prefix), - ), - ) - : testSources; - const runtimeMatches = []; - const testMatches = []; - const violations = []; - - for (const file of filteredRuntimeSources) { - const counts = monitor.occurrences - .map((rule) => ({ - ...rule, - count: countOccurrences(file.sourceCode, rule.pattern), - })) - .filter((rule) => rule.count > 0); - - if (counts.length === 0) { - continue; - } - - runtimeMatches.push({ - relativePath: file.relativePath, - counts, - }); - - for (const rule of counts) { - if (rule.count > rule.maxCount) { - violations.push( - `${file.relativePath} -> ${rule.pattern} (${rule.count} > ${rule.maxCount})`, - ); - } - } - } - - for (const file of filteredTestSources) { - const counts = monitor.occurrences - .map((rule) => ({ - ...rule, - count: countOccurrences( - file.rawSourceCode ?? file.sourceCode, - rule.pattern, - ), - })) - .filter((rule) => rule.count > 0); - - if (counts.length === 0) { - continue; - } - - testMatches.push({ - relativePath: file.relativePath, - counts, - }); - } - - return { - ...monitor, - runtimeMatches, - testMatches, - violations, - }; -} - -function getImportStatus(result) { - return result.violations.length > 0 - ? "违规" - : result.references.length === 0 && result.existingTargets.length === 0 - ? "已删除" - : result.references.length === 0 - ? "零引用" - : "受控"; -} - -function getCommandStatus(result) { - const flattenedReferences = [...result.referencesByCommand.values()].flat(); - const uniqueReferences = [...new Set(flattenedReferences)].sort(); - return result.violations.length > 0 - ? "违规" - : uniqueReferences.length === 0 - ? "零引用" - : "受控"; -} - -function getTextStatus(result) { - return result.violations.length > 0 - ? "违规" - : result.references.length === 0 - ? "零引用" - : "受控"; -} - -function isStatusClassificationDrift(status, classification) { - return ( - (status === "已删除" || status === "零引用") && - classification !== "dead-candidate" - ); -} - -function printImportReport(result) { - const status = getImportStatus(result); - - console.log( - `- [${status}] ${result.id} (${result.classification}):${result.description}`, - ); - console.log(` 目标文件:${result.targets.join(", ")}`); - console.log(` 允许引用:${result.allowedPaths.join(", ") || "无"}`); - if (result.missingTargets.length > 0) { - console.log(` 已删除目标:\n${formatPaths(result.missingTargets)}`); - } - console.log(` 实际引用:\n${formatPaths(result.references)}`); - console.log(` 测试引用:\n${formatPaths(result.testReferences)}`); - - if (result.violations.length > 0) { - console.log(` 违规引用:\n${formatPaths(result.violations)}`); - } -} - -function printCommandReport(result) { - const status = getCommandStatus(result); - - console.log( - `- [${status}] ${result.id} (${result.classification}):${result.description}`, - ); - console.log(` 命令:${result.commands.join(", ")}`); - console.log(` 允许引用:${result.allowedPaths.join(", ") || "无"}`); - - for (const command of result.commands) { - const references = result.referencesByCommand.get(command) ?? []; - const testReferences = result.testReferencesByCommand.get(command) ?? []; - console.log(` ${command}:\n${formatPaths(references)}`); - console.log(` ${command}(测试):\n${formatPaths(testReferences)}`); - } - - if (result.violations.length > 0) { - console.log(` 违规引用:\n${formatPaths(result.violations)}`); - } -} - -function printTextReport(result) { - const status = getTextStatus(result); - - console.log( - `- [${status}] ${result.id} (${result.classification}):${result.description}`, - ); - const keywords = [ - ...result.patterns, - ...(result.regexPatterns ?? []).map((pattern) => `regex:${pattern}`), - ]; - console.log(` 关键字:${keywords.join(", ")}`); - console.log(` 允许引用:${result.allowedPaths.join(", ") || "无"}`); - console.log(` 实际引用:\n${formatPaths(result.references)}`); - console.log(` 测试引用:\n${formatPaths(result.testReferences)}`); - - if (result.violations.length > 0) { - console.log(` 违规引用:\n${formatPaths(result.violations)}`); - } -} - -function printTextCountReport(result) { - const status = getTextCountStatus(result); - - console.log( - `- [${status}] ${result.id} (${result.classification}):${result.description}`, - ); - console.log( - ` 次数规则:${result.occurrences - .map((rule) => `${rule.pattern} <= ${rule.maxCount}`) - .join(";")}`, - ); - console.log( - ` 实际命中:\n${formatPaths( - result.runtimeMatches.map( - (item) => - `${item.relativePath} -> ${item.counts - .map((rule) => `${rule.pattern} (${rule.count})`) - .join(";")}`, - ), - )}`, - ); - console.log( - ` 测试命中:\n${formatPaths( - result.testMatches.map( - (item) => - `${item.relativePath} -> ${item.counts - .map((rule) => `${rule.pattern} (${rule.count})`) - .join(";")}`, - ), - )}`, - ); - - if (result.violations.length > 0) { - console.log(` 违规引用:\n${formatPaths(result.violations)}`); - } -} - -function getTextCountStatus(result) { - return result.violations.length > 0 - ? "违规" - : result.runtimeMatches.length === 0 - ? "零引用" - : "受控"; -} - -export function buildLegacySurfaceReport() { - const { runtimeSources, testSources } = collectSources(); - const { - runtimeSources: frontendRuntimeTextSources, - testSources: frontendTestTextSources, - } = collectTextSources(sourceRoots, sourceExtensions); - const { runtimeSources: rustRuntimeSources, testSources: rustTestSources } = - collectTextSources(rustSourceRoots, rustSourceExtensions); - const importResults = importSurfaceMonitors.map((monitor) => - evaluateImportMonitor(monitor, runtimeSources, testSources), - ); - const commandResults = commandSurfaceMonitors.map((monitor) => - evaluateCommandMonitor(monitor, runtimeSources, testSources), - ); - const frontendTextResults = frontendTextSurfaceMonitors.map((monitor) => - evaluateTextMonitor( - monitor, - frontendRuntimeTextSources, - frontendTestTextSources, - ), - ); - const rustTextResults = rustTextSurfaceMonitors.map((monitor) => - evaluateTextMonitor(monitor, rustRuntimeSources, rustTestSources), - ); - const rustTextCountResults = rustTextCountMonitors.map((monitor) => - evaluateTextCountMonitor(monitor, rustRuntimeSources, rustTestSources), - ); - - const zeroReferenceCandidates = importResults - .filter( - (result) => - result.references.length === 0 && result.existingTargets.length > 0, - ) - .map((result) => `${result.id} (${result.description})`); - const classificationDriftCandidates = [ - ...importResults - .filter((result) => - isStatusClassificationDrift( - getImportStatus(result), - result.classification, - ), - ) - .map( - (result) => - `${result.id} -> ${result.classification} / ${getImportStatus(result)}`, - ), - ...commandResults - .filter((result) => - isStatusClassificationDrift( - getCommandStatus(result), - result.classification, - ), - ) - .map( - (result) => - `${result.id} -> ${result.classification} / ${getCommandStatus(result)}`, - ), - ...frontendTextResults - .filter((result) => - isStatusClassificationDrift(getTextStatus(result), result.classification), - ) - .map( - (result) => - `${result.id} -> ${result.classification} / ${getTextStatus(result)}`, - ), - ...rustTextResults - .filter((result) => - isStatusClassificationDrift(getTextStatus(result), result.classification), - ) - .map( - (result) => - `${result.id} -> ${result.classification} / ${getTextStatus(result)}`, - ), - ...rustTextCountResults - .filter((result) => - isStatusClassificationDrift( - getTextCountStatus(result), - result.classification, - ), - ) - .map( - (result) => - `${result.id} -> ${result.classification} / ${getTextCountStatus(result)}`, - ), - ]; - const violations = [ - ...importResults.flatMap((result) => - result.violations.map((item) => `${result.id} -> ${item}`), - ), - ...commandResults.flatMap((result) => - result.violations.map((item) => `${result.id} -> ${item}`), - ), - ...frontendTextResults.flatMap((result) => - result.violations.map((item) => `${result.id} -> ${item}`), - ), - ...rustTextResults.flatMap((result) => - result.violations.map((item) => `${result.id} -> ${item}`), - ), - ...rustTextCountResults.flatMap((result) => - result.violations.map((item) => `${result.id} -> ${item}`), - ), - ]; - - return { - repoRoot, - runtimeSources, - testSources, - rustRuntimeSources, - rustTestSources, - importResults, - commandResults, - frontendTextResults, - rustTextResults, - rustTextCountResults, - zeroReferenceCandidates, - classificationDriftCandidates, - violations, - }; -} - -function serializeMapEntries(map) { - return Object.fromEntries(map.entries()); -} - -export function toSerializableLegacySurfaceReport(report) { - return { - repoRoot: report.repoRoot, - summary: { - runtimeSourceCount: report.runtimeSources.length, - testSourceCount: report.testSources.length, - rustRuntimeSourceCount: report.rustRuntimeSources.length, - rustTestSourceCount: report.rustTestSources.length, - zeroReferenceCandidates: report.zeroReferenceCandidates, - classificationDriftCandidates: report.classificationDriftCandidates, - violations: report.violations, - }, - importResults: report.importResults, - commandResults: report.commandResults.map((result) => ({ - ...result, - referencesByCommand: serializeMapEntries(result.referencesByCommand), - testReferencesByCommand: serializeMapEntries(result.testReferencesByCommand), - })), - frontendTextResults: report.frontendTextResults, - rustTextResults: report.rustTextResults, - rustTextCountResults: report.rustTextCountResults, - }; -} - -export function printLegacySurfaceReport(report) { - console.log("[lime] legacy surface report"); - console.log(""); - console.log("## 入口引用"); - for (const result of report.importResults) { - printImportReport(result); - } - - console.log(""); - console.log("## 命令边界"); - for (const result of report.commandResults) { - printCommandReport(result); - } - - console.log(""); - console.log("## 前端护栏"); - for (const result of report.frontendTextResults) { - printTextReport(result); - } - - console.log(""); - console.log("## Rust 护栏"); - for (const result of report.rustTextResults) { - printTextReport(result); - } - for (const result of report.rustTextCountResults) { - printTextCountReport(result); - } - - console.log(""); - console.log("## 摘要"); - console.log(`- 扫描文件数:${report.runtimeSources.length}`); - console.log(`- 测试文件数:${report.testSources.length}`); - console.log(`- Rust 扫描文件数:${report.rustRuntimeSources.length}`); - console.log(`- Rust 测试文件数:${report.rustTestSources.length}`); - console.log(`- 零引用候选:${report.zeroReferenceCandidates.length}`); - for (const candidate of report.zeroReferenceCandidates) { - console.log(` - ${candidate}`); - } - console.log(`- 分类漂移候选:${report.classificationDriftCandidates.length}`); - for (const candidate of report.classificationDriftCandidates) { - console.log(` - ${candidate}`); - } - console.log(`- 边界违规:${report.violations.length}`); - for (const violation of report.violations) { - console.log(` - ${violation}`); - } -} +import { + buildLegacySurfaceReport, + printLegacySurfaceReport, + toSerializableLegacySurfaceReport, +} from "./lib/legacy-surface-report-core.mjs"; + +export { + buildLegacySurfaceReport, + printLegacySurfaceReport, + toSerializableLegacySurfaceReport, +} from "./lib/legacy-surface-report-core.mjs"; function parseCliArgs(argv) { const options = { diff --git a/src-tauri/src/app/runner.rs b/src-tauri/src/app/runner.rs index 0ff2c1a43..58d869aee 100644 --- a/src-tauri/src/app/runner.rs +++ b/src-tauri/src/app/runner.rs @@ -1454,6 +1454,7 @@ pub fn run() { commands::aster_agent_cmd::command_api::runtime_api::agent_runtime_export_handoff_bundle, commands::aster_agent_cmd::command_api::runtime_api::agent_runtime_export_evidence_pack, commands::aster_agent_cmd::command_api::runtime_api::agent_runtime_export_review_decision_template, + commands::aster_agent_cmd::command_api::runtime_api::agent_runtime_save_review_decision, commands::aster_agent_cmd::command_api::runtime_api::agent_runtime_export_replay_case, commands::aster_agent_cmd::command_api::runtime_api::agent_runtime_get_tool_inventory, commands::aster_agent_cmd::command_api::subagent_api::agent_runtime_spawn_subagent, diff --git a/src-tauri/src/commands/aster_agent_cmd/command_api.rs b/src-tauri/src/commands/aster_agent_cmd/command_api.rs index 7a80e2e96..19aaf9a4c 100644 --- a/src-tauri/src/commands/aster_agent_cmd/command_api.rs +++ b/src-tauri/src/commands/aster_agent_cmd/command_api.rs @@ -52,7 +52,8 @@ pub(crate) use runtime_api::{ agent_runtime_export_replay_case, agent_runtime_get_session, agent_runtime_get_thread_read, agent_runtime_get_tool_inventory, agent_runtime_interrupt_turn, agent_runtime_promote_queued_turn, agent_runtime_remove_queued_turn, - agent_runtime_replay_request, agent_runtime_resume_thread, agent_runtime_submit_turn, + agent_runtime_replay_request, agent_runtime_resume_thread, agent_runtime_save_review_decision, + agent_runtime_submit_turn, }; pub(crate) use session_api::{ agent_runtime_create_session, agent_runtime_list_sessions, agent_runtime_update_session, diff --git a/src-tauri/src/commands/aster_agent_cmd/command_api/runtime_api.rs b/src-tauri/src/commands/aster_agent_cmd/command_api/runtime_api.rs index 8a0eab255..e9ba8b93b 100644 --- a/src-tauri/src/commands/aster_agent_cmd/command_api/runtime_api.rs +++ b/src-tauri/src/commands/aster_agent_cmd/command_api/runtime_api.rs @@ -12,7 +12,8 @@ use crate::services::runtime_replay_case_service::{ export_runtime_replay_case, RuntimeReplayCaseExportResult, }; use crate::services::runtime_review_decision_service::{ - export_runtime_review_decision_template, RuntimeReviewDecisionTemplateExportResult, + export_runtime_review_decision_template, save_runtime_review_decision, + RuntimeReviewDecisionContent, RuntimeReviewDecisionTemplateExportResult, }; use crate::services::thread_reliability_projection_service::sync_thread_reliability_projection; use std::path::PathBuf; @@ -454,6 +455,54 @@ pub async fn agent_runtime_export_review_decision_template( ) } +/// 统一运行时:保存当前会话的人工审核结果。 +#[tauri::command] +pub async fn agent_runtime_save_review_decision( + app: AppHandle, + state: State<'_, AsterAgentState>, + db: State<'_, DbConnection>, + api_key_provider_service: State<'_, ApiKeyProviderServiceState>, + logs: State<'_, LogState>, + config_manager: State<'_, GlobalConfigManagerState>, + mcp_manager: State<'_, McpManagerState>, + automation_state: State<'_, AutomationServiceState>, + request: AgentRuntimeSaveReviewDecisionRequest, +) -> Result { + let session_id = request.session_id.trim().to_string(); + tracing::info!("[AsterAgent] 保存 review decision: {}", session_id); + let context = load_runtime_export_context( + &app, + state.inner(), + db.inner(), + api_key_provider_service.inner(), + logs.inner(), + config_manager.inner(), + mcp_manager.inner(), + automation_state.inner(), + &session_id, + "保存 review decision", + ) + .await?; + + save_runtime_review_decision( + &context.detail, + &context.thread_read, + &context.workspace_root, + RuntimeReviewDecisionContent { + decision_status: request.decision_status, + decision_summary: request.decision_summary, + chosen_fix_strategy: request.chosen_fix_strategy, + risk_level: request.risk_level, + risk_tags: request.risk_tags, + human_reviewer: request.human_reviewer, + reviewed_at: request.reviewed_at, + followup_actions: request.followup_actions, + regression_requirements: request.regression_requirements, + notes: request.notes, + }, + ) +} + /// 统一运行时:导出当前会话的 replay case。 #[tauri::command] pub async fn agent_runtime_export_replay_case( diff --git a/src-tauri/src/commands/aster_agent_cmd/dto.rs b/src-tauri/src/commands/aster_agent_cmd/dto.rs index 71b4121ac..7eaf82d8d 100644 --- a/src-tauri/src/commands/aster_agent_cmd/dto.rs +++ b/src-tauri/src/commands/aster_agent_cmd/dto.rs @@ -1601,6 +1601,32 @@ pub struct AgentRuntimeUpdateSessionRequest { pub recent_team_selection: Option, } +#[derive(Debug, Deserialize)] +pub struct AgentRuntimeSaveReviewDecisionRequest { + #[serde(alias = "sessionId")] + pub session_id: String, + #[serde(default, alias = "decisionStatus")] + pub decision_status: String, + #[serde(default, alias = "decisionSummary")] + pub decision_summary: String, + #[serde(default, alias = "chosenFixStrategy")] + pub chosen_fix_strategy: String, + #[serde(default, alias = "riskLevel")] + pub risk_level: String, + #[serde(default, alias = "riskTags")] + pub risk_tags: Vec, + #[serde(default, alias = "humanReviewer")] + pub human_reviewer: String, + #[serde(default, alias = "reviewedAt")] + pub reviewed_at: Option, + #[serde(default, alias = "followupActions")] + pub followup_actions: Vec, + #[serde(default, alias = "regressionRequirements")] + pub regression_requirements: Vec, + #[serde(default)] + pub notes: String, +} + /// 自动续写参数 #[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)] pub struct AutoContinuePayload { diff --git a/src-tauri/src/commands/aster_agent_cmd/mod.rs b/src-tauri/src/commands/aster_agent_cmd/mod.rs index bc45f0308..5ff0c0737 100644 --- a/src-tauri/src/commands/aster_agent_cmd/mod.rs +++ b/src-tauri/src/commands/aster_agent_cmd/mod.rs @@ -306,10 +306,10 @@ pub(crate) use command_api::{ agent_runtime_get_session, agent_runtime_get_thread_read, agent_runtime_get_tool_inventory, agent_runtime_interrupt_turn, agent_runtime_list_sessions, agent_runtime_promote_queued_turn, agent_runtime_remove_queued_turn, agent_runtime_replay_request, agent_runtime_resume_subagent, - agent_runtime_resume_thread, agent_runtime_send_subagent_input, agent_runtime_spawn_subagent, - agent_runtime_submit_turn, agent_runtime_update_session, agent_runtime_wait_subagents, - aster_agent_configure_from_pool, aster_agent_configure_provider, aster_agent_init, - aster_agent_reset, aster_agent_status, + agent_runtime_resume_thread, agent_runtime_save_review_decision, + agent_runtime_send_subagent_input, agent_runtime_spawn_subagent, agent_runtime_submit_turn, + agent_runtime_update_session, agent_runtime_wait_subagents, aster_agent_configure_from_pool, + aster_agent_configure_provider, aster_agent_init, aster_agent_reset, aster_agent_status, }; #[allow(unused_imports)] pub(crate) use dto::{ @@ -322,8 +322,9 @@ pub(crate) use dto::{ AgentRuntimeReplayRequestRequest, AgentRuntimeReplayedActionRequiredView, AgentRuntimeRequestView, AgentRuntimeRespondActionRequest, AgentRuntimeResumeSubagentRequest, AgentRuntimeResumeSubagentResponse, AgentRuntimeResumeThreadRequest, - AgentRuntimeSendSubagentInputRequest, AgentRuntimeSendSubagentInputResponse, - AgentRuntimeSessionDetail, AgentRuntimeSpawnSubagentRequest, AgentRuntimeSpawnSubagentResponse, + AgentRuntimeSaveReviewDecisionRequest, AgentRuntimeSendSubagentInputRequest, + AgentRuntimeSendSubagentInputResponse, AgentRuntimeSessionDetail, + AgentRuntimeSpawnSubagentRequest, AgentRuntimeSpawnSubagentResponse, AgentRuntimeSubmitTurnRequest, AgentRuntimeThreadDiagnostics, AgentRuntimeThreadReadModel, AgentRuntimeToolInventoryRequest, AgentRuntimeUpdateSessionRequest, AgentRuntimeWaitSubagentsRequest, AgentRuntimeWaitSubagentsResponse, AsterAgentStatus, diff --git a/src-tauri/src/commands/aster_agent_cmd/runtime_turn.rs b/src-tauri/src/commands/aster_agent_cmd/runtime_turn.rs index 12e856618..1cade54d8 100644 --- a/src-tauri/src/commands/aster_agent_cmd/runtime_turn.rs +++ b/src-tauri/src/commands/aster_agent_cmd/runtime_turn.rs @@ -31,19 +31,31 @@ fn emit_runtime_side_event( } } -fn summarize_artifact_document_issues(issues: &[String]) -> String { - let parts = issues - .iter() - .map(|issue| issue.trim()) - .filter(|issue| !issue.is_empty()) - .take(3) - .collect::>(); - - if parts.is_empty() { - "结构已按最小可用方案落盘。".to_string() - } else { - parts.join(";") +fn build_artifact_document_warning_message( + status: &str, + fallback_used: bool, + issues: &[String], +) -> String { + if status == "failed" { + return "结构化文稿未完整生成,已保留一份可继续编辑的恢复稿。".to_string(); } + + if fallback_used + || issues + .iter() + .any(|issue| issue.contains("Markdown 正文自动恢复")) + { + return "已根据正文整理出一份可继续编辑的草稿。".to_string(); + } + + if issues + .iter() + .any(|issue| issue.contains("不完整的 ArtifactDocument JSON")) + { + return "已补全文稿结构,可继续查看和编辑。".to_string(); + } + + "已整理为可继续编辑的文稿。".to_string() } fn merge_turn_context_with_artifact_output_schema( @@ -213,15 +225,19 @@ fn maybe_persist_artifact_document_after_stream( let (code, prefix) = if persisted.status == "failed" { ( ARTIFACT_DOCUMENT_FAILED_WARNING_CODE, - "ArtifactDocument 未通过完整校验,已以失败态文档落盘", + "ArtifactDocument 已落盘", ) } else { ( ARTIFACT_DOCUMENT_REPAIRED_WARNING_CODE, - "ArtifactDocument 已自动修复后落盘", + "ArtifactDocument 已落盘", ) }; - let detail = summarize_artifact_document_issues(&persisted.issues); + let detail = build_artifact_document_warning_message( + persisted.status.as_str(), + persisted.fallback_used, + &persisted.issues, + ); emit_runtime_side_event( app, event_name, diff --git a/src-tauri/src/services/artifact_document_validator.rs b/src-tauri/src/services/artifact_document_validator.rs index 9d6f88581..971a23287 100644 --- a/src-tauri/src/services/artifact_document_validator.rs +++ b/src-tauri/src/services/artifact_document_validator.rs @@ -34,6 +34,10 @@ const ARTIFACT_BLOCK_TYPE_VALUES: &[&str] = &[ "divider", ]; const MAX_BLOCK_COUNT: usize = 40; +const MARKDOWN_RECOVERY_REASON: &str = + "模型未返回合法的 ArtifactDocument JSON,已按 Markdown 正文自动恢复为可渲染文档。"; +const TRUNCATED_JSON_RECOVERY_REASON: &str = + "检测到不完整的 ArtifactDocument JSON,已做闭合修复后继续校验。"; #[derive(Debug, Clone, Default, PartialEq, Eq)] pub struct ArtifactDocumentValidationContext { @@ -64,7 +68,18 @@ pub fn validate_or_fallback_artifact_document( raw_text: &str, context: &ArtifactDocumentValidationContext, ) -> ArtifactDocumentValidationOutcome { - let Some(candidate) = extract_artifact_document_candidate(raw_text) else { + let Some((candidate, repaired_truncated_json)) = extract_artifact_document_candidate(raw_text) + else { + if let Some(recovered_candidate) = build_markdown_recovery_candidate(raw_text, context) { + let mut outcome = + validate_or_repair_artifact_document_value(&recovered_candidate, raw_text, context); + outcome.repaired = true; + outcome.fallback_used = true; + outcome + .issues + .insert(0, MARKDOWN_RECOVERY_REASON.to_string()); + return outcome; + } return build_failed_fallback_document( raw_text, "模型未返回合法的 ArtifactDocument JSON,已回退为失败态文档。", @@ -72,7 +87,14 @@ pub fn validate_or_fallback_artifact_document( ); }; - validate_or_repair_artifact_document_value(&candidate, raw_text, context) + let mut outcome = validate_or_repair_artifact_document_value(&candidate, raw_text, context); + if repaired_truncated_json { + outcome.repaired = true; + outcome + .issues + .insert(0, TRUNCATED_JSON_RECOVERY_REASON.to_string()); + } + outcome } pub fn validate_or_repair_artifact_document_value( @@ -303,7 +325,53 @@ fn build_failed_fallback_document( } } -fn extract_artifact_document_candidate(raw_text: &str) -> Option { +fn build_markdown_recovery_candidate( + raw_text: &str, + context: &ArtifactDocumentValidationContext, +) -> Option { + let trimmed = raw_text.trim(); + if trimmed.is_empty() { + return None; + } + + let (heading_title, markdown_body) = extract_markdown_heading_and_body(trimmed); + let title = normalize_title( + heading_title + .or_else(|| context.title_hint.clone()) + .or_else(|| extract_first_non_empty_line(trimmed)), + ); + let body = if markdown_body.trim().is_empty() { + trimmed.to_string() + } else { + markdown_body + }; + let kind = normalize_enum(context.kind_hint.clone(), ARTIFACT_KIND_VALUES, "analysis"); + let mut document = Map::new(); + document.insert( + "schemaVersion".to_string(), + Value::String(ARTIFACT_DOCUMENT_SCHEMA_VERSION.to_string()), + ); + document.insert("kind".to_string(), Value::String(kind)); + document.insert("title".to_string(), Value::String(title)); + document.insert("status".to_string(), Value::String("draft".to_string())); + document.insert("language".to_string(), Value::String("zh-CN".to_string())); + if let Some(summary) = extract_markdown_summary(body.as_str()) { + document.insert("summary".to_string(), Value::String(summary)); + } + document.insert( + "blocks".to_string(), + Value::Array(vec![Value::Object(build_fallback_rich_text_block( + "block-1", + body.as_str(), + Some("markdown_recovery"), + ))]), + ); + document.insert("sources".to_string(), Value::Array(Vec::new())); + document.insert("metadata".to_string(), Value::Object(Map::new())); + Some(Value::Object(document)) +} + +fn extract_artifact_document_candidate(raw_text: &str) -> Option<(Value, bool)> { let trimmed = raw_text.trim(); if trimmed.is_empty() { return None; @@ -314,6 +382,7 @@ fn extract_artifact_document_candidate(raw_text: &str) -> Option { strip_outer_code_fence(trimmed), extract_first_fenced_payload(trimmed).unwrap_or_default(), extract_braced_json_candidate(trimmed).unwrap_or_default(), + extract_unclosed_json_candidate(trimmed).unwrap_or_default(), ]; for candidate in candidates { @@ -321,17 +390,24 @@ fn extract_artifact_document_candidate(raw_text: &str) -> Option { if normalized.is_empty() { continue; } - let Ok(parsed) = serde_json::from_str::(normalized) else { - continue; - }; - if let Some(document) = unwrap_artifact_document_envelope(&parsed) { - return Some(document.clone()); + if let Some(document) = parse_artifact_document_candidate(normalized) { + return Some((document, false)); + } + if let Some(repaired) = repair_json_candidate(normalized) { + if let Some(document) = parse_artifact_document_candidate(repaired.as_str()) { + return Some((document, true)); + } } } None } +fn parse_artifact_document_candidate(candidate: &str) -> Option { + let parsed = serde_json::from_str::(candidate).ok()?; + unwrap_artifact_document_envelope(&parsed).cloned() +} + fn unwrap_artifact_document_envelope(value: &Value) -> Option<&Value> { let record = value.as_object()?; if is_artifact_document_record(record) { @@ -396,6 +472,79 @@ fn extract_braced_json_candidate(raw: &str) -> Option { Some(raw[start..=end].trim().to_string()) } +fn extract_unclosed_json_candidate(raw: &str) -> Option { + let start = raw.find('{').or_else(|| raw.find('['))?; + Some(raw[start..].trim().to_string()) +} + +fn repair_json_candidate(raw: &str) -> Option { + let trimmed = raw.trim(); + if trimmed.is_empty() { + return None; + } + + let mut repaired = String::new(); + let mut expected_closers = Vec::new(); + let mut in_string = false; + let mut escaping = false; + + for ch in trimmed.chars() { + repaired.push(ch); + if in_string { + if escaping { + escaping = false; + continue; + } + match ch { + '\\' => escaping = true, + '"' => in_string = false, + _ => {} + } + continue; + } + + match ch { + '"' => in_string = true, + '{' => expected_closers.push('}'), + '[' => expected_closers.push(']'), + '}' | ']' => { + let expected = expected_closers.pop()?; + if ch != expected { + return None; + } + } + _ => {} + } + } + + if in_string { + repaired.push('"'); + } + repaired = repair_json_tail(repaired); + while let Some(closer) = expected_closers.pop() { + repaired = repair_json_tail(repaired); + repaired.push(closer); + } + Some(repaired) +} + +fn repair_json_tail(mut value: String) -> String { + loop { + let trimmed_len = value.trim_end().len(); + value.truncate(trimmed_len); + if value.ends_with(':') { + value.push_str(" null"); + break; + } + if value.ends_with(',') { + value.pop(); + continue; + } + break; + } + value +} + fn normalize_sources( value: Option<&Value>, repaired: &mut bool, @@ -841,6 +990,102 @@ fn build_fallback_markdown(raw_text: &str, reason: &str) -> String { } } +fn extract_markdown_heading_and_body(raw_text: &str) -> (Option, String) { + let lines = raw_text.lines().collect::>(); + let first_content_index = lines + .iter() + .position(|line| !line.trim().is_empty()) + .unwrap_or(0); + let first_line = lines + .get(first_content_index) + .map(|line| line.trim()) + .unwrap_or_default(); + let Some(heading) = first_line.strip_prefix('#') else { + return (None, raw_text.trim().to_string()); + }; + let title = normalize_text(Some(heading.trim_start_matches('#').trim())); + if title.is_none() { + return (None, raw_text.trim().to_string()); + } + + let mut body_start = first_content_index + 1; + while body_start < lines.len() && lines[body_start].trim().is_empty() { + body_start += 1; + } + let body = lines[body_start..].join("\n").trim().to_string(); + (title, body) +} + +fn extract_first_non_empty_line(raw_text: &str) -> Option { + raw_text + .lines() + .map(str::trim) + .find(|line| !line.is_empty()) + .map(ToString::to_string) +} + +fn extract_markdown_summary(markdown: &str) -> Option { + let mut in_code_block = false; + let mut parts = Vec::new(); + + for line in markdown.lines() { + let trimmed = line.trim(); + if trimmed.starts_with("```") { + in_code_block = !in_code_block; + continue; + } + if in_code_block || trimmed.is_empty() { + if !parts.is_empty() { + break; + } + continue; + } + if trimmed.starts_with('#') { + continue; + } + + let normalized = strip_markdown_summary_prefix(trimmed); + if normalized.is_empty() { + continue; + } + parts.push(normalized); + if parts.len() >= 2 { + break; + } + } + + if parts.is_empty() { + None + } else { + Some(truncate_text(parts.join(" ").as_str(), 180)) + } +} + +fn strip_markdown_summary_prefix(line: &str) -> String { + let trimmed = line.trim(); + let without_bullet = trimmed + .strip_prefix("- ") + .or_else(|| trimmed.strip_prefix("* ")) + .or_else(|| trimmed.strip_prefix("> ")) + .unwrap_or(trimmed) + .trim(); + let without_ordered = without_bullet + .find(". ") + .and_then(|index| { + if without_bullet[..index] + .chars() + .all(|ch| ch.is_ascii_digit()) + { + without_bullet.get(index + 2..) + } else { + None + } + }) + .unwrap_or(without_bullet) + .trim(); + without_ordered.to_string() +} + fn normalize_title(value: Option) -> String { value .map(|title| truncate_text(&title, 120)) @@ -1132,4 +1377,53 @@ mod tests { == Some("rich_text") ); } + + #[test] + fn validate_or_fallback_should_repair_truncated_json_before_markdown_recovery() { + let mut context = base_context(); + context.source_policy = Some("none".to_string()); + + let outcome = validate_or_fallback_artifact_document( + "{\n \"schemaVersion\": \"artifact_document.v1\",\n \"kind\": \"analysis\",\n \"title\": \"结构化报告\",\n \"status\": \"ready\",\n \"blocks\": [\n { \"type\": \"hero_summary\", \"summary\": \"摘要\" }\n ],\n \"sources\": [],\n \"metadata\": {\n \"theme\": \"knowledge\"\n }\n", + &context, + ); + + assert_eq!(outcome.status, "ready"); + assert!(!outcome.fallback_used); + assert!(outcome.repaired); + assert!(outcome + .issues + .iter() + .any(|issue| issue.contains("不完整的 ArtifactDocument JSON"))); + assert_eq!(outcome.title, "结构化报告"); + } + + #[test] + fn validate_or_fallback_should_recover_markdown_when_sources_not_required() { + let mut context = base_context(); + context.source_policy = Some("none".to_string()); + context.title_hint = None; + + let outcome = validate_or_fallback_artifact_document( + "# 前端概念方案\n\n我将为你整理一份通用的前端概念方案框架。\n\n## 信息架构\n- 页面结构\n", + &context, + ); + + assert_eq!(outcome.status, "draft"); + assert!(outcome.fallback_used); + assert_eq!(outcome.title, "前端概念方案"); + assert!(outcome.repaired); + assert!(outcome + .issues + .iter() + .any(|issue| issue.contains("Markdown 正文自动恢复"))); + assert_eq!( + outcome.document.get("status").and_then(Value::as_str), + Some("draft") + ); + assert_eq!( + outcome.document.get("summary").and_then(Value::as_str), + Some("我将为你整理一份通用的前端概念方案框架。") + ); + } } diff --git a/src-tauri/src/services/runtime_review_decision_service.rs b/src-tauri/src/services/runtime_review_decision_service.rs index 71856a75c..d3e89737e 100644 --- a/src-tauri/src/services/runtime_review_decision_service.rs +++ b/src-tauri/src/services/runtime_review_decision_service.rs @@ -1,8 +1,8 @@ -//! Runtime review decision 模板导出服务 +//! Runtime review decision 模板导出与保存服务 //! //! 将外部 Claude Code / Codex 的分析结论回挂为 //! Lime 工作区内可版本化的人工审核与决策记录模板。 -//! 这条链只导出 review-decision 模板,不在 Lime 内自动批准或自动应用修复。 +//! 这条链只导出与保存 review-decision,不在 Lime 内自动批准或自动应用修复。 use crate::agent::SessionDetail; use crate::commands::aster_agent_cmd::AgentRuntimeThreadReadModel; @@ -59,12 +59,15 @@ pub struct RuntimeReviewDecisionTemplateExportResult { pub pending_request_count: usize, pub queued_turn_count: usize, pub default_decision_status: String, + pub decision: RuntimeReviewDecisionContent, + pub decision_status_options: Vec, + pub risk_level_options: Vec, pub review_checklist: Vec, pub analysis_artifacts: Vec, pub artifacts: Vec, } -#[derive(Debug, Clone, Serialize)] +#[derive(Debug, Clone, Serialize, Deserialize)] #[serde(rename_all = "camelCase")] struct ReviewDecisionDocument { schema_version: String, @@ -72,20 +75,20 @@ struct ReviewDecisionDocument { exported_at: String, source: ReviewDecisionSource, review_context: ReviewDecisionContext, - decision: ReviewDecisionContent, + decision: RuntimeReviewDecisionContent, decision_status_options: Vec, risk_level_options: Vec, review_checklist: Vec, } -#[derive(Debug, Clone, Serialize)] +#[derive(Debug, Clone, Serialize, Deserialize)] #[serde(rename_all = "camelCase")] struct ReviewDecisionSource { derived_from: Vec, upstream_alignment: ReviewDecisionUpstreamAlignment, } -#[derive(Debug, Clone, Serialize)] +#[derive(Debug, Clone, Serialize, Deserialize)] #[serde(rename_all = "camelCase")] struct ReviewDecisionUpstreamAlignment { execution_environment_reference: String, @@ -93,7 +96,7 @@ struct ReviewDecisionUpstreamAlignment { product_surface: String, } -#[derive(Debug, Clone, Serialize)] +#[derive(Debug, Clone, Serialize, Deserialize)] #[serde(rename_all = "camelCase")] struct ReviewDecisionContext { session_id: String, @@ -111,7 +114,7 @@ struct ReviewDecisionContext { analysis_artifacts: Vec, } -#[derive(Debug, Clone, Serialize)] +#[derive(Debug, Clone, Serialize, Deserialize)] #[serde(rename_all = "camelCase")] struct ReviewDecisionArtifactReference { kind: String, @@ -119,25 +122,43 @@ struct ReviewDecisionArtifactReference { relative_path: String, } -#[derive(Debug, Clone, Serialize)] +#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)] #[serde(rename_all = "camelCase")] -struct ReviewDecisionContent { - decision_status: String, - decision_summary: String, - chosen_fix_strategy: String, - risk_level: String, - risk_tags: Vec, - human_reviewer: String, - reviewed_at: Option, - followup_actions: Vec, - regression_requirements: Vec, - notes: String, +pub struct RuntimeReviewDecisionContent { + pub decision_status: String, + pub decision_summary: String, + pub chosen_fix_strategy: String, + pub risk_level: String, + pub risk_tags: Vec, + pub human_reviewer: String, + pub reviewed_at: Option, + pub followup_actions: Vec, + pub regression_requirements: Vec, + pub notes: String, } pub fn export_runtime_review_decision_template( detail: &SessionDetail, thread_read: &AgentRuntimeThreadReadModel, workspace_root: &Path, +) -> Result { + sync_runtime_review_decision(detail, thread_read, workspace_root, None) +} + +pub fn save_runtime_review_decision( + detail: &SessionDetail, + thread_read: &AgentRuntimeThreadReadModel, + workspace_root: &Path, + decision: RuntimeReviewDecisionContent, +) -> Result { + sync_runtime_review_decision(detail, thread_read, workspace_root, Some(decision)) +} + +fn sync_runtime_review_decision( + detail: &SessionDetail, + thread_read: &AgentRuntimeThreadReadModel, + workspace_root: &Path, + decision_override: Option, ) -> Result { let session_id = detail.id.trim(); if session_id.is_empty() { @@ -167,7 +188,13 @@ pub fn export_runtime_review_decision_template( })?; let review_checklist = build_review_checklist(); - let document = build_review_decision_document(&analysis, &exported_at, &review_checklist); + let existing_decision = load_existing_review_decision_document(&review_absolute_root)? + .map(|document| document.decision); + let mut document = build_review_decision_document(&analysis, &exported_at, &review_checklist); + let decision = decision_override + .or(existing_decision) + .unwrap_or_else(|| document.decision.clone()); + document.decision = normalize_review_decision_content(decision, &exported_at); let markdown = build_review_decision_markdown(&document); let json = serde_json::to_string_pretty(&document) .map_err(|error| format!("序列化 review decision json 失败: {error}"))?; @@ -208,6 +235,9 @@ pub fn export_runtime_review_decision_template( pending_request_count: analysis.pending_request_count, queued_turn_count: analysis.queued_turn_count, default_decision_status: DEFAULT_DECISION_STATUS.to_string(), + decision: document.decision, + decision_status_options: document.decision_status_options, + risk_level_options: document.risk_level_options, review_checklist, analysis_artifacts: analysis.artifacts.clone(), artifacts, @@ -259,7 +289,7 @@ fn build_review_decision_document( }) .collect(), }, - decision: ReviewDecisionContent { + decision: RuntimeReviewDecisionContent { decision_status: DEFAULT_DECISION_STATUS.to_string(), decision_summary: String::new(), chosen_fix_strategy: String::new(), @@ -271,19 +301,8 @@ fn build_review_decision_document( regression_requirements: Vec::new(), notes: String::new(), }, - decision_status_options: vec![ - "accepted".to_string(), - "deferred".to_string(), - "rejected".to_string(), - "needs_more_evidence".to_string(), - DEFAULT_DECISION_STATUS.to_string(), - ], - risk_level_options: vec![ - "low".to_string(), - "medium".to_string(), - "high".to_string(), - DEFAULT_RISK_LEVEL.to_string(), - ], + decision_status_options: build_decision_status_options(), + risk_level_options: build_risk_level_options(), review_checklist: review_checklist.to_vec(), } } @@ -302,6 +321,21 @@ fn build_review_decision_markdown(document: &ReviewDecisionDocument) -> String { .map(|artifact| format!("- `{}`:`{}`", artifact.title, artifact.relative_path)) .collect::>() .join("\n"); + let decision_status_options = document + .decision_status_options + .iter() + .map(|status| format!("`{status}`")) + .collect::>() + .join(" / "); + let risk_tags = format_markdown_inline_list(&document.decision.risk_tags, "待填写"); + let regression_requirements = + format_markdown_list(&document.decision.regression_requirements, "- 待填写"); + let followup_actions = format_markdown_list(&document.decision.followup_actions, "- 待填写"); + let decision_summary = + format_markdown_text_block(&document.decision.decision_summary, "待填写。"); + let chosen_fix_strategy = + format_markdown_text_block(&document.decision.chosen_fix_strategy, "待填写。"); + let notes = format_markdown_text_block(&document.decision.notes, "待填写。"); format!( "# Lime 人工审核与决策记录\n\n\ @@ -330,22 +364,22 @@ fn build_review_decision_markdown(document: &ReviewDecisionDocument) -> String { {checklist}\n\n\ ## 4. 决策状态\n\ - 当前值:`{decision_status}`\n\ -- 可选值:`accepted` / `deferred` / `rejected` / `needs_more_evidence`\n\n\ +- 可选值:{decision_status_options}\n\n\ ## 5. 决策摘要\n\ -待填写。\n\n\ +{decision_summary}\n\n\ ## 6. 采用的修复策略\n\ -待填写。\n\n\ +{chosen_fix_strategy}\n\n\ ## 7. 风险等级与标签\n\ - 风险等级:`{risk_level}`\n\ -- 风险标签:待填写\n\n\ +- 风险标签:{risk_tags}\n\n\ ## 8. 回归要求\n\ -- 待填写\n\n\ +{regression_requirements}\n\n\ ## 9. 后续动作\n\ -- 待填写\n\n\ +{followup_actions}\n\n\ ## 10. 审核备注\n\ -- 审核人:待填写\n\ -- 审核时间:待填写\n\ -- 备注:待填写\n", +- 审核人:{human_reviewer}\n\ +- 审核时间:{reviewed_at}\n\ +- 备注:\n{notes}\n", decision_status = document.decision.decision_status, exported_at = document.exported_at, title = empty_fallback(&document.review_context.title, "未命名"), @@ -374,7 +408,25 @@ fn build_review_decision_markdown(document: &ReviewDecisionDocument) -> String { } else { checklist }, + decision_status_options = if decision_status_options.is_empty() { + format!("`{DEFAULT_DECISION_STATUS}`") + } else { + decision_status_options + }, + decision_summary = decision_summary, + chosen_fix_strategy = chosen_fix_strategy, risk_level = document.decision.risk_level, + risk_tags = risk_tags, + regression_requirements = regression_requirements, + followup_actions = followup_actions, + human_reviewer = empty_fallback(&document.decision.human_reviewer, "待填写"), + reviewed_at = document + .decision + .reviewed_at + .as_deref() + .filter(|value: &&str| !value.trim().is_empty()) + .unwrap_or("待填写"), + notes = notes, ) } @@ -425,6 +477,162 @@ fn write_review_decision_artifact( }) } +fn load_existing_review_decision_document( + review_absolute_root: &Path, +) -> Result, String> { + let json_path = review_absolute_root.join(REVIEW_DECISION_JSON_FILE_NAME); + if !json_path.exists() { + return Ok(None); + } + + let contents = fs::read_to_string(&json_path).map_err(|error| { + format!( + "读取已有 review decision json 失败 {}: {error}", + json_path.display() + ) + })?; + let document = serde_json::from_str::(&contents).map_err(|error| { + format!( + "解析已有 review decision json 失败 {}: {error}", + json_path.display() + ) + })?; + Ok(Some(document)) +} + +fn normalize_review_decision_content( + decision: RuntimeReviewDecisionContent, + exported_at: &str, +) -> RuntimeReviewDecisionContent { + let decision_status = normalize_review_decision_status(&decision.decision_status); + let decision_summary = normalize_string(&decision.decision_summary); + let chosen_fix_strategy = normalize_string(&decision.chosen_fix_strategy); + let risk_level = normalize_review_risk_level(&decision.risk_level); + let risk_tags = normalize_string_list(&decision.risk_tags); + let human_reviewer = normalize_string(&decision.human_reviewer); + let followup_actions = normalize_string_list(&decision.followup_actions); + let regression_requirements = normalize_string_list(&decision.regression_requirements); + let notes = normalize_string(&decision.notes); + let reviewed_at = normalize_optional_string(decision.reviewed_at.as_deref()).or_else(|| { + let has_review_content = decision_status != DEFAULT_DECISION_STATUS + || !decision_summary.is_empty() + || !chosen_fix_strategy.is_empty() + || risk_level != DEFAULT_RISK_LEVEL + || !risk_tags.is_empty() + || !human_reviewer.is_empty() + || !followup_actions.is_empty() + || !regression_requirements.is_empty() + || !notes.is_empty(); + has_review_content.then(|| exported_at.to_string()) + }); + + RuntimeReviewDecisionContent { + decision_status, + decision_summary, + chosen_fix_strategy, + risk_level, + risk_tags, + human_reviewer, + reviewed_at, + followup_actions, + regression_requirements, + notes, + } +} + +fn build_decision_status_options() -> Vec { + vec![ + "accepted".to_string(), + "deferred".to_string(), + "rejected".to_string(), + "needs_more_evidence".to_string(), + DEFAULT_DECISION_STATUS.to_string(), + ] +} + +fn build_risk_level_options() -> Vec { + vec![ + "low".to_string(), + "medium".to_string(), + "high".to_string(), + DEFAULT_RISK_LEVEL.to_string(), + ] +} + +fn normalize_review_decision_status(value: &str) -> String { + match value.trim() { + "accepted" => "accepted".to_string(), + "deferred" => "deferred".to_string(), + "rejected" => "rejected".to_string(), + "needs_more_evidence" => "needs_more_evidence".to_string(), + "pending_review" => "pending_review".to_string(), + _ => DEFAULT_DECISION_STATUS.to_string(), + } +} + +fn normalize_review_risk_level(value: &str) -> String { + match value.trim() { + "low" => "low".to_string(), + "medium" => "medium".to_string(), + "high" => "high".to_string(), + "unknown" => "unknown".to_string(), + _ => DEFAULT_RISK_LEVEL.to_string(), + } +} + +fn normalize_string(value: &str) -> String { + value.trim().to_string() +} + +fn normalize_optional_string(value: Option<&str>) -> Option { + value + .map(str::trim) + .filter(|value| !value.is_empty()) + .map(ToString::to_string) +} + +fn normalize_string_list(values: &[String]) -> Vec { + values + .iter() + .map(|value| value.trim()) + .filter(|value| !value.is_empty()) + .map(ToString::to_string) + .collect() +} + +fn format_markdown_text_block(value: &str, placeholder: &str) -> String { + let trimmed = value.trim(); + if trimmed.is_empty() { + placeholder.to_string() + } else { + trimmed.to_string() + } +} + +fn format_markdown_list(values: &[String], placeholder: &str) -> String { + if values.is_empty() { + placeholder.to_string() + } else { + values + .iter() + .map(|value| format!("- {value}")) + .collect::>() + .join("\n") + } +} + +fn format_markdown_inline_list(values: &[String], placeholder: &str) -> String { + if values.is_empty() { + placeholder.to_string() + } else { + values + .iter() + .map(|value| format!("`{value}`")) + .collect::>() + .join(" / ") + } +} + fn empty_fallback<'a>(value: &'a str, fallback: &'a str) -> &'a str { if value.trim().is_empty() { fallback @@ -622,4 +830,64 @@ mod tests { assert!(json.contains("\"executionEnvironmentReference\": \"codex\"")); assert!(json.contains("\"runtimeFactSource\": \"aster-rust\"")); } + + #[test] + fn should_save_runtime_review_decision_and_keep_it_on_reexport() { + let temp_dir = TempDir::new().expect("temp dir"); + let detail = build_detail(); + let thread_read = build_thread_read(); + + export_runtime_review_decision_template(&detail, &thread_read, temp_dir.path()) + .expect("export"); + + let saved = save_runtime_review_decision( + &detail, + &thread_read, + temp_dir.path(), + RuntimeReviewDecisionContent { + decision_status: "accepted".to_string(), + decision_summary: "确认最小修复落在 current 主链。".to_string(), + chosen_fix_strategy: "先补 runtime save 命令,再补 Harness UI。".to_string(), + risk_level: "medium".to_string(), + risk_tags: vec!["runtime".to_string(), "harness".to_string()], + human_reviewer: "Lime Maintainer".to_string(), + reviewed_at: Some("2026-03-27T10:30:00Z".to_string()), + followup_actions: vec!["补 UI 回归".to_string()], + regression_requirements: vec![ + "npm run test:contracts".to_string(), + "Rust 定向测试".to_string(), + ], + notes: "不要把 compat 命令重新接回主线。".to_string(), + }, + ) + .expect("save"); + + assert_eq!(saved.decision.decision_status, "accepted"); + assert_eq!(saved.decision.risk_level, "medium"); + assert_eq!(saved.decision.risk_tags, vec!["runtime", "harness"]); + assert_eq!(saved.decision.human_reviewer, "Lime Maintainer"); + + let json_path = temp_dir + .path() + .join(".lime/harness/sessions/session-1/review/review-decision.json"); + let markdown_path = temp_dir + .path() + .join(".lime/harness/sessions/session-1/review/review-decision.md"); + + let json = fs::read_to_string(&json_path).expect("json"); + assert!(json.contains("\"decisionStatus\": \"accepted\"")); + assert!(json.contains("\"humanReviewer\": \"Lime Maintainer\"")); + assert!(json.contains("\"regressionRequirements\": [")); + + let markdown = fs::read_to_string(&markdown_path).expect("markdown"); + assert!(markdown.contains("确认最小修复落在 current 主链。")); + assert!(markdown.contains("补 UI 回归")); + assert!(markdown.contains("Lime Maintainer")); + + let reexported = + export_runtime_review_decision_template(&detail, &thread_read, temp_dir.path()) + .expect("re-export"); + assert_eq!(reexported.decision.decision_status, "accepted"); + assert_eq!(reexported.decision.human_reviewer, "Lime Maintainer"); + } } diff --git a/src/components/agent/chat/components/HarnessStatusPanel.test.tsx b/src/components/agent/chat/components/HarnessStatusPanel.test.tsx index 6d3c315df..fa2deac52 100644 --- a/src/components/agent/chat/components/HarnessStatusPanel.test.tsx +++ b/src/components/agent/chat/components/HarnessStatusPanel.test.tsx @@ -11,6 +11,7 @@ const { exportAgentRuntimeHandoffBundleMock, exportAgentRuntimeReplayCaseMock, exportAgentRuntimeReviewDecisionTemplateMock, + saveAgentRuntimeReviewDecisionMock, mockToast, } = vi.hoisted(() => ({ exportAgentRuntimeAnalysisHandoffMock: vi.fn(), @@ -18,6 +19,7 @@ const { exportAgentRuntimeHandoffBundleMock: vi.fn(), exportAgentRuntimeReplayCaseMock: vi.fn(), exportAgentRuntimeReviewDecisionTemplateMock: vi.fn(), + saveAgentRuntimeReviewDecisionMock: vi.fn(), mockToast: { success: vi.fn(), error: vi.fn(), @@ -38,6 +40,7 @@ vi.mock("@/lib/api/agentRuntime", async () => { exportAgentRuntimeReplayCase: exportAgentRuntimeReplayCaseMock, exportAgentRuntimeReviewDecisionTemplate: exportAgentRuntimeReviewDecisionTemplateMock, + saveAgentRuntimeReviewDecision: saveAgentRuntimeReviewDecisionMock, }; }); @@ -121,6 +124,29 @@ function renderPanel( return rendered; } +function setInputValue( + input: HTMLInputElement | HTMLTextAreaElement | HTMLSelectElement, + value: string, +) { + const prototype = + input instanceof HTMLTextAreaElement + ? HTMLTextAreaElement.prototype + : input instanceof HTMLSelectElement + ? HTMLSelectElement.prototype + : HTMLInputElement.prototype; + const descriptor = Object.getOwnPropertyDescriptor(prototype, "value"); + descriptor?.set?.call(input, value); + input.dispatchEvent(new Event("input", { bubbles: true })); + input.dispatchEvent(new Event("change", { bubbles: true })); +} + +function findButtonByText(text: string): HTMLButtonElement | null { + return Array.from(document.body.querySelectorAll("button")).find( + (button): button is HTMLButtonElement => + button.textContent?.trim() === text, + ) as HTMLButtonElement | null; +} + function createToolInventory(): AgentRuntimeToolInventory { return { request: { @@ -744,6 +770,76 @@ describe("HarnessStatusPanel", () => { ); }); + it("复制回归命令在未导出时应先自动导出 Replay 样本,再复制 promote / eval / trend 命令", async () => { + exportAgentRuntimeReplayCaseMock.mockResolvedValue({ + session_id: "session-replay-copy-1", + thread_id: "thread-replay-copy-1", + workspace_id: "workspace-replay-copy-1", + workspace_root: "/tmp/workspace-replay-copy-1", + replay_relative_root: + ".lime/harness/sessions/session-replay-copy-1/replay", + replay_absolute_root: + "/tmp/workspace-replay-copy-1/.lime/harness/sessions/session-replay-copy-1/replay", + handoff_bundle_relative_root: + ".lime/harness/sessions/session-replay-copy-1", + evidence_pack_relative_root: + ".lime/harness/sessions/session-replay-copy-1/evidence", + exported_at: "2026-03-27T10:12:00.000Z", + thread_status: "waiting_request", + latest_turn_status: "completed", + pending_request_count: 0, + queued_turn_count: 0, + linked_handoff_artifact_count: 4, + linked_evidence_artifact_count: 4, + recent_artifact_count: 2, + artifacts: [ + { + kind: "grader", + title: "评分说明", + relative_path: + ".lime/harness/sessions/session-replay-copy-1/replay/grader.md", + absolute_path: + "/tmp/workspace-replay-copy-1/.lime/harness/sessions/session-replay-copy-1/replay/grader.md", + bytes: 320, + }, + ], + }); + + renderPanel({ + diagnosticRuntimeContext: { + sessionId: "session-replay-copy-1", + workspaceId: "workspace-replay-copy-1", + providerType: "openai", + model: "gpt-5.4", + executionStrategy: "react", + activeTheme: "default", + selectedTeamLabel: null, + }, + }); + + const copyButton = document.body.querySelector( + 'button[aria-label="复制回归沉淀与验证命令"]', + ) as HTMLButtonElement | null; + + await act(async () => { + copyButton?.click(); + await Promise.resolve(); + await Promise.resolve(); + }); + + expect(exportAgentRuntimeReplayCaseMock).toHaveBeenCalledWith( + "session-replay-copy-1", + ); + expect(navigator.clipboard.writeText).toHaveBeenCalledWith( + 'npm run harness:eval:promote -- --session-id "session-replay-copy-1" --slug "session-replay-copy-1" --title "Replay case session-replay-copy-1"\n' + + "npm run harness:eval\n" + + "npm run harness:eval:trend\n", + ); + expect(mockToast.success).toHaveBeenCalledWith( + "已复制回归沉淀、验证与趋势命令", + ); + }); + it("存在 sessionId 时应支持导出人工审核记录并展示审核模板与清单", async () => { exportAgentRuntimeReviewDecisionTemplateMock.mockResolvedValue({ session_id: "session-review-1", @@ -769,6 +865,26 @@ describe("HarnessStatusPanel", () => { pending_request_count: 1, queued_turn_count: 0, default_decision_status: "pending_review", + decision: { + decision_status: "pending_review", + decision_summary: "", + chosen_fix_strategy: "", + risk_level: "unknown", + risk_tags: [], + human_reviewer: "", + reviewed_at: undefined, + followup_actions: [], + regression_requirements: [], + notes: "", + }, + decision_status_options: [ + "accepted", + "deferred", + "rejected", + "needs_more_evidence", + "pending_review", + ], + risk_level_options: ["low", "medium", "high", "unknown"], review_checklist: [ "先阅读 analysis-brief.md 与 analysis-context.json。", "确认最终决策由人工审核者填写。", @@ -845,6 +961,256 @@ describe("HarnessStatusPanel", () => { expect(mockToast.success).toHaveBeenCalledWith("已导出 2 个人工审核文件"); }); + it("应支持填写并保存人工审核结果", async () => { + exportAgentRuntimeReviewDecisionTemplateMock.mockResolvedValue({ + session_id: "session-review-2", + thread_id: "thread-review-2", + workspace_id: "workspace-review-2", + workspace_root: "/tmp/workspace-review-2", + review_relative_root: ".lime/harness/sessions/session-review-2/review", + review_absolute_root: + "/tmp/workspace-review-2/.lime/harness/sessions/session-review-2/review", + analysis_relative_root: + ".lime/harness/sessions/session-review-2/analysis", + analysis_absolute_root: + "/tmp/workspace-review-2/.lime/harness/sessions/session-review-2/analysis", + handoff_bundle_relative_root: ".lime/harness/sessions/session-review-2", + evidence_pack_relative_root: + ".lime/harness/sessions/session-review-2/evidence", + replay_case_relative_root: + ".lime/harness/sessions/session-review-2/replay", + exported_at: "2026-03-27T10:28:00.000Z", + title: "把外部分析结论回挂为人工审核记录", + thread_status: "waiting_request", + latest_turn_status: "action_required", + pending_request_count: 1, + queued_turn_count: 0, + default_decision_status: "pending_review", + decision: { + decision_status: "pending_review", + decision_summary: "", + chosen_fix_strategy: "", + risk_level: "unknown", + risk_tags: [], + human_reviewer: "", + reviewed_at: undefined, + followup_actions: [], + regression_requirements: [], + notes: "", + }, + decision_status_options: [ + "accepted", + "deferred", + "rejected", + "needs_more_evidence", + "pending_review", + ], + risk_level_options: ["low", "medium", "high", "unknown"], + review_checklist: ["先阅读 analysis-brief.md 与 analysis-context.json。"], + analysis_artifacts: [ + { + kind: "analysis_brief", + title: "外部分析简报", + relative_path: + ".lime/harness/sessions/session-review-2/analysis/analysis-brief.md", + absolute_path: + "/tmp/workspace-review-2/.lime/harness/sessions/session-review-2/analysis/analysis-brief.md", + bytes: 320, + }, + ], + artifacts: [ + { + kind: "review_decision_markdown", + title: "人工审核记录", + relative_path: + ".lime/harness/sessions/session-review-2/review/review-decision.md", + absolute_path: + "/tmp/workspace-review-2/.lime/harness/sessions/session-review-2/review/review-decision.md", + bytes: 512, + }, + ], + }); + saveAgentRuntimeReviewDecisionMock.mockResolvedValue({ + session_id: "session-review-2", + thread_id: "thread-review-2", + workspace_id: "workspace-review-2", + workspace_root: "/tmp/workspace-review-2", + review_relative_root: ".lime/harness/sessions/session-review-2/review", + review_absolute_root: + "/tmp/workspace-review-2/.lime/harness/sessions/session-review-2/review", + analysis_relative_root: + ".lime/harness/sessions/session-review-2/analysis", + analysis_absolute_root: + "/tmp/workspace-review-2/.lime/harness/sessions/session-review-2/analysis", + handoff_bundle_relative_root: ".lime/harness/sessions/session-review-2", + evidence_pack_relative_root: + ".lime/harness/sessions/session-review-2/evidence", + replay_case_relative_root: + ".lime/harness/sessions/session-review-2/replay", + exported_at: "2026-03-27T10:32:00.000Z", + title: "把外部分析结论回挂为人工审核记录", + thread_status: "waiting_request", + latest_turn_status: "action_required", + pending_request_count: 1, + queued_turn_count: 0, + default_decision_status: "pending_review", + decision: { + decision_status: "accepted", + decision_summary: "确认最小修复可以接受。", + chosen_fix_strategy: "先补 runtime save,再补 UI 回归。", + risk_level: "medium", + risk_tags: ["runtime", "ui"], + human_reviewer: "Lime Maintainer", + reviewed_at: "2026-03-27T10:32:00.000Z", + followup_actions: ["补充 HarnessStatusPanel 测试"], + regression_requirements: ["npm run test:contracts", "Rust 定向测试"], + notes: "保持 review decision 主链单一。", + }, + decision_status_options: [ + "accepted", + "deferred", + "rejected", + "needs_more_evidence", + "pending_review", + ], + risk_level_options: ["low", "medium", "high", "unknown"], + review_checklist: ["先阅读 analysis-brief.md 与 analysis-context.json。"], + analysis_artifacts: [ + { + kind: "analysis_brief", + title: "外部分析简报", + relative_path: + ".lime/harness/sessions/session-review-2/analysis/analysis-brief.md", + absolute_path: + "/tmp/workspace-review-2/.lime/harness/sessions/session-review-2/analysis/analysis-brief.md", + bytes: 320, + }, + ], + artifacts: [ + { + kind: "review_decision_markdown", + title: "人工审核记录", + relative_path: + ".lime/harness/sessions/session-review-2/review/review-decision.md", + absolute_path: + "/tmp/workspace-review-2/.lime/harness/sessions/session-review-2/review/review-decision.md", + bytes: 512, + }, + ], + }); + + renderPanel({ + diagnosticRuntimeContext: { + sessionId: "session-review-2", + workspaceId: "workspace-review-2", + providerType: "openai", + model: "gpt-5.4", + executionStrategy: "react", + activeTheme: "default", + selectedTeamLabel: null, + }, + }); + + const fillButton = document.body.querySelector( + 'button[aria-label="填写人工审核结果"]', + ) as HTMLButtonElement | null; + + await act(async () => { + fillButton?.click(); + await Promise.resolve(); + await Promise.resolve(); + }); + + const statusSelect = document.body.querySelector( + 'select[aria-label="决策状态"]', + ) as HTMLSelectElement | null; + const riskSelect = document.body.querySelector( + 'select[aria-label="风险等级"]', + ) as HTMLSelectElement | null; + const reviewerInput = document.body.querySelector( + 'input[aria-label="审核人"]', + ) as HTMLInputElement | null; + const riskTagsInput = document.body.querySelector( + 'input[aria-label="风险标签"]', + ) as HTMLInputElement | null; + const summaryTextarea = document.body.querySelector( + 'textarea[aria-label="决策摘要"]', + ) as HTMLTextAreaElement | null; + const strategyTextarea = document.body.querySelector( + 'textarea[aria-label="采用的修复策略"]', + ) as HTMLTextAreaElement | null; + const regressionsTextarea = document.body.querySelector( + 'textarea[aria-label="回归要求"]', + ) as HTMLTextAreaElement | null; + const followupsTextarea = document.body.querySelector( + 'textarea[aria-label="后续动作"]', + ) as HTMLTextAreaElement | null; + const notesTextarea = document.body.querySelector( + 'textarea[aria-label="审核备注"]', + ) as HTMLTextAreaElement | null; + + await act(async () => { + if (statusSelect) { + setInputValue(statusSelect, "accepted"); + } + if (riskSelect) { + setInputValue(riskSelect, "medium"); + } + if (reviewerInput) { + setInputValue(reviewerInput, "Lime Maintainer"); + } + if (riskTagsInput) { + setInputValue(riskTagsInput, "runtime, ui"); + } + if (summaryTextarea) { + setInputValue(summaryTextarea, "确认最小修复可以接受。"); + } + if (strategyTextarea) { + setInputValue(strategyTextarea, "先补 runtime save,再补 UI 回归。"); + } + if (regressionsTextarea) { + setInputValue(regressionsTextarea, "npm run test:contracts\nRust 定向测试"); + } + if (followupsTextarea) { + setInputValue(followupsTextarea, "补充 HarnessStatusPanel 测试"); + } + if (notesTextarea) { + setInputValue(notesTextarea, "保持 review decision 主链单一。"); + } + await Promise.resolve(); + }); + + const saveButton = findButtonByText("保存审核结果"); + + await act(async () => { + saveButton?.click(); + await Promise.resolve(); + await Promise.resolve(); + }); + + expect(exportAgentRuntimeReviewDecisionTemplateMock).toHaveBeenCalledWith( + "session-review-2", + ); + expect(saveAgentRuntimeReviewDecisionMock).toHaveBeenCalledWith({ + session_id: "session-review-2", + decision_status: "accepted", + decision_summary: "确认最小修复可以接受。", + chosen_fix_strategy: "先补 runtime save,再补 UI 回归。", + risk_level: "medium", + risk_tags: ["runtime", "ui"], + human_reviewer: "Lime Maintainer", + reviewed_at: undefined, + followup_actions: ["补充 HarnessStatusPanel 测试"], + regression_requirements: ["npm run test:contracts", "Rust 定向测试"], + notes: "保持 review decision 主链单一。", + }); + expect(document.body.textContent).toContain("当前人工审核结论"); + expect(document.body.textContent).toContain("确认最小修复可以接受。"); + expect(document.body.textContent).toContain("Lime Maintainer"); + expect(document.body.textContent).toContain("补充 HarnessStatusPanel 测试"); + expect(mockToast.success).toHaveBeenCalledWith("已保存人工审核结果"); + }); + it("一键复制给 AI 在未导出时应先自动导出再复制 copy_prompt", async () => { exportAgentRuntimeAnalysisHandoffMock.mockResolvedValue({ session_id: "session-analysis-copy-1", diff --git a/src/components/agent/chat/components/HarnessStatusPanel.tsx b/src/components/agent/chat/components/HarnessStatusPanel.tsx index e82496ab7..b64ab4b71 100644 --- a/src/components/agent/chat/components/HarnessStatusPanel.tsx +++ b/src/components/agent/chat/components/HarnessStatusPanel.tsx @@ -38,6 +38,7 @@ import type { AgentRuntimeAnalysisHandoff, AgentRuntimeEvidencePack, AgentRuntimeHandoffBundle, + AgentRuntimeSaveReviewDecisionRequest, AgentRuntimeReplayCase, AgentRuntimeReviewDecisionTemplate, AgentRuntimeToolInventory, @@ -54,6 +55,7 @@ import { exportAgentRuntimeHandoffBundle, exportAgentRuntimeReplayCase, exportAgentRuntimeReviewDecisionTemplate, + saveAgentRuntimeReviewDecision, } from "@/lib/api/agentRuntime"; import { Badge } from "@/components/ui/badge"; import { Button } from "@/components/ui/button"; @@ -104,6 +106,7 @@ import { resolveTeamWorkspaceStableProcessingLabel } from "../utils/teamWorkspac import type { CompatSubagentRuntimeSnapshot } from "../utils/compatSubagentRuntime"; import type { TeamRoleDefinition } from "../utils/teamDefinitions"; import { AgentThreadReliabilityPanel } from "./AgentThreadReliabilityPanel"; +import { RuntimeReviewDecisionDialog } from "./RuntimeReviewDecisionDialog"; interface HarnessEnvironmentSummary { skillsCount: number; @@ -524,6 +527,77 @@ function formatReviewDecisionStatusLabel(status?: string): string { } } +function formatReviewDecisionRiskLevelLabel(riskLevel?: string): string { + switch (riskLevel?.trim()) { + case "low": + return "低"; + case "medium": + return "中"; + case "high": + return "高"; + case "unknown": + return "未定"; + default: + return riskLevel?.trim() || "未知"; + } +} + +function slugifyHarnessCase(value: string): string { + const normalized = value + .trim() + .toLowerCase() + .replace(/[^a-z0-9]+/g, "-") + .replace(/^-+|-+$/g, ""); + return normalized || "replay-case"; +} + +function quoteShellArg(value: string): string { + return JSON.stringify(value); +} + +function buildReplayPromotionContext(params: { + replayCase: AgentRuntimeReplayCase; + analysisTitle?: string | null; + reviewTitle?: string | null; +}) { + const titleSource = + params.reviewTitle?.trim() || + params.analysisTitle?.trim() || + `Replay case ${params.replayCase.session_id}`; + const slugSource = + params.reviewTitle?.trim() || + params.analysisTitle?.trim() || + params.replayCase.session_id; + + return { + suiteId: "repo-promoted-replays", + title: titleSource, + slug: slugifyHarnessCase(slugSource), + }; +} + +function buildReplayPromotionCommand(params: { + replayCase: AgentRuntimeReplayCase; + analysisTitle?: string | null; + reviewTitle?: string | null; +}): string { + const context = buildReplayPromotionContext(params); + return [ + "npm run harness:eval:promote --", + `--session-id ${quoteShellArg(params.replayCase.session_id)}`, + `--slug ${quoteShellArg(context.slug)}`, + `--title ${quoteShellArg(context.title)}`, + ].join(" "); +} + +function buildReplayEvalCommand(): string { + return "npm run harness:eval"; +} + +function buildReplayTrendCommand(): string { + return "npm run harness:eval:trend"; +} + function describeAction(action: HarnessFileAction): string { switch (action) { case "read": @@ -1599,7 +1673,10 @@ export function HarnessStatusPanel({ ); const [reviewDecisionTemplate, setReviewDecisionTemplate] = useState(null); + const [reviewDecisionEditorOpen, setReviewDecisionEditorOpen] = + useState(false); const [reviewDecisionExporting, setReviewDecisionExporting] = useState(false); + const [reviewDecisionSaving, setReviewDecisionSaving] = useState(false); const [reviewDecisionExportError, setReviewDecisionExportError] = useState< string | null >(null); @@ -1623,8 +1700,10 @@ export function HarnessStatusPanel({ setAnalysisExportError(null); setAnalysisExporting(false); setReviewDecisionTemplate(null); + setReviewDecisionEditorOpen(false); setReviewDecisionExportError(null); setReviewDecisionExporting(false); + setReviewDecisionSaving(false); }, [currentSessionId]); const registerSectionRef = useCallback( @@ -1689,7 +1768,7 @@ export function HarnessStatusPanel({ const handleExportReplayCase = useCallback(async () => { if (!currentSessionId) { toast.error("当前没有可导出的会话上下文"); - return; + return null; } setReplayExporting(true); @@ -1698,11 +1777,13 @@ export function HarnessStatusPanel({ const replay = await exportAgentRuntimeReplayCase(currentSessionId); setReplayCase(replay); toast.success(`已导出 ${replay.artifacts.length} 个 Replay 样本文件`); + return replay; } catch (error) { const message = error instanceof Error ? error.message : "导出 Replay 样本失败"; setReplayExportError(message); toast.error(message); + return null; } finally { setReplayExporting(false); } @@ -1754,6 +1835,44 @@ export function HarnessStatusPanel({ } }, [analysisHandoff, handleExportAnalysisHandoff]); + const handleCopyReplayPromotionCommand = useCallback(async () => { + if (typeof navigator === "undefined" || !navigator.clipboard?.writeText) { + toast.error("当前环境不支持剪贴板复制"); + return; + } + + const replay = replayCase || (await handleExportReplayCase()); + if (!replay) { + return; + } + + const promoteCommand = buildReplayPromotionCommand({ + replayCase: replay, + analysisTitle: analysisHandoff?.title, + reviewTitle: reviewDecisionTemplate?.title, + }); + const evalCommand = buildReplayEvalCommand(); + const trendCommand = buildReplayTrendCommand(); + + try { + await navigator.clipboard.writeText( + `${promoteCommand}\n${evalCommand}\n${trendCommand}\n`, + ); + toast.success("已复制回归沉淀、验证与趋势命令"); + } catch (error) { + toast.error( + error instanceof Error + ? error.message + : "复制回归沉淀、验证与趋势命令失败", + ); + } + }, [ + analysisHandoff?.title, + handleExportReplayCase, + replayCase, + reviewDecisionTemplate?.title, + ]); + const handleExportReviewDecisionTemplate = useCallback(async () => { if (!currentSessionId) { toast.error("当前没有可导出的会话上下文"); @@ -1779,6 +1898,38 @@ export function HarnessStatusPanel({ } }, [currentSessionId]); + const handleOpenReviewDecisionEditor = useCallback(async () => { + const template = + reviewDecisionTemplate || (await handleExportReviewDecisionTemplate()); + if (!template) { + return; + } + + setReviewDecisionTemplate(template); + setReviewDecisionEditorOpen(true); + }, [handleExportReviewDecisionTemplate, reviewDecisionTemplate]); + + const handleSaveReviewDecision = useCallback( + async (request: AgentRuntimeSaveReviewDecisionRequest) => { + setReviewDecisionSaving(true); + setReviewDecisionExportError(null); + try { + const template = await saveAgentRuntimeReviewDecision(request); + setReviewDecisionTemplate(template); + setReviewDecisionEditorOpen(false); + toast.success("已保存人工审核结果"); + } catch (error) { + const message = + error instanceof Error ? error.message : "保存人工审核结果失败"; + setReviewDecisionExportError(message); + toast.error(message); + } finally { + setReviewDecisionSaving(false); + } + }, + [], + ); + const hasToolInventorySection = toolInventoryLoading || Boolean(toolInventoryError) || @@ -3123,6 +3274,24 @@ export function HarnessStatusPanel({ ? "刷新 Replay 样本" : "导出 Replay 样本"} + {replayCase ? ( + {reviewDecisionTemplate ? (