From 208a231113afab5e5455733df733da4027f94f7d Mon Sep 17 00:00:00 2001 From: zhuyongxin Date: Fri, 17 Jul 2026 18:53:29 +0800 Subject: [PATCH] test(graph): establish diagnosis stategraph suite --- devflow/index.md | 1 + .../acceptance.md | 50 +++ .../brief.md | 20 + .../decisions.md | 121 ++++++ .../evidence.md | 42 ++ .../.archive-ready | 1 + .../.committed | 3 + .../design.md | 83 ++++ .../proposal.md | 63 +++ .../spec.md | 25 ++ .../spec.md | 121 ++++++ .../tasks.md | 35 ++ .../spec.md | 18 +- .../spec.md | 124 ++++++ ...va => DiagnosisGraphNodeContractTest.java} | 82 +++- .../DiagnosisGraphTestSuiteStructureTest.java | 44 ++ ...t.java => DiagnosisGraphWorkflowTest.java} | 44 +- .../agent/hook/VerifierInputHookTest.java | 380 ------------------ .../ChatServiceGraphIntegrationTest.java | 6 + 19 files changed, 868 insertions(+), 395 deletions(-) create mode 100644 devflow/projects/2026-07-17-chat-diagnosis-stategraph-test-suite/acceptance.md create mode 100644 devflow/projects/2026-07-17-chat-diagnosis-stategraph-test-suite/brief.md create mode 100644 devflow/projects/2026-07-17-chat-diagnosis-stategraph-test-suite/decisions.md create mode 100644 devflow/projects/2026-07-17-chat-diagnosis-stategraph-test-suite/evidence.md create mode 100644 openspec/changes/archive/2026-07-17-chat-diagnosis-stategraph-test-suite/.archive-ready create mode 100644 openspec/changes/archive/2026-07-17-chat-diagnosis-stategraph-test-suite/.committed create mode 100644 openspec/changes/archive/2026-07-17-chat-diagnosis-stategraph-test-suite/design.md create mode 100644 openspec/changes/archive/2026-07-17-chat-diagnosis-stategraph-test-suite/proposal.md create mode 100644 openspec/changes/archive/2026-07-17-chat-diagnosis-stategraph-test-suite/specs/chat-diagnosis-stategraph-design-freeze/spec.md create mode 100644 openspec/changes/archive/2026-07-17-chat-diagnosis-stategraph-test-suite/specs/chat-diagnosis-stategraph-test-suite/spec.md create mode 100644 openspec/changes/archive/2026-07-17-chat-diagnosis-stategraph-test-suite/tasks.md create mode 100644 openspec/specs/chat-diagnosis-stategraph-test-suite/spec.md rename src/test/java/com/superbiz/agent/graph/diagnosis/{DiagnosisRealGraphIntegrationTest.java => DiagnosisGraphNodeContractTest.java} (82%) create mode 100644 src/test/java/com/superbiz/agent/graph/diagnosis/DiagnosisGraphTestSuiteStructureTest.java rename src/test/java/com/superbiz/agent/graph/diagnosis/{DiagnosisGraphRoutingTest.java => DiagnosisGraphWorkflowTest.java} (93%) delete mode 100644 src/test/java/com/superbiz/agent/hook/VerifierInputHookTest.java diff --git a/devflow/index.md b/devflow/index.md index 90679c7..dd58c21 100644 --- a/devflow/index.md +++ b/devflow/index.md @@ -4,6 +4,7 @@ | 日期 | slug | 说明 | 领域 | 关键词 | 关联 OpenSpec | 状态 | |---|---|---|---|---|---|---| +| 2026-07-17 | chat-diagnosis-stategraph-test-suite | 建立 Workflow、Node Contract、Chat Integration 三层权威测试体系并退役旧 Hook implementation tests。 | Chat diagnosis orchestration/testing | workflow test, node contract, Chat integration, coverage matrix, Hook test retirement | openspec/changes/archive/2026-07-17-chat-diagnosis-stategraph-test-suite | archived | | 2026-07-17 | chat-diagnosis-stategraph-chatservice-cutover | 将复杂 Chat 单轨切换到 Diagnosis StateGraph,并增加 Run 级 orchestration trace 和 verified-only Verifier 输入。 | Chat diagnosis orchestration/production cutover | ChatService, CompiledGraph stream, runId metadata, orchestration trace, verified-only prompt, V012 | openspec/changes/archive/2026-07-17-chat-diagnosis-stategraph-chatservice-cutover | archived | | 2026-07-17 | chat-diagnosis-stategraph-real-nodes | 接入真实 Agent/Java Nodes、显式 Gatekeeper、可信输入投影、关键证据补查与安全 Fallback,暂不切换生产入口。 | Chat diagnosis orchestration/nodes | ReactAgent adapter, Gatekeeper node, verified input, evidence retry, safe fallback, CompiledGraph | openspec/changes/archive/2026-07-17-chat-diagnosis-stategraph-real-nodes | archived | | 2026-07-17 | chat-diagnosis-stategraph-routing-skeleton | 实现未接生产入口的 Diagnosis StateGraph 骨架、有限路由和 Fake Node 测试。 | Chat diagnosis orchestration/graph | StateGraph, fake node, conditional edge, retry counter, orchestration events, trace builder | openspec/changes/archive/2026-07-17-chat-diagnosis-stategraph-routing-skeleton | archived | diff --git a/devflow/projects/2026-07-17-chat-diagnosis-stategraph-test-suite/acceptance.md b/devflow/projects/2026-07-17-chat-diagnosis-stategraph-test-suite/acceptance.md new file mode 100644 index 0000000..15aa646 --- /dev/null +++ b/devflow/projects/2026-07-17-chat-diagnosis-stategraph-test-suite/acceptance.md @@ -0,0 +1,50 @@ +# Chat Diagnosis StateGraph Test Suite 验收 + +## 结果 + +已接受。OpenSpec tasks 21/21 完成,阶段 4为 test-only,无生产行为变更。 + +## 验证 + +### 静态验证 + +- `git diff --check`:通过。 +- `git diff --name-only HEAD -- src/main`:0 个文件。 +- test inventory:三个权威类均存在;`ChatServiceSequentialAgentTest`/`VerifierInputHookTest` 均不存在;`ScriptedDiagnosisGraphActions` 定义 1 处。 +- `openspec validate chat-diagnosis-stategraph-test-suite --strict`:通过。 +- `openspec validate --specs --strict`:15 passed,0 failed。 + +### 脚本验证 + +- 三层权威 suite:4 suites / 43 tests,0 failures/errors/skipped。 +- 完整保留安全回归:31 suites / 126 tests,0 failures/errors/skipped。 +- `mvn -q -DskipTests test-compile`:通过。 + +### 浏览器/人工验证 + +- 未运行;阶段 4仅重构自动化测试,且用户要求最终人工/live 验收到阶段 5统一执行。 + +### 未验证 + +- 未使用 Maven 启动应用做 live E2E。 +- 未检查 `logs/`。 +- 未执行 `scripts/query_mysql.py`。 +- 剩余风险:真实模型、工具、Flyway/MySQL 和最终 Trace 内容仍需阶段 5 E2E/log/DB 证据。 + +## 已完成范围 + +- 建立 Workflow、Node Contract、Chat Integration 三层权威测试体系。 +- 补齐 ceiling/second LOW_CONFID、tool-failure legal snapshot、partial-pass REJECT 和 Verifier invalid status 边界。 +- 删除旧 Hook implementation test,并保留 parser/Gatekeeper/projection/Composer 等安全回归。 +- 用结构测试和 source gate 防止旧 Sequential/Hook 测试回归。 + +## 已知限制 + +- 生产 `VerifierInputHook`/`VerifierContextHolder` 类型仍存在但已无生产/权威测试消费者;阶段 5清理。 +- component tests 与权威层存在有意的分层 overlap,详见 `evidence.md`。 + +## 交接 + +- OpenSpec archive:已同步 2 份 delta specs,并归档到 `openspec/changes/archive/2026-07-17-chat-diagnosis-stategraph-test-suite/`。 +- 下一步:审查精确 Git diff并完成阶段 4独立提交,之后启动阶段 5。 +- OpenSpec 归档确认:用户已授权直接执行后续归档;归档已完成。 diff --git a/devflow/projects/2026-07-17-chat-diagnosis-stategraph-test-suite/brief.md b/devflow/projects/2026-07-17-chat-diagnosis-stategraph-test-suite/brief.md new file mode 100644 index 0000000..749d585 --- /dev/null +++ b/devflow/projects/2026-07-17-chat-diagnosis-stategraph-test-suite/brief.md @@ -0,0 +1,20 @@ +# Chat Diagnosis StateGraph Test Suite Brief + +## 背景 + +- 用户目标:将 ISS-011 阶段 4作为独立 sm-flow,建立以 Graph 路径和外部行为为中心的新测试体系。 +- 当前问题:阶段 1–3 覆盖充分但组织分散,旧 Hook payload test 仍会约束已退出生产的隐式状态机。 +- 关联 OpenSpec:`openspec/changes/chat-diagnosis-stategraph-test-suite/` +- devflow 分档:complex;接口影响 L1 test-only。 + +## 范围 + +- 本次要做:Workflow/Node Contract/Chat Integration 三层权威入口、Issue 路径矩阵补齐、旧 Hook test 退役、保留安全回归。 +- 本次不做:不改生产源码,不删除生产 Hook/ThreadLocal 类型,不运行 live E2E/log/DB 验收。 +- 影响区域:Diagnosis Graph tests、ChatService integration test、OpenSpec/devflow。 + +## OpenSpec 对齐 + +- proposal/design/specs:已覆盖,2 个 delta capabilities。 +- tasks:21/21 已完成。 +- 生产行为:无变化,`src/main` diff=0。 diff --git a/devflow/projects/2026-07-17-chat-diagnosis-stategraph-test-suite/decisions.md b/devflow/projects/2026-07-17-chat-diagnosis-stategraph-test-suite/decisions.md new file mode 100644 index 0000000..835d88f --- /dev/null +++ b/devflow/projects/2026-07-17-chat-diagnosis-stategraph-test-suite/decisions.md @@ -0,0 +1,121 @@ +# Chat Diagnosis StateGraph Test Suite Decisions + +## Entry Summary + +- 问题:阶段 1–3 的安全覆盖已充足,但测试命名/组织仍是逐步实现产物,尚未形成 ISS-011 指定的 Workflow、Node Contract、Chat Integration 三层权威体系。 +- 期望:阶段 4只重构测试结构并补齐矩阵,不改变生产行为;归档并提交后才进入阶段 5。 +- 分档:complex(路径矩阵广、涉及旧安全测试退役,但接口影响为 L1 test-only)。 +- Change:`chat-diagnosis-stategraph-test-suite`。 +- 授权:用户已要求直接实现,阶段门禁与阶段 5 才 live E2E 的约束不变。 + +## Context Sources + +- ISS-011 阶段 4、测试策略和验收标准。 +- `chat-diagnosis-stategraph-design-freeze` 的 test migration requirement。 +- 阶段 1–3 archive/acceptance 和当前 119-test focused baseline。 +- `DiagnosisGraphRoutingTest`、各 Node tests、`DiagnosisRealGraphIntegrationTest`、`ChatDiagnosisGraphRuntimeTest`、`ChatServiceGraphIntegrationTest`。 +- `VerifierInputHookTest`、protocol parser tests、`ExecutorGatekeeperServiceTest` 与 Trace/Controller/Repository/Eval tests。 + +## Question Pool + +| # | 维度 | 问题 | 模式 | 状态 | +|---|---|---|---|---| +| Q1 | 术语 | Workflow、Node Contract、Chat Integration 的职责边界是什么? | evidence-driven | 已解决 | +| Q2 | 边界 | 是否把所有现有 Graph unit tests 合并成三个巨型类? | evidence-driven | 已解决 | +| Q3 | 旧测试 | `VerifierInputHookTest` 在显式 Graph Gatekeeper 后应保留、改写还是删除? | evidence-driven | 已解决 | +| Q4 | 验收 | 如何证明阶段 4矩阵完整且没有恢复固定 Sequential 顺序? | evidence-driven | 已解决 | +| Q5 | 阶段 | 是否允许为测试可测性改生产代码或执行 live E2E? | user-interview(既有冻结规则) | 已确认 | + +## Evidence-driven Findings + +- Q1:Issue 已明确三类测试;现有 `DiagnosisGraphRoutingTest` 对应 Workflow,分散 Node/Protocol tests 对应 Node Contract,`ChatServiceGraphIntegrationTest` 对应外部生命周期。 +- Q2:现有细粒度测试失败定位清晰,全部合并会制造大文件;应保留专用 unit tests,同时新增/重命名三层权威入口并共享夹具。 +- Q3:生产 Graph Verifier 已不注册 Hook,Hook test 仍验证 raw/full-trace/ThreadLocal payload,与 verified-only 生产协议冲突;parser/Gatekeeper/投影行为已有独立 tests,故阶段 4删除 Hook test,生产类型留到阶段 5。 +- Q4:以 Issue 必需路径清单建立 requirement-to-test matrix;source check 禁止 `SequentialAgent`/`VerifierInputHookTest` 成为新 suite 依赖,并运行保留安全回归。 + +## User-interview Confirmation + +| 问题 | 用户原话/既有确认 | 状态 | OpenSpec 回写 | +|---|---|---|---| +| Q5 阶段边界 | “端到端只在最后阶段全部完成后才验证;每个阶段如果有必要添加单元测试验收的话,就加” | 已确认 | proposal | + +## Grill-with-docs Result + +- 术语不进入业务 glossary:Workflow/Node Contract/Integration 是测试架构术语,不改变 Session、Run、Trace、Gatekeeper 或 evidence gap 领域定义。 +- 具体场景压力测试:同 session 多 run 属于 Chat Integration;Gatekeeper REJECT/LOW_CONFID 与 retry exhaustion 属于 Workflow;failed binding 过滤与 verified-only payload 属于 Node Contract。 +- 旧 Hook test 的有效行为已分别迁移到 `ExecutorEvidenceParserTest`、`ExecutorGatekeeperServiceTest`、`GatekeeperNodeTest` 和 `VerifiedInputNodeTest`;删除不会丢失安全真理源。 +- 没有难以逆转的新架构取舍,不创建 ADR;测试组织可以在保持行为矩阵的前提下继续演进。 + +## Discover Status + +- `devflow/index.md`:命中阶段 0–3 archive。 +- 生产调用链:阶段 3 已冻结且测试阶段默认不修改。 +- 接口影响:L1 test-only;无 API/DTO/DB/Prompt/运行时消费者变化。 +- 未解决问题:0。 +- Draft 产物:proposal + decisions;尚未生成 design/spec/tasks,尚未修改测试代码。 + +## Architecture Audit + +### Test ownership map + +`ISS-011 path matrix -> DiagnosisGraphWorkflowTest -> ScriptedDiagnosisGraphActions -> real DiagnosisGraphFactory` 负责控制流;`Node/protocol contracts -> DiagnosisGraphNodeContractTest + focused component tests -> real Node actions/parsers/Gatekeeper` 负责安全投影;`public lifecycle -> ChatServiceGraphIntegrationTest -> Run repositories/Eval/Trace mapping` 负责外部行为。Controller、Repository、Trace、Eval 和 protocol tests 是三层体系的下游安全消费者,不应被重写成 Graph 内部顺序断言。 + +### Data and lifecycle ownership + +- Workflow fixtures 只拥有脚本状态、调用计数和 events,不创建 Session/Run。 +- Node Contract fixtures 只拥有 invoker input/output 和 mock Gatekeeper current-run result,不持久化生产实体。 +- Chat Integration fixtures 只观察 ChatService public result 和 current Run persistence,不推断内部 Node 次序。 +- `VerifierInputHookTest` 删除后不产生数据契约缺口:Executor parser、Gatekeeper、passed-binding projection 各自已有单一测试所有者。 + +### Coupling risks + +- 最大风险是 Workflow 与 Node Contract 都断言完整路径而重复;设计将 Fake route matrix 与 real-node input/security matrix分开。 +- `ScriptedDiagnosisGraphActions` 是唯一 Fake topology fixture;不新增第二套 Graph builder。 +- 测试-only阶段禁止 `src/main` diff,避免为测试便利扩大 production API。 +- 类重命名使用 Git rename,旧名称仅允许出现在 OpenSpec/devflow迁移说明中。 + +### Cross-artifact alignment + +| 上游 → 下游 | 检查内容 | 状态 | +|---|---|---| +| brief/proposal → proposal | 三层体系、旧 Hook test退役、保留安全回归、阶段 5 E2E 延期 | 已对齐 | +| proposal → design | rename-not-copy、职责归属、production diff=0、rollback | 已对齐 | +| design → specs/tasks | Workflow/Node/Integration矩阵、Hook test删除、source inventory和验证门禁 | 已对齐 | +| specs → tasks | 每类 scenario 均有 rename、补缺、回归或静态验收任务 | 已对齐 | + +### Audit result + +架构审计未发现业务 glossary、阶段 0–3 specs 或生产行为冲突。新 capability 只描述测试验证系统,modified design-freeze requirement 完整保留并细化旧测试替换边界。接口影响保持 L1,cross-artifact gap=0,无需回写生产 spec 或创建 ADR。 + +## Commit Gate + +- schema:spec-driven;proposal/design/2 delta specs/tasks 全部 done,applyRequires=`tasks` 已满足。 +- OpenSpec:当前 change strict pass;15 个主 specs strict pass。 +- Cross-artifact:4/4 已对齐,gap=0。 +- Question pool:4 个 evidence-driven 已查证,1 个 user-interview 由用户既有原话确认,无未决项。 +- Interface impact:L1 test-only;design 有独立影响/回滚章节,默认 `src/main` diff=0。 +- Preflight:`git diff --check` 通过,当前仅 Draft OpenSpec/devflow 与根目录计划文件,无测试/生产代码修改。 +- 结论:Draft OpenSpec 达到可执行状态,创建 `.committed` 后进入 Apply。 + +## Apply Completion + +- `DiagnosisGraphRoutingTest` 已以 Git rename 演进为 `DiagnosisGraphWorkflowTest`;所有 scripted run 统一断言 events 与真实 sequence 完全一致。 +- `DiagnosisRealGraphIntegrationTest` 已演进为 `DiagnosisGraphNodeContractTest`;补齐合法工具失败限制、partial-pass REJECT 和 Verifier invalid 无伪 verdict。 +- Workflow 显式断言 Gatekeeper ceiling 将模型 PASS 限制为 LOW_CONFID 且不触发 evidence retry,以及第二次 LOW_CONFID 不再补证据。 +- `ChatServiceGraphIntegrationTest` 增加 SessionContextHolder finally cleanup 断言;public Run/Trace/Eval/multi-run 覆盖保持。 +- `VerifierInputHookTest` 已删除;生产 Hook/ThreadLocal 类型未修改,留待阶段 5清理。 +- 新增 `DiagnosisGraphTestSuiteStructureTest`,保证三个权威类存在、Sequential/Hook实现测试不存在且权威测试不引用旧实现。 + +## Verification Summary + +- 新权威层:4 suites / 43 tests,0 failures/errors/skipped。 +- 完整保留安全回归:31 suites / 126 tests,0 failures/errors/skipped。 +- Maven test compilation:通过。 +- OpenSpec:当前 change strict pass;主 specs 15/15 strict pass。 +- 静态门禁:`git diff --check` 通过;required authoritative classes=3;legacy tests=0;scripted fixture definitions=1;`src/main` diff=0。 +- 按阶段门禁未运行 Maven live E2E、未检查 `logs/`、未执行 `scripts/query_mysql.py`;统一保留到阶段 5。 + +## Archive Result + +- 2 份 delta specs 已同步:新增 test-suite capability 6 条 requirements,修改 design-freeze test migration requirement 1 条。 +- OpenSpec 已归档到 `openspec/changes/archive/2026-07-17-chat-diagnosis-stategraph-test-suite/`。 diff --git a/devflow/projects/2026-07-17-chat-diagnosis-stategraph-test-suite/evidence.md b/devflow/projects/2026-07-17-chat-diagnosis-stategraph-test-suite/evidence.md new file mode 100644 index 0000000..282124f --- /dev/null +++ b/devflow/projects/2026-07-17-chat-diagnosis-stategraph-test-suite/evidence.md @@ -0,0 +1,42 @@ +# Chat Diagnosis StateGraph Test Suite Evidence + +## Requirement-to-test Matrix + +| ISS-011 路径/边界 | 权威测试 | 辅助证据 | +|---|---|---| +| PASS 与精确 event/transition | `DiagnosisGraphWorkflowTest.normalPassPathUsesCompiledGraphAndPreservesEventOrder` | `DiagnosisGraphNodeContractTest.compiledRealNodeGraphCompletesPassPathInExactOrder` | +| Planner 一次技术重试/耗尽/非重试失败 | `DiagnosisGraphWorkflowTest.planner*` | Planner adapter focused test | +| Executor FAILED/TOOL_BLOCKED/INVALID/no-evidence | `DiagnosisGraphWorkflowTest.executor*` | `DiagnosisGraphNodeContractTest.legalSnapshotAfterToolFailureRemainsCompletedAndReachesGatekeeper` | +| Gatekeeper PASS/LOW/REJECT/unknown | `DiagnosisGraphWorkflowTest.gatekeeper*` | Gatekeeper Node/service tests | +| partial-pass REJECT 不泄漏 claim | Workflow unsafe route | `DiagnosisGraphNodeContractTest.gatekeeperRejectSkipsVerifierAndUsesPreVerificationFallback` | +| verified-only passed-binding projection | Workflow VerifiedInput route | `VerifiedInputNodeTest` + Verifier adapter test | +| Verifier 技术重试/耗尽/非重试失败 | `DiagnosisGraphWorkflowTest.verifier*` | `DiagnosisGraphNodeContractTest.verifierInvalidOutputSetsExecutionStatusWithoutFabricatedVerdict` | +| critical evidence retry、无 gap、non-critical、ceiling、第二次 LOW | `DiagnosisGraphWorkflowTest.*LowConfidence*` / `secondLowConfidence*` | EvidenceRetryPrepareNodeTest + real full-snapshot integration | +| 第二轮完整 snapshot 再验真 | Workflow evidence retry | `DiagnosisGraphNodeContractTest.criticalGapPerformsOneIncrementalRoundAndRevalidatesCompleteSnapshot` | +| Verifier REJECT 安全表达 | `DiagnosisGraphWorkflowTest.verifierRejectStillRunsComposerWithSafeMaterial` | ComposerSafeInputBuilderTest | +| Composer retry/耗尽/非重试/安全 Fallback | `DiagnosisGraphWorkflowTest.composer*` | ComposerNodeAdapterTest + FallbackNodeTest | +| ChatResult/Run/Trace/evaluation/cleanup | `ChatServiceGraphIntegrationTest` | Controller/Trace/Repository/Eval tests | +| 同 session 多 run 隔离 | `ChatServiceGraphIntegrationTest.sameSessionCreatesDistinctRunIdsAndKeepsTracePerRun` | Run repository/Trace exact-run tests | +| 三层结构与旧实现测试退役 | `DiagnosisGraphTestSuiteStructureTest` | source inventory command | + +## 旧 Hook test 安全映射 + +| 旧行为 | 新真理源 | +|---|---| +| fenced/prefixed/malformed Executor JSON | `ExecutorEvidenceParserTest` | +| invocation/raw_path/evidence excerpt真实性 | `ExecutorGatekeeperServiceTest` | +| Gatekeeper status/severity/audit | `GatekeeperNodeTest` | +| passed binding 精确投影 | `VerifiedInputNodeTest` | +| Verifier verified-only payload | `VerifierNodeAdapterTest` + `ChatVerifierPromptContractTest` | + +## 验证证据 + +- 31 suites / 126 tests:0 failures、0 errors、0 skipped。 +- test compilation、当前 change strict、主 specs 15/15、diff check 全通过。 +- `src/main` diff=0;三个权威类存在;旧 Sequential/Hook tests不存在;Scripted fixture 定义唯一。 + +## Intentional overlap and limits + +- Workflow 与 Node Contract 都覆盖 PASS/REJECT,但前者验证 route/event,后者验证真实输入/安全材料;这是分层证据,不是复制 fixture。 +- 细粒度 component tests 保留以定位失败,不要求全部搬进三个权威类。 +- 真实模型/工具、日志和数据库行为未在阶段 4验证,统一由阶段 5 live E2E承担。 diff --git a/openspec/changes/archive/2026-07-17-chat-diagnosis-stategraph-test-suite/.archive-ready b/openspec/changes/archive/2026-07-17-chat-diagnosis-stategraph-test-suite/.archive-ready new file mode 100644 index 0000000..8b13789 --- /dev/null +++ b/openspec/changes/archive/2026-07-17-chat-diagnosis-stategraph-test-suite/.archive-ready @@ -0,0 +1 @@ + diff --git a/openspec/changes/archive/2026-07-17-chat-diagnosis-stategraph-test-suite/.committed b/openspec/changes/archive/2026-07-17-chat-diagnosis-stategraph-test-suite/.committed new file mode 100644 index 0000000..c852650 --- /dev/null +++ b/openspec/changes/archive/2026-07-17-chat-diagnosis-stategraph-test-suite/.committed @@ -0,0 +1,3 @@ +committed_at: 2026-07-17 +checkpoint: Commit +authorization: user-requested-direct-implementation diff --git a/openspec/changes/archive/2026-07-17-chat-diagnosis-stategraph-test-suite/design.md b/openspec/changes/archive/2026-07-17-chat-diagnosis-stategraph-test-suite/design.md new file mode 100644 index 0000000..c676f0b --- /dev/null +++ b/openspec/changes/archive/2026-07-17-chat-diagnosis-stategraph-test-suite/design.md @@ -0,0 +1,83 @@ +## Context + +阶段 3 已将复杂 Chat 单轨切换到 Diagnosis StateGraph,并以 119 个 focused tests 证明生产入口、路由、Node、Trace 和 Eval。当前测试实现按阶段累积:`DiagnosisGraphRoutingTest` 是完整 Fake workflow,`DiagnosisRealGraphIntegrationTest` 验证真实 Node assembly,多个专用 Node/protocol tests 锁定局部契约,`ChatServiceGraphIntegrationTest` 验证 Run 生命周期;另有 `VerifierInputHookTest` 仍锁定已退出生产路径的 raw/full-trace/ThreadLocal payload。 + +ISS-011 阶段 4只改变测试架构。生产源码、API、DTO、DB、Prompt 和路由均不得修改。目标是建立三个权威入口,同时保留小而精确的专用 tests,避免把全部覆盖合并成难维护的大类。 + +## Goals / Non-Goals + +**Goals:** + +- 用 `DiagnosisGraphWorkflowTest` 表达所有条件边、有限重试、终止和 event 顺序。 +- 用 `DiagnosisGraphNodeContractTest` 表达真实 Node assembly 的输入投影、config identity、Gatekeeper/verified evidence、status/verdict 和安全材料边界。 +- 用 `ChatServiceGraphIntegrationTest` 表达 public ChatResult、Run/Trace/evaluation、SUCCESS/FAILED 和 multi-run 隔离。 +- 建立 Issue 阶段 4路径矩阵,补齐 partial-pass REJECT、tool-failure-but-valid-output、invalid Verifier 无伪 verdict 等高价值缺口。 +- 删除 `VerifierInputHookTest`,并证明其仍有价值的 parser/Gatekeeper/投影行为已由独立 tests承接。 + +**Non-Goals:** + +- 不修改生产实现或为了测试引入额外 public seam。 +- 不删除生产 `VerifierInputHook`/`VerifierContextHolder` 类型;阶段 5清理。 +- 不强制删除所有细粒度 Node/protocol tests。 +- 不运行 Maven live E2E、日志或真实数据库验收。 + +## Decisions + +### 1. 重命名现有权威类,不复制路由测试 + +将 `DiagnosisGraphRoutingTest` 原样演进为 `DiagnosisGraphWorkflowTest`。它已经使用 `ScriptedDiagnosisGraphActions` 驱动真实 CompiledGraph,覆盖 counter、edge、sequence 和 trace;复制一份新类只会形成两个路由真理源。 + +替代方案是保留旧类再加 facade/suite class,但 Maven/JUnit suite 依赖和重复发现没有新增验证价值,因此拒绝。 + +### 2. 真实 Node assembly integration 成为 Node Contract 权威入口 + +将 `DiagnosisRealGraphIntegrationTest` 演进为 `DiagnosisGraphNodeContractTest`。该类跨 Planner/Executor/Gatekeeper/VerifiedInput/Verifier/Composer 观察 invoker input、Gatekeeper调用次数、完整 snapshot 和 safe fallback,最符合“节点输入输出契约”的架构边界。 + +专用 `PlannerNodeAdapterTest`、`VerifiedInputNodeTest` 等继续保留,用于精确定位单组件失败;Node Contract 不复制它们全部断言,只补跨节点组合缺口。 + +### 3. 覆盖矩阵按行为归属,不按历史类迁移 + +- Workflow:PASS、Planner/Verifier/Composer retry 与 exhaustion、Executor status、Gatekeeper outcome、evidence retry、Verifier REJECT、终止/trace。 +- Node Contract:adapter/config、合法 no-evidence/工具失败说明、partial-pass REJECT、verified-only projection、ceiling、完整 snapshot revalidation、Composer/Fallback材料。 +- Chat Integration:Run start/complete/fail、metrics/Eval、orchestration trace、self-evaluation、cleanup、multi-run。 + +同一安全边界可以有 unit + integration 两层证据,但不得复制大段 fixture;复用现有 helpers 和 constants。 + +### 4. 删除 Hook test,不迁移旧 payload + +`VerifierInputHookTest` 的 raw Executor、full tool trace、ThreadLocal 和 Hook Gatekeeper payload 与阶段 3 verified-only生产协议冲突。其 parser sanitization 由 `ExecutorEvidenceParserTest`,引用真实性由 `ExecutorGatekeeperServiceTest`,Graph status/audit 由 `GatekeeperNodeTest`,passed-binding projection 由 `VerifiedInputNodeTest` 覆盖。 + +阶段 4删除测试但保留生产类型,避免把阶段 5清理提前混入测试重构。若阶段 5删除类型,已经没有测试迁移阻力。 + +### 5. 测试失败按三类处理,默认 production diff=0 + +若新矩阵发现失败:规格缺口则先修正 OpenSpec;生产实现偏离既有 specs 才允许修复生产代码;测试命名/fixture 偏差只修测试。阶段 4正常验收要求 `src/main` 无 diff,从而保持提交职责单一。 + +## Interface Impact + +- 等级:L1 test-only。 +- 外部行为、API、DTO、DB、Prompt、Graph State 和消费者:无变化。 +- 测试发现生产偏差时才可能升级影响级别,并必须回写 OpenSpec;当前没有此类发现。 +- 回滚:revert 测试重构提交;不涉及数据或运行时迁移。 + +## Risks / Trade-offs + +- [类重命名让历史链接失效] → devflow/archive 记录 old→new 映射,Git rename 可追踪历史。 +- [三个权威类变成巨型文件] → 保留细粒度专用 unit tests,权威类只承载跨组件/路径矩阵。 +- [删除 Hook test 丢失安全边界] → 删除前运行 parser/Gatekeeper/VerifiedInput保留集合并在 acceptance 记录映射。 +- [仅有测试名没有真实矩阵] → tasks 明确补 3 个跨节点缺口并对照 Issue 全列表审计。 +- [repository/integration tests 写本地日志被误认为 live E2E] → 验收明确区分 Maven tests 与 Maven 应用启动;阶段 4不检查日志/数据库。 + +## Migration Plan + +1. Rename Routing→Workflow 和 RealGraphIntegration→NodeContract,不改变测试逻辑,先跑 GREEN。 +2. 以 Issue 矩阵审计现有 method coverage,补 partial-pass REJECT、tool failure但合法结构、Verifier failure无伪 verdict 等缺口。 +3. 删除 `VerifierInputHookTest`,运行对应 parser/Gatekeeper/VerifiedInput回归证明安全覆盖未丢失。 +4. 扩展/复核 ChatService integration 与保留 Controller/Trace/Repository/Composer/Eval tests。 +5. 运行 test compilation、OpenSpec/static gates,归档并独立提交。 + +回滚只需 revert 阶段 4提交,不影响生产运行。 + +## Open Questions + +无。测试职责、旧 Hook test 退役和 E2E 阶段边界均由代码证据、阶段 0 spec 与用户规则确定。 diff --git a/openspec/changes/archive/2026-07-17-chat-diagnosis-stategraph-test-suite/proposal.md b/openspec/changes/archive/2026-07-17-chat-diagnosis-stategraph-test-suite/proposal.md new file mode 100644 index 0000000..d1fc330 --- /dev/null +++ b/openspec/changes/archive/2026-07-17-chat-diagnosis-stategraph-test-suite/proposal.md @@ -0,0 +1,63 @@ +# Chat Diagnosis StateGraph Test Suite + +## Why + +阶段 1–3 已建立完整 Graph 路由、真实 Node 和 ChatService cutover 测试,但核心覆盖仍分散在 `DiagnosisGraphRoutingTest`、多个单 Node test、真实 Graph integration 和已迁移前的 `VerifierInputHookTest`。阶段 4 需要让测试结构与 ISS-011 的正式架构一致:以 Graph workflow、Node contract、ChatService 外部生命周期三层为权威入口,不继续让 Sequential/Hook 内部实现成为长期约束。 + +## What Changes + +- 将纯条件边/重试/终止矩阵收敛为 `DiagnosisGraphWorkflowTest`,保留 Fake Node、精确事件序列和有界循环断言。 +- 建立 `DiagnosisGraphNodeContractTest`,以显式输入白名单、status/verdict 分离、Gatekeeper/verified evidence、retry snapshot 和 Fallback 安全材料为跨节点契约。 +- 扩展 `ChatServiceGraphIntegrationTest`,覆盖 public ChatResult、Run/Trace/self-evaluation/Eval、失败生命周期和同 session 多 run 隔离。 +- 删除已被显式 Graph Node 契约替代的 `VerifierInputHookTest`,但阶段 4 不删除生产 Hook/ThreadLocal 类型;生产清理仍属于阶段 5。 +- 保留 Gatekeeper、Controller、Trace、Repository、Composer、protocol parser、no-evidence、REJECT 和 Eval 的独立安全回归。 +- 增加覆盖矩阵/源级检查,证明 Issue 阶段 4 的必需路径均有权威测试且不存在固定 Sequential 顺序断言。 + +## Capabilities + +### New Capabilities + +- `chat-diagnosis-stategraph-test-suite`:规定 Diagnosis Graph 的 Workflow、Node Contract、ChatService Integration 三层测试体系、必需路径矩阵和旧实现测试退役边界。 + +### Modified Capabilities + +- `chat-diagnosis-stategraph-design-freeze`:将已冻结的“route/node-contract/Chat integration 替换旧 Sequential/Hook 测试”要求落实为具体权威测试类和保留回归集合。 + +## Scope + +### In Scope + +- 测试类重命名/收敛、跨节点契约测试、ChatService integration 分支补齐、覆盖矩阵和测试辅助夹具复用。 +- 删除 `VerifierInputHookTest`,避免已退出生产路径的隐式 Gatekeeper/ThreadLocal payload 继续约束新架构。 +- 运行新的 Graph suite 与必须保留的 Controller/Trace/Repository/Gatekeeper/Composer/Eval 回归。 + +### Out of Scope + +- 不改变生产 Graph 路由、Node、ChatService、Prompt、数据库或 API 行为;若测试发现生产偏差,按实现期冲突规则单独分类。 +- 不删除 `VerifierInputHook`、`VerifierContextHolder` 或其他生产兼容代码;阶段 5 统一清理。 +- 不运行 Maven live E2E、检查 `logs/` 或查询真实数据库;仍保留到阶段 5。 +- 不把所有细粒度 unit test 强制合并成一个巨型文件;三层权威入口与可复用小测试可以共存。 + +## Context Constraints + +- `ChatServiceSequentialAgentTest` 已在阶段 3 由 Graph integration 替代,阶段 4 不恢复任何固定 Agent 顺序断言。 +- Workflow 只验证 Graph 节点/条件边/计数/事件;Node Contract 只验证输入投影、标准输出和安全边界;Chat integration 只从 public service/Run persistence 观察行为。 +- 必须复用 `ScriptedDiagnosisGraphActions` 和现有 protocol/node helpers,禁止复制大段 JSON fixture 或另造第二套 Graph factory。 +- Gatekeeper ceiling 导致的 LOW_CONFID 不得触发 evidence retry;只有有效 critical evidence gap 且总轮次未耗尽才允许补证据。 +- 前置 Fallback 不得泄漏任何 Executor claim;后置 Composer Fallback 只能使用 Verifier 允许材料。 + +## Acceptance + +- `DiagnosisGraphWorkflowTest` 覆盖 PASS、Planner/Verifier/Composer 技术重试与耗尽、Executor failures/no-evidence、Gatekeeper PASS/LOW_CONFID/REJECT、evidence retry、Verifier REJECT 和全部安全终止。 +- `DiagnosisGraphNodeContractTest` 覆盖四类 Agent Adapter、Gatekeeper、Verified Input、Evidence Retry、Fallback、config identity、unknown/failure fail-closed 和 verified-only 输入。 +- `ChatServiceGraphIntegrationTest` 覆盖 SUCCESS、handled Fallback、unhandled/no-answer FAILED、Run metrics/Eval/trace、clean-up 和同 session 多 run 隔离。 +- `VerifierInputHookTest` 不再存在;保留回归集合全部通过,且生产源码在本阶段无行为 diff。 +- Maven test compilation、OpenSpec strict、主 specs strict、`git diff --check` 和测试体系源级检查通过。 +- 阶段 4明确记录未运行最终 live E2E/log/DB 验收。 + +## Risks + +- 仅重命名测试可能掩盖覆盖缺口;必须建立 Issue 路径到 test method 的显式矩阵并补齐缺失分支。 +- 过度合并会降低失败定位;保留专用 unit tests,只让三层类成为架构入口而非唯一文件。 +- 删除 Hook test 可能丢失 parser/Gatekeeper 边界;删除前必须证明对应行为已由 protocol parser、Gatekeeper Node/service 和 verified input tests覆盖。 +- 测试重构若意外修改生产代码会模糊阶段边界;默认生产源码 diff 必须为零。 diff --git a/openspec/changes/archive/2026-07-17-chat-diagnosis-stategraph-test-suite/specs/chat-diagnosis-stategraph-design-freeze/spec.md b/openspec/changes/archive/2026-07-17-chat-diagnosis-stategraph-test-suite/specs/chat-diagnosis-stategraph-design-freeze/spec.md new file mode 100644 index 0000000..4018e6b --- /dev/null +++ b/openspec/changes/archive/2026-07-17-chat-diagnosis-stategraph-test-suite/specs/chat-diagnosis-stategraph-design-freeze/spec.md @@ -0,0 +1,25 @@ +## MODIFIED Requirements + +### Requirement: Test migration design SHALL preserve safety behavior + +The project SHALL replace Sequential/Hook implementation tests with authoritative `DiagnosisGraphWorkflowTest`, `DiagnosisGraphNodeContractTest`, and `ChatServiceGraphIntegrationTest` layers while retaining focused public/security contract tests. Fixed Agent call order and legacy Hook payload shape SHALL NOT remain correctness criteria. + +#### Scenario: Old tests are replaced + +- **WHEN** StateGraph tests become authoritative +- **THEN** `ChatServiceSequentialAgentTest` SHALL NOT exist +- **AND** `VerifierInputHookTest` SHALL NOT exist after explicit Gatekeeper/Verified Input coverage is established +- **AND** no replacement test SHALL restore fixed Sequential Agent order as a public requirement + +#### Scenario: Safety tests are retained + +- **WHEN** the new suite is assembled +- **THEN** Gatekeeper, Controller, Run/Trace, Repository, Composer, parser/projection, no-evidence, REJECT, Eval, and multi-run safety contracts SHALL remain covered +- **AND** Workflow and Node Contract matrices SHALL cover bounded retry, fail-closed routing, verified-only inputs, and deterministic Fallback + +#### Scenario: Authoritative test layers are inspected + +- **WHEN** a maintainer needs to locate ISS-011 orchestration verification +- **THEN** Workflow tests SHALL own route/loop/event behavior +- **AND** Node Contract tests SHALL own input/output/security projection behavior +- **AND** Chat integration tests SHALL own Run lifecycle and public result behavior diff --git a/openspec/changes/archive/2026-07-17-chat-diagnosis-stategraph-test-suite/specs/chat-diagnosis-stategraph-test-suite/spec.md b/openspec/changes/archive/2026-07-17-chat-diagnosis-stategraph-test-suite/specs/chat-diagnosis-stategraph-test-suite/spec.md new file mode 100644 index 0000000..52e4f5b --- /dev/null +++ b/openspec/changes/archive/2026-07-17-chat-diagnosis-stategraph-test-suite/specs/chat-diagnosis-stategraph-test-suite/spec.md @@ -0,0 +1,121 @@ +## ADDED Requirements + +### Requirement: Diagnosis Graph tests SHALL have three authoritative layers + +The test suite SHALL use Workflow, Node Contract, and ChatService Integration layers as the authoritative ISS-011 verification structure while retaining focused component tests for failure localization. + +#### Scenario: Test architecture is inspected +- **WHEN** the Diagnosis Graph test sources are listed +- **THEN** `DiagnosisGraphWorkflowTest`, `DiagnosisGraphNodeContractTest`, and `ChatServiceGraphIntegrationTest` SHALL exist +- **AND** no `ChatServiceSequentialAgentTest` SHALL exist + +#### Scenario: Focused unit tests are retained +- **WHEN** a single parser, adapter, Gatekeeper, projection, retry preparer, trace builder, or Fallback contract fails +- **THEN** a focused component test SHALL be able to identify that boundary +- **AND** the suite SHALL NOT require all component assertions to be duplicated in the three authoritative classes + +### Requirement: Workflow tests SHALL cover every bounded routing class + +`DiagnosisGraphWorkflowTest` SHALL execute the real compiled topology with scripted nodes and SHALL verify path order, retry ownership, bounded termination, and orchestration events without invoking models or tools. + +#### Scenario: Normal and Planner paths are tested +- **WHEN** the Workflow suite runs +- **THEN** it SHALL cover PASS, Planner INVALID_OUTPUT/RETRYABLE_FAILED one-time retry success, second technical failure, NON_RETRYABLE_FAILED, and fail-closed unknown status +- **AND** Planner terminal failures SHALL skip Executor and reach Fallback + +#### Scenario: Executor and Gatekeeper paths are tested +- **WHEN** the Workflow suite runs +- **THEN** it SHALL cover Executor FAILED, TOOL_BLOCKED, INVALID_OUTPUT, legal no-evidence, Gatekeeper PASS, LOW_CONFID with and without verified binding, REJECT, and unknown result +- **AND** unsafe pre-verification outcomes SHALL skip Verifier + +#### Scenario: Verifier evidence retry paths are tested +- **WHEN** the Workflow suite runs +- **THEN** it SHALL cover Verifier one-time technical retry, retry exhaustion, NON_RETRYABLE_FAILED, REJECT, critical evidence retry, no valid gap, non-critical gap, ceiling-driven LOW_CONFID, and second LOW_CONFID termination +- **AND** evidence retry SHALL occur at most once + +#### Scenario: Composer paths are tested +- **WHEN** the Workflow suite runs +- **THEN** it SHALL cover Composer one-time technical retry, retry exhaustion, NON_RETRYABLE_FAILED, normal completion, and deterministic post-verification Fallback +- **AND** every terminal path SHALL have events matching the executed node sequence + +### Requirement: Node Contract tests SHALL enforce explicit safe state projection + +`DiagnosisGraphNodeContractTest` and focused component tests SHALL verify that Nodes receive only allowed state, use the current RunnableConfig identity, return standardized statuses, and never promote unverified material. + +#### Scenario: Agent and Gatekeeper config is inspected +- **WHEN** Planner, Executor, Verifier, Composer, or Gatekeeper is invoked +- **THEN** the exact current RunnableConfig SHALL be preserved +- **AND** Gatekeeper SHALL validate the current run exactly once per Executor round + +#### Scenario: Executor returns a legal limited result +- **WHEN** tool data is empty or a tool failed but Executor still returns a legal `executor_evidence_v2` with limitations +- **THEN** Executor status SHALL be COMPLETED +- **AND** the workflow SHALL continue to Gatekeeper + +#### Scenario: Gatekeeper returns partial or unsafe evidence +- **WHEN** Gatekeeper is REJECT with any partial passed binding, LOW_CONFID with zero passed binding, missing, or unknown +- **THEN** the path SHALL fail closed to pre-verification Fallback +- **AND** no Executor claim from a passed or failed binding SHALL appear in the answer + +#### Scenario: Verified-only input is projected +- **WHEN** Gatekeeper PASS or continuable LOW_CONFID reaches Verified Input +- **THEN** Verifier SHALL receive only claims/bindings matched to passed checked bindings and their `matched_text` +- **AND** it SHALL NOT receive unreferenced tool results, raw Executor text, or full tool trace + +#### Scenario: Verifier execution fails +- **WHEN** Verifier output is invalid or invocation fails +- **THEN** only `verifier_status` SHALL express the execution failure +- **AND** no model/effective diagnostic verdict SHALL be fabricated + +#### Scenario: Evidence retry revalidates the full snapshot +- **WHEN** one critical evidence retry occurs +- **THEN** the second Executor input SHALL contain prior verified material and incremental-query constraints +- **AND** the second complete Executor snapshot SHALL pass through Gatekeeper again without reusing the first verdict + +### Requirement: Chat integration tests SHALL verify Run-owned public behavior + +`ChatServiceGraphIntegrationTest` SHALL verify the public ChatResult and current DiagnosisRun lifecycle without binding to internal Agent call order. + +#### Scenario: Safe answer completes +- **WHEN** Graph returns a Composer or handled Fallback answer +- **THEN** ChatResult SHALL preserve answer/sessionId/runId +- **AND** the current Run SHALL persist SUCCESS, agent_flow, metrics, self-evaluation, non-empty orchestration trace, and Eval invocation + +#### Scenario: Unsafe completion fails +- **WHEN** Graph throws an unhandled failure, returns no state, or has a blank final answer +- **THEN** the current Run SHALL persist FAILED when possible +- **AND** Eval SHALL NOT run +- **AND** only a real partial trace SHALL be retained + +#### Scenario: Multiple runs share one session +- **WHEN** two complex Chat requests use the same sessionId +- **THEN** each SHALL receive a distinct runId +- **AND** trace/evaluation/metrics SHALL remain owned by their current run + +### Requirement: Legacy implementation tests SHALL retire without losing safety regressions + +The stage 4 suite SHALL remove tests that make Sequential or Hook payload internals correctness criteria while preserving equivalent public and security contracts. + +#### Scenario: Hook implementation test is retired +- **WHEN** explicit Gatekeeper and Verified Input Nodes are authoritative +- **THEN** `VerifierInputHookTest` SHALL NOT exist +- **AND** parser sanitization, Gatekeeper reference fidelity, passed-binding projection, no-evidence, REJECT, and safe Composer behavior SHALL remain covered by independent tests + +#### Scenario: Retained regression suite runs +- **WHEN** stage 4 is accepted +- **THEN** Controller, Trace, Repository, Gatekeeper service, Composer/protocol, Eval, Workflow, Node Contract, and Chat integration tests SHALL pass +- **AND** Maven test compilation SHALL pass + +### Requirement: Stage 4 SHALL remain a test-only change + +The change SHALL reorganize and strengthen automated tests without changing production runtime behavior and SHALL defer final live verification to stage 5. + +#### Scenario: Source diff is inspected +- **WHEN** stage 4 implementation completes +- **THEN** no file under `src/main` SHALL be changed by this stage +- **AND** OpenSpec/devflow/test files MAY change + +#### Scenario: Stage 4 verification completes +- **WHEN** the test suite and static gates pass +- **THEN** Maven live startup, `logs/` inspection, and `scripts/query_mysql.py` database verification SHALL remain not run +- **AND** acceptance SHALL record them as reserved for stage 5 diff --git a/openspec/changes/archive/2026-07-17-chat-diagnosis-stategraph-test-suite/tasks.md b/openspec/changes/archive/2026-07-17-chat-diagnosis-stategraph-test-suite/tasks.md new file mode 100644 index 0000000..abe5e26 --- /dev/null +++ b/openspec/changes/archive/2026-07-17-chat-diagnosis-stategraph-test-suite/tasks.md @@ -0,0 +1,35 @@ +## 1. Authoritative Test Layer Renames + +- [x] 1.1 Rename `DiagnosisGraphRoutingTest` to `DiagnosisGraphWorkflowTest` with Git history preserved and no assertion loss. +- [x] 1.2 Rename `DiagnosisRealGraphIntegrationTest` to `DiagnosisGraphNodeContractTest` with existing real-node assertions preserved. +- [x] 1.3 Run the two renamed classes alone and prove the rename baseline is green before adding coverage. + +## 2. Workflow Coverage Matrix + +- [x] 2.1 Audit Workflow methods against every stage-4 Planner, Executor, Gatekeeper, Verifier, evidence-retry, Composer, and termination scenario. +- [x] 2.2 Add explicit ceiling-driven LOW_CONFID and second-LOW_CONFID no-more-retry assertions where current routing coverage is only implicit. +- [x] 2.3 Assert all unsafe pre-verification paths skip Verifier and every terminal path emits only its actual ordered events. +- [x] 2.4 Keep `ScriptedDiagnosisGraphActions` as the only scripted workflow fixture and avoid duplicate Graph topology/factory setup. + +## 3. Node Contract Coverage Matrix + +- [x] 3.1 Add a legal tool-failure/no-evidence Executor snapshot case proving COMPLETED continues to Gatekeeper. +- [x] 3.2 Strengthen Gatekeeper REJECT with partial passed binding and prove pre-verification Fallback exposes no Executor claim. +- [x] 3.3 Add Verifier invalid/failure contract assertions proving execution status is set without model/effective verdict fabrication. +- [x] 3.4 Preserve current config identity, checked-binding projection, LOW_CONFID ceiling, full-snapshot revalidation, and Composer safe-material assertions. +- [x] 3.5 Run all focused Adapter, Gatekeeper, VerifiedInput, protocol parser, retry preparer, and Fallback tests. + +## 4. Chat Integration And Legacy Test Retirement + +- [x] 4.1 Confirm `ChatServiceGraphIntegrationTest` covers compatible ChatResult, SUCCESS/Fallback/FAILED, metrics/Eval, self-evaluation/trace, cleanup, and multi-run isolation. +- [x] 4.2 Delete `VerifierInputHookTest` without changing production Hook/ThreadLocal types in stage 4. +- [x] 4.3 Run replacement parser, Gatekeeper service/node, VerifiedInput, no-evidence, REJECT, Composer safety, Controller, Trace, Repository, and Eval regressions. +- [x] 4.4 Add a source/test inventory check proving the three authoritative classes exist and Sequential/Hook implementation tests do not. + +## 5. Verification And Handoff + +- [x] 5.1 Run the complete new Graph Workflow/Node Contract/Chat Integration suite and record exact test counts. +- [x] 5.2 Run Maven test compilation and all retained public/security contract tests touched by the migration. +- [x] 5.3 Run current change strict validation, all main specs strict validation, `git diff --check`, and prove `src/main` has no stage-4 diff. +- [x] 5.4 Record the Issue requirement-to-test coverage matrix and any intentional overlap/known limits in devflow acceptance evidence. +- [x] 5.5 Record that Maven live E2E, `logs/`, and `scripts/query_mysql.py` verification were not run and remain reserved for stage 5. diff --git a/openspec/specs/chat-diagnosis-stategraph-design-freeze/spec.md b/openspec/specs/chat-diagnosis-stategraph-design-freeze/spec.md index 8ee7c47..1fbc9ed 100644 --- a/openspec/specs/chat-diagnosis-stategraph-design-freeze/spec.md +++ b/openspec/specs/chat-diagnosis-stategraph-design-freeze/spec.md @@ -109,16 +109,24 @@ Orchestration audit SHALL remain separate from self evaluation, AgentStep, ToolI ### Requirement: Test migration design SHALL preserve safety behavior -The baseline SHALL identify Sequential/Hook implementation tests to replace and public/security contract tests to retain or extend. Fixed Agent call order SHALL NOT remain a correctness criterion. +The project SHALL replace Sequential/Hook implementation tests with authoritative `DiagnosisGraphWorkflowTest`, `DiagnosisGraphNodeContractTest`, and `ChatServiceGraphIntegrationTest` layers while retaining focused public/security contract tests. Fixed Agent call order and legacy Hook payload shape SHALL NOT remain correctness criteria. #### Scenario: Old tests are replaced - **WHEN** StateGraph tests become authoritative -- **THEN** `ChatServiceSequentialAgentTest` SHALL be replaced by route, node-contract, and Chat integration coverage -- **AND** `VerifierInputHookTest` SHALL be removed or rewritten for explicit nodes +- **THEN** `ChatServiceSequentialAgentTest` SHALL NOT exist +- **AND** `VerifierInputHookTest` SHALL NOT exist after explicit Gatekeeper/Verified Input coverage is established +- **AND** no replacement test SHALL restore fixed Sequential Agent order as a public requirement #### Scenario: Safety tests are retained - **WHEN** the new suite is assembled -- **THEN** Gatekeeper, Controller, Run/Trace, Repository, Composer, no-evidence, REJECT, and Eval safety contracts SHALL remain covered -- **AND** fixed-order-only assertions SHALL be removed +- **THEN** Gatekeeper, Controller, Run/Trace, Repository, Composer, parser/projection, no-evidence, REJECT, Eval, and multi-run safety contracts SHALL remain covered +- **AND** Workflow and Node Contract matrices SHALL cover bounded retry, fail-closed routing, verified-only inputs, and deterministic Fallback + +#### Scenario: Authoritative test layers are inspected + +- **WHEN** a maintainer needs to locate ISS-011 orchestration verification +- **THEN** Workflow tests SHALL own route/loop/event behavior +- **AND** Node Contract tests SHALL own input/output/security projection behavior +- **AND** Chat integration tests SHALL own Run lifecycle and public result behavior diff --git a/openspec/specs/chat-diagnosis-stategraph-test-suite/spec.md b/openspec/specs/chat-diagnosis-stategraph-test-suite/spec.md new file mode 100644 index 0000000..0bf46ad --- /dev/null +++ b/openspec/specs/chat-diagnosis-stategraph-test-suite/spec.md @@ -0,0 +1,124 @@ +# chat-diagnosis-stategraph-test-suite Specification + +## Purpose +TBD - created by archiving change chat-diagnosis-stategraph-test-suite. Update Purpose after archive. +## Requirements +### Requirement: Diagnosis Graph tests SHALL have three authoritative layers + +The test suite SHALL use Workflow, Node Contract, and ChatService Integration layers as the authoritative ISS-011 verification structure while retaining focused component tests for failure localization. + +#### Scenario: Test architecture is inspected +- **WHEN** the Diagnosis Graph test sources are listed +- **THEN** `DiagnosisGraphWorkflowTest`, `DiagnosisGraphNodeContractTest`, and `ChatServiceGraphIntegrationTest` SHALL exist +- **AND** no `ChatServiceSequentialAgentTest` SHALL exist + +#### Scenario: Focused unit tests are retained +- **WHEN** a single parser, adapter, Gatekeeper, projection, retry preparer, trace builder, or Fallback contract fails +- **THEN** a focused component test SHALL be able to identify that boundary +- **AND** the suite SHALL NOT require all component assertions to be duplicated in the three authoritative classes + +### Requirement: Workflow tests SHALL cover every bounded routing class + +`DiagnosisGraphWorkflowTest` SHALL execute the real compiled topology with scripted nodes and SHALL verify path order, retry ownership, bounded termination, and orchestration events without invoking models or tools. + +#### Scenario: Normal and Planner paths are tested +- **WHEN** the Workflow suite runs +- **THEN** it SHALL cover PASS, Planner INVALID_OUTPUT/RETRYABLE_FAILED one-time retry success, second technical failure, NON_RETRYABLE_FAILED, and fail-closed unknown status +- **AND** Planner terminal failures SHALL skip Executor and reach Fallback + +#### Scenario: Executor and Gatekeeper paths are tested +- **WHEN** the Workflow suite runs +- **THEN** it SHALL cover Executor FAILED, TOOL_BLOCKED, INVALID_OUTPUT, legal no-evidence, Gatekeeper PASS, LOW_CONFID with and without verified binding, REJECT, and unknown result +- **AND** unsafe pre-verification outcomes SHALL skip Verifier + +#### Scenario: Verifier evidence retry paths are tested +- **WHEN** the Workflow suite runs +- **THEN** it SHALL cover Verifier one-time technical retry, retry exhaustion, NON_RETRYABLE_FAILED, REJECT, critical evidence retry, no valid gap, non-critical gap, ceiling-driven LOW_CONFID, and second LOW_CONFID termination +- **AND** evidence retry SHALL occur at most once + +#### Scenario: Composer paths are tested +- **WHEN** the Workflow suite runs +- **THEN** it SHALL cover Composer one-time technical retry, retry exhaustion, NON_RETRYABLE_FAILED, normal completion, and deterministic post-verification Fallback +- **AND** every terminal path SHALL have events matching the executed node sequence + +### Requirement: Node Contract tests SHALL enforce explicit safe state projection + +`DiagnosisGraphNodeContractTest` and focused component tests SHALL verify that Nodes receive only allowed state, use the current RunnableConfig identity, return standardized statuses, and never promote unverified material. + +#### Scenario: Agent and Gatekeeper config is inspected +- **WHEN** Planner, Executor, Verifier, Composer, or Gatekeeper is invoked +- **THEN** the exact current RunnableConfig SHALL be preserved +- **AND** Gatekeeper SHALL validate the current run exactly once per Executor round + +#### Scenario: Executor returns a legal limited result +- **WHEN** tool data is empty or a tool failed but Executor still returns a legal `executor_evidence_v2` with limitations +- **THEN** Executor status SHALL be COMPLETED +- **AND** the workflow SHALL continue to Gatekeeper + +#### Scenario: Gatekeeper returns partial or unsafe evidence +- **WHEN** Gatekeeper is REJECT with any partial passed binding, LOW_CONFID with zero passed binding, missing, or unknown +- **THEN** the path SHALL fail closed to pre-verification Fallback +- **AND** no Executor claim from a passed or failed binding SHALL appear in the answer + +#### Scenario: Verified-only input is projected +- **WHEN** Gatekeeper PASS or continuable LOW_CONFID reaches Verified Input +- **THEN** Verifier SHALL receive only claims/bindings matched to passed checked bindings and their `matched_text` +- **AND** it SHALL NOT receive unreferenced tool results, raw Executor text, or full tool trace + +#### Scenario: Verifier execution fails +- **WHEN** Verifier output is invalid or invocation fails +- **THEN** only `verifier_status` SHALL express the execution failure +- **AND** no model/effective diagnostic verdict SHALL be fabricated + +#### Scenario: Evidence retry revalidates the full snapshot +- **WHEN** one critical evidence retry occurs +- **THEN** the second Executor input SHALL contain prior verified material and incremental-query constraints +- **AND** the second complete Executor snapshot SHALL pass through Gatekeeper again without reusing the first verdict + +### Requirement: Chat integration tests SHALL verify Run-owned public behavior + +`ChatServiceGraphIntegrationTest` SHALL verify the public ChatResult and current DiagnosisRun lifecycle without binding to internal Agent call order. + +#### Scenario: Safe answer completes +- **WHEN** Graph returns a Composer or handled Fallback answer +- **THEN** ChatResult SHALL preserve answer/sessionId/runId +- **AND** the current Run SHALL persist SUCCESS, agent_flow, metrics, self-evaluation, non-empty orchestration trace, and Eval invocation + +#### Scenario: Unsafe completion fails +- **WHEN** Graph throws an unhandled failure, returns no state, or has a blank final answer +- **THEN** the current Run SHALL persist FAILED when possible +- **AND** Eval SHALL NOT run +- **AND** only a real partial trace SHALL be retained + +#### Scenario: Multiple runs share one session +- **WHEN** two complex Chat requests use the same sessionId +- **THEN** each SHALL receive a distinct runId +- **AND** trace/evaluation/metrics SHALL remain owned by their current run + +### Requirement: Legacy implementation tests SHALL retire without losing safety regressions + +The stage 4 suite SHALL remove tests that make Sequential or Hook payload internals correctness criteria while preserving equivalent public and security contracts. + +#### Scenario: Hook implementation test is retired +- **WHEN** explicit Gatekeeper and Verified Input Nodes are authoritative +- **THEN** `VerifierInputHookTest` SHALL NOT exist +- **AND** parser sanitization, Gatekeeper reference fidelity, passed-binding projection, no-evidence, REJECT, and safe Composer behavior SHALL remain covered by independent tests + +#### Scenario: Retained regression suite runs +- **WHEN** stage 4 is accepted +- **THEN** Controller, Trace, Repository, Gatekeeper service, Composer/protocol, Eval, Workflow, Node Contract, and Chat integration tests SHALL pass +- **AND** Maven test compilation SHALL pass + +### Requirement: Stage 4 SHALL remain a test-only change + +The change SHALL reorganize and strengthen automated tests without changing production runtime behavior and SHALL defer final live verification to stage 5. + +#### Scenario: Source diff is inspected +- **WHEN** stage 4 implementation completes +- **THEN** no file under `src/main` SHALL be changed by this stage +- **AND** OpenSpec/devflow/test files MAY change + +#### Scenario: Stage 4 verification completes +- **WHEN** the test suite and static gates pass +- **THEN** Maven live startup, `logs/` inspection, and `scripts/query_mysql.py` database verification SHALL remain not run +- **AND** acceptance SHALL record them as reserved for stage 5 diff --git a/src/test/java/com/superbiz/agent/graph/diagnosis/DiagnosisRealGraphIntegrationTest.java b/src/test/java/com/superbiz/agent/graph/diagnosis/DiagnosisGraphNodeContractTest.java similarity index 82% rename from src/test/java/com/superbiz/agent/graph/diagnosis/DiagnosisRealGraphIntegrationTest.java rename to src/test/java/com/superbiz/agent/graph/diagnosis/DiagnosisGraphNodeContractTest.java index 37c9bbb..02995d5 100644 --- a/src/test/java/com/superbiz/agent/graph/diagnosis/DiagnosisRealGraphIntegrationTest.java +++ b/src/test/java/com/superbiz/agent/graph/diagnosis/DiagnosisGraphNodeContractTest.java @@ -14,6 +14,7 @@ import java.util.Deque; import static org.junit.jupiter.api.Assertions.assertEquals; import static org.junit.jupiter.api.Assertions.assertFalse; +import static org.junit.jupiter.api.Assertions.assertTrue; import static org.mockito.ArgumentMatchers.anyMap; import static org.mockito.ArgumentMatchers.eq; import static org.mockito.Mockito.mock; @@ -21,7 +22,44 @@ import static org.mockito.Mockito.times; import static org.mockito.Mockito.verify; import static org.mockito.Mockito.when; -class DiagnosisRealGraphIntegrationTest { +class DiagnosisGraphNodeContractTest { + + @Test + void legalSnapshotAfterToolFailureRemainsCompletedAndReachesGatekeeper() + throws Exception { + ExecutorGatekeeperService gatekeeper = mock(ExecutorGatekeeperService.class); + when(gatekeeper.validateRun(eq("run-tool-limited"), anyMap(), anyMap())) + .thenReturn(Map.of( + "status", "pass", + "severity", "none", + "checked_bindings", List.of(), + "failed_rules", List.of())); + String limitedOutput = """ + {"answer_version":"executor_evidence_v2","claims":[], + "hypotheses":[],"recommended_actions":[], + "missing_info":["query_logs failed: timeout; no evidence available"]} + """; + String verifierOutput = """ + {"verdict":"LOW_CONFID","groundedness_score":0.0, + "critical_fact_count":0,"claim_checks":[],"facts_checked":[], + "rationale":"工具失败已被合法表达为证据限制"} + """; + DiagnosisGraphActions actions = new DiagnosisRealGraphActionsFactory().create( + constant(plannerOutput()), constant(limitedOutput), + constant(verifierOutput), constant(composerOutput()), gatekeeper); + + OverAllState state = new DiagnosisGraphFactory().compile(actions) + .invoke(initialState(), config("run-tool-limited")) + .orElseThrow(); + + assertEquals("COMPLETED", DiagnosisGraphState.stringValue( + state, DiagnosisGraphState.EXECUTOR_STATUS)); + assertEquals(List.of("planner", "executor", "gatekeeper", "verified_input", + "verifier", "composer"), + eventNodes(state)); + verify(gatekeeper, times(1)).validateRun( + eq("run-tool-limited"), anyMap(), anyMap()); + } @Test void legalNoEvidenceWithVerifiedBindingContinuesUnderLowConfidenceCeiling() @@ -128,7 +166,20 @@ class DiagnosisRealGraphIntegrationTest { .thenReturn(Map.of( "status", "fail", "severity", "reject", - "checked_bindings", List.of(), + "checked_bindings", List.of( + Map.of( + "claim_id", "c1", + "status", "pass", + "tool_name", "query_metrics", + "source_invocation_id", 17L, + "raw_path", "$.alerts[0]", + "matched_text", "active=50 max=50"), + Map.of( + "claim_id", "c2", + "status", "fail", + "tool_name", "query_logs", + "source_invocation_id", 18L, + "raw_path", "$.results[0]")), "failed_rules", List.of("evidence.raw_path"))); DiagnosisGraphActions actions = new DiagnosisRealGraphActionsFactory().create( constant(plannerOutput()), @@ -148,6 +199,33 @@ class DiagnosisRealGraphIntegrationTest { eventNodes(state)); } + @Test + void verifierInvalidOutputSetsExecutionStatusWithoutFabricatedVerdict() + throws Exception { + ExecutorGatekeeperService gatekeeper = mock(ExecutorGatekeeperService.class); + when(gatekeeper.validateRun(eq("run-verifier-invalid"), anyMap(), anyMap())) + .thenReturn(passResult()); + QueueInvoker verifier = new QueueInvoker("not-json", "still-not-json"); + DiagnosisGraphActions actions = new DiagnosisRealGraphActionsFactory().create( + constant(plannerOutput()), constant(executorOutput()), verifier, + constant(composerOutput()), gatekeeper); + + OverAllState state = new DiagnosisGraphFactory().compile(actions) + .invoke(initialState(), config("run-verifier-invalid")) + .orElseThrow(); + + assertEquals("INVALID_OUTPUT", DiagnosisGraphState.stringValue( + state, DiagnosisGraphState.VERIFIER_STATUS)); + assertTrue(state.value(DiagnosisGraphState.VERIFIER_MODEL_VERDICT).isEmpty()); + assertTrue(state.value(DiagnosisGraphState.EFFECTIVE_VERDICT).isEmpty()); + assertFalse(DiagnosisGraphState.stringValue( + state, DiagnosisGraphState.FINAL_ANSWER) + .contains("连接数达到上限")); + assertEquals(List.of("planner", "executor", "gatekeeper", "verified_input", + "verifier", "verifier", "fallback"), + eventNodes(state)); + } + @Test void exhaustedComposerRetryUsesVerifiedMaterialWithoutRerunningPredecessors() throws Exception { diff --git a/src/test/java/com/superbiz/agent/graph/diagnosis/DiagnosisGraphTestSuiteStructureTest.java b/src/test/java/com/superbiz/agent/graph/diagnosis/DiagnosisGraphTestSuiteStructureTest.java new file mode 100644 index 0000000..18856f7 --- /dev/null +++ b/src/test/java/com/superbiz/agent/graph/diagnosis/DiagnosisGraphTestSuiteStructureTest.java @@ -0,0 +1,44 @@ +package com.superbiz.agent.graph.diagnosis; + +import org.junit.jupiter.api.Test; + +import java.io.IOException; +import java.nio.file.Files; +import java.nio.file.Path; +import java.util.List; + +import static org.junit.jupiter.api.Assertions.assertFalse; +import static org.junit.jupiter.api.Assertions.assertTrue; + +class DiagnosisGraphTestSuiteStructureTest { + + private static final Path TEST_ROOT = Path.of( + "src", "test", "java", "com", "superbiz", "agent"); + + @Test + void authoritativeGraphLayersExistWithoutSequentialOrHookImplementationTests() + throws IOException { + List authoritativeTests = List.of( + graphTest("DiagnosisGraphWorkflowTest.java"), + graphTest("DiagnosisGraphNodeContractTest.java"), + TEST_ROOT.resolve(Path.of( + "service", "ChatServiceGraphIntegrationTest.java"))); + + authoritativeTests.forEach(path -> assertTrue( + Files.isRegularFile(path), "missing authoritative test: " + path)); + assertFalse(Files.exists(TEST_ROOT.resolve(Path.of( + "service", "ChatServiceSequentialAgentTest.java")))); + assertFalse(Files.exists(TEST_ROOT.resolve(Path.of( + "hook", "VerifierInputHookTest.java")))); + + for (Path path : authoritativeTests) { + String source = Files.readString(path); + assertFalse(source.contains("SequentialAgent"), path.toString()); + assertFalse(source.contains("VerifierInputHook"), path.toString()); + } + } + + private Path graphTest(String fileName) { + return TEST_ROOT.resolve(Path.of("graph", "diagnosis", fileName)); + } +} diff --git a/src/test/java/com/superbiz/agent/graph/diagnosis/DiagnosisGraphRoutingTest.java b/src/test/java/com/superbiz/agent/graph/diagnosis/DiagnosisGraphWorkflowTest.java similarity index 93% rename from src/test/java/com/superbiz/agent/graph/diagnosis/DiagnosisGraphRoutingTest.java rename to src/test/java/com/superbiz/agent/graph/diagnosis/DiagnosisGraphWorkflowTest.java index cdd8c2a..3e90712 100644 --- a/src/test/java/com/superbiz/agent/graph/diagnosis/DiagnosisGraphRoutingTest.java +++ b/src/test/java/com/superbiz/agent/graph/diagnosis/DiagnosisGraphWorkflowTest.java @@ -22,7 +22,7 @@ import java.util.stream.Stream; import static org.junit.jupiter.api.Assertions.assertEquals; import static org.junit.jupiter.api.Assertions.assertFalse; -class DiagnosisGraphRoutingTest { +class DiagnosisGraphWorkflowTest { private static final String THREAD_ID = "run-routing-test"; @@ -193,15 +193,15 @@ class DiagnosisGraphRoutingTest { } @Test - void gatekeeperLowConfidenceWithVerifiedBindingRunsVerifiedInput() + void gatekeeperCeilingLowConfidenceWithVerifiedBindingDoesNotRetryEvidence() throws Exception { ScriptedDiagnosisGraphActions script = new ScriptedDiagnosisGraphActions(); planner(script, PlannerStatus.COMPLETED); executor(script, ExecutorStatus.COMPLETED); gatekeeper(script, GatekeeperStatus.LOW_CONFID, 1); verifiedInput(script); - verifier(script, VerifierStatus.COMPLETED, Verdict.LOW_CONFID, - Verdict.LOW_CONFID, Map.of()); + verifier(script, VerifierStatus.COMPLETED, Verdict.PASS, + Verdict.LOW_CONFID, Verdict.LOW_CONFID, Map.of()); composerCompleted(script); OverAllState state = run(script); @@ -209,6 +209,12 @@ class DiagnosisGraphRoutingTest { assertEquals(1, script.calls(DiagnosisGraphTopology.Node.VERIFIED_INPUT)); assertEquals(1, script.calls(DiagnosisGraphTopology.Node.VERIFIER)); assertEquals(0, script.calls(DiagnosisGraphTopology.Node.EVIDENCE_RETRY)); + assertEquals(Verdict.PASS, + DiagnosisGraphState.enumValue( + state, + DiagnosisGraphState.VERIFIER_MODEL_VERDICT, + Verdict.class) + .orElseThrow()); assertEquals(Verdict.LOW_CONFID, DiagnosisGraphState.enumValue( state, @@ -272,7 +278,7 @@ class DiagnosisGraphRoutingTest { } @Test - void evidenceRetryRunsOnceAndResetsPlannerStageCounter() + void secondLowConfidenceDoesNotRetryEvidenceAgainAndResetsPlannerCounter() throws Exception { ScriptedDiagnosisGraphActions script = new ScriptedDiagnosisGraphActions(); planner(script, PlannerStatus.INVALID_OUTPUT); @@ -294,6 +300,7 @@ class DiagnosisGraphRoutingTest { assertEquals(4, script.calls(DiagnosisGraphTopology.Node.PLANNER)); assertEquals(2, script.calls(DiagnosisGraphTopology.Node.EXECUTOR)); assertEquals(1, script.calls(DiagnosisGraphTopology.Node.EVIDENCE_RETRY)); + assertEquals(1, script.calls(DiagnosisGraphTopology.Node.COMPOSER)); assertEquals(1, DiagnosisGraphState.intValue( state, DiagnosisGraphState.EVIDENCE_RETRY_COUNT)); assertEquals(1, DiagnosisGraphState.intValue( @@ -436,10 +443,16 @@ class DiagnosisGraphRoutingTest { CompiledGraph graph = new DiagnosisGraphFactory().compile(script.actions()); assertEquals(DiagnosisGraphTopology.RECURSION_LIMIT, graph.getMaxIterations()); - return graph.invoke( + OverAllState state = graph.invoke( initialState(), RunnableConfig.builder().threadId(THREAD_ID).build()) .orElseThrow(); + assertEquals(script.sequence(), DiagnosisGraphState.listValue( + state, DiagnosisGraphState.ORCHESTRATION_EVENTS) + .stream() + .map(event -> ((OrchestrationEvent) event).node()) + .toList()); + return state; } private Map initialState() { @@ -548,10 +561,25 @@ class DiagnosisGraphRoutingTest { Verdict verdict, Verdict ceiling, Map verifierOutput) { + verifier(script, status, verdict, verdict, ceiling, verifierOutput); + } + + private void verifier( + ScriptedDiagnosisGraphActions script, + VerifierStatus status, + Verdict modelVerdict, + Verdict effectiveVerdict, + Verdict ceiling, + Map verifierOutput) { Map update = new java.util.LinkedHashMap<>(); update.put(DiagnosisGraphState.VERIFIER_STATUS, status.name()); - if (verdict != null) { - update.put(DiagnosisGraphState.EFFECTIVE_VERDICT, verdict.name()); + if (modelVerdict != null) { + update.put(DiagnosisGraphState.VERIFIER_MODEL_VERDICT, + modelVerdict.name()); + } + if (effectiveVerdict != null) { + update.put(DiagnosisGraphState.EFFECTIVE_VERDICT, + effectiveVerdict.name()); } if (ceiling != null) { update.put(DiagnosisGraphState.VERIFIER_VERDICT_CEILING, diff --git a/src/test/java/com/superbiz/agent/hook/VerifierInputHookTest.java b/src/test/java/com/superbiz/agent/hook/VerifierInputHookTest.java deleted file mode 100644 index 3100f0c..0000000 --- a/src/test/java/com/superbiz/agent/hook/VerifierInputHookTest.java +++ /dev/null @@ -1,380 +0,0 @@ -package com.superbiz.agent.hook; - -import com.alibaba.cloud.ai.graph.RunnableConfig; -import com.alibaba.cloud.ai.graph.agent.hook.messages.AgentCommand; -import com.fasterxml.jackson.databind.JsonNode; -import com.fasterxml.jackson.databind.ObjectMapper; -import com.superbiz.agent.domain.entity.ToolInvocation; -import com.superbiz.agent.repository.ToolInvocationRepository; -import com.superbiz.agent.service.ExecutorGatekeeperService; -import com.superbiz.agent.service.ToolTraceSummaryService; -import com.superbiz.agent.util.VerifierContextHolder; -import org.junit.jupiter.api.AfterEach; -import org.junit.jupiter.api.Test; -import org.springframework.ai.chat.messages.AssistantMessage; -import org.springframework.ai.chat.messages.Message; -import org.springframework.ai.chat.messages.UserMessage; - -import java.util.List; -import java.util.Map; - -import static org.junit.jupiter.api.Assertions.assertEquals; -import static org.junit.jupiter.api.Assertions.assertFalse; -import static org.junit.jupiter.api.Assertions.assertNotNull; -import static org.junit.jupiter.api.Assertions.assertTrue; -import static org.mockito.ArgumentMatchers.anyString; -import static org.mockito.Mockito.mock; -import static org.mockito.Mockito.when; - -class VerifierInputHookTest { - - private final ObjectMapper objectMapper = new ObjectMapper(); - - @AfterEach - void tearDown() { - VerifierContextHolder.clear(); - } - - @Test - void beforeModelAddsStructuredExecutorOutputWhenJsonContractIsValid() throws Exception { - ToolTraceSummaryService traceSummaryService = mock(ToolTraceSummaryService.class); - when(traceSummaryService.buildVerifierTraceSummary(anyString(), anyString())).thenReturn(List.of( - Map.of("trace_ref", "trace-1", "tool_name", "query_metrics") - )); - VerifierInputHook hook = new VerifierInputHook(traceSummaryService); - VerifierContextHolder.setOriginalQuery("分析 MySQL 连接池耗尽"); - - String executorOutput = """ - { - "answer_version": "executor_evidence_v1", - "diagnosis_summary": "连接池已满,但缺少泄漏证据。", - "claims": [ - { - "claim_id": "claim-1", - "claim_type": "symptom", - "claim_text": "连接池 active 达到上限", - "support_level": "direct", - "evidence_bindings": [ - { - "source_type": "tool_trace", - "source_id": "trace-1", - "tool_name": "query_metrics", - "source_invocation_ids": [101], - "evidence_excerpt": "active=50 max=50" - } - ] - } - ], - "hypotheses": [], - "recommended_actions": [], - "missing_info": ["缺少泄漏检测日志"], - "user_facing_answer": "已确认连接池 active 达到上限。" - } - """; - - AgentCommand command = hook.beforeModel( - List.of(new AssistantMessage(executorOutput)), - RunnableConfig.builder().addMetadata("sessionId", "structured-session").build() - ); - - JsonNode payload = readPayload(command); - assertEquals("valid", payload.path("executor_output_parse_status").path("status").asText()); - assertEquals("executor_evidence_v1", - payload.path("executor_structured_output").path("answer_version").asText()); - assertEquals("连接池 active 达到上限", - payload.path("executor_structured_output").path("claims").get(0).path("claim_text").asText()); - assertNotNull(VerifierContextHolder.getExecutorStructuredOutput()); - assertEquals("valid", VerifierContextHolder.getExecutorOutputParseStatus().get("status")); - } - - @Test - void beforeModelAddsStructuredExecutorOutputWhenV2ContractHasNoUserFacingAnswer() throws Exception { - ToolTraceSummaryService traceSummaryService = mock(ToolTraceSummaryService.class); - when(traceSummaryService.buildVerifierTraceSummary(anyString(), anyString())).thenReturn(List.of( - Map.of("trace_ref", "trace-1", "tool_name", "query_metrics") - )); - ToolInvocationRepository invocationRepository = mock(ToolInvocationRepository.class); - when(invocationRepository.findBySessionIdOrderByIdAsc("structured-v2-session")).thenReturn(List.of( - invocation(101L, "structured-v2-session", "query_metrics", "$.alerts[0]", "active=50 max=50") - )); - VerifierInputHook hook = new VerifierInputHook(traceSummaryService, - new ExecutorGatekeeperService(invocationRepository)); - VerifierContextHolder.setOriginalQuery("分析 MySQL 连接池耗尽"); - - String executorOutput = """ - { - "answer_version": "executor_evidence_v2", - "claims": [ - { - "claim_id": "claim-1", - "claim_type": "symptom", - "claim_text": "连接池 active 达到上限", - "support_level": "direct", - "evidence_bindings": [ - { - "source_type": "tool_trace", - "source_id": "trace-1", - "tool_name": "query_metrics", - "source_invocation_id": 101, - "raw_path": "$.alerts[0]", - "evidence_excerpt": "active=50 max=50" - } - ] - } - ], - "hypotheses": [], - "recommended_actions": [], - "missing_info": [] - } - """; - - AgentCommand command = hook.beforeModel( - List.of(new AssistantMessage(executorOutput)), - RunnableConfig.builder().addMetadata("sessionId", "structured-v2-session").build() - ); - - JsonNode payload = readPayload(command); - assertEquals("valid", payload.path("executor_output_parse_status").path("status").asText()); - assertEquals("executor_evidence_v2", - payload.path("executor_structured_output").path("answer_version").asText()); - assertFalse(payload.path("executor_structured_output").has("user_facing_answer")); - assertEquals("连接池 active 达到上限", - payload.path("executor_structured_output").path("claims").get(0).path("claim_text").asText()); - assertEquals("pass", payload.path("gatekeeper_result").path("status").asText()); - assertEquals("none", payload.path("gatekeeper_result").path("severity").asText()); - assertEquals("pass", VerifierContextHolder.getGatekeeperResult().get("status")); - } - - @Test - void beforeModelBackfillsOnlyUniqueInvocationIdAndDoesNotPassWithoutRawPath() throws Exception { - ToolTraceSummaryService traceSummaryService = mock(ToolTraceSummaryService.class); - when(traceSummaryService.buildVerifierTraceSummary(anyString(), anyString())).thenReturn(List.of( - Map.of( - "trace_ref", "metrics-1", - "tool_name", "query_metrics", - "source_invocation_ids", List.of(101L) - ) - )); - ToolInvocationRepository invocationRepository = mock(ToolInvocationRepository.class); - when(invocationRepository.findBySessionIdOrderByIdAsc("backfill-session")).thenReturn(List.of( - invocation(101L, "backfill-session", "query_metrics", "$.alerts[0]", - "CPU 使用率持续超过 80%,当前值为 92%") - )); - VerifierInputHook hook = new VerifierInputHook(traceSummaryService, - new ExecutorGatekeeperService(invocationRepository)); - - String executorOutput = """ - { - "answer_version": "executor_evidence_v2", - "claims": [ - { - "claim_id": "claim-1", - "claim_type": "symptom", - "claim_text": "payment-service CPU 使用率超过 92%", - "support_level": "direct", - "evidence_bindings": [ - { - "source_type": "tool_trace", - "source_id": "prometheus-alert-HighCPUUsage", - "tool_name": "queryPrometheusAlerts", - "evidence_excerpt": "CPU 使用率持续超过 80%,当前值为 92%" - } - ] - } - ], - "hypotheses": [], - "recommended_actions": [ - { - "action_text": "restart payment-service", - "reason": "cpu alert is firing", - "evidence_bindings": [ - { - "source_type": "tool_trace", - "source_id": "prometheus-alert-HighCPUUsage", - "tool_name": "queryPrometheusAlerts", - "evidence_excerpt": "CPU usage is 92%" - } - ] - } - ], - "missing_info": [] - } - """; - - AgentCommand command = hook.beforeModel( - List.of(new AssistantMessage(executorOutput)), - RunnableConfig.builder().addMetadata("sessionId", "backfill-session").build() - ); - - JsonNode payload = readPayload(command); - JsonNode binding = payload.path("executor_structured_output") - .path("claims").get(0) - .path("evidence_bindings").get(0); - assertEquals("query_metrics", binding.path("tool_name").asText()); - assertEquals(101L, binding.path("source_invocation_id").asLong()); - JsonNode actionBinding = payload.path("executor_structured_output") - .path("recommended_actions").get(0) - .path("evidence_bindings").get(0); - assertEquals("query_metrics", actionBinding.path("tool_name").asText()); - assertEquals(101L, actionBinding.path("source_invocation_id").asLong()); - assertFalse(binding.has("raw_path")); - assertEquals("fail", payload.path("gatekeeper_result").path("status").asText()); - assertEquals("low_confid", payload.path("gatekeeper_result").path("severity").asText()); - assertEquals("evidence.invocation_auto_backfill", - payload.path("gatekeeper_result").path("warnings").get(0).path("rule").asText()); - } - - @Test - void beforeModelAddsFailingGatekeeperResultForFabricatedInvocationId() throws Exception { - ToolTraceSummaryService traceSummaryService = mock(ToolTraceSummaryService.class); - when(traceSummaryService.buildVerifierTraceSummary(anyString(), anyString())).thenReturn(List.of()); - ToolInvocationRepository invocationRepository = mock(ToolInvocationRepository.class); - when(invocationRepository.findBySessionIdOrderByIdAsc("fabricated-invocation-session")).thenReturn(List.of( - invocation(101L, "fabricated-invocation-session", "query_metrics", "$.alerts[0]", "active=50 max=50") - )); - VerifierInputHook hook = new VerifierInputHook(traceSummaryService, - new ExecutorGatekeeperService(invocationRepository)); - - String executorOutput = """ - { - "answer_version": "executor_evidence_v2", - "claims": [ - { - "claim_id": "claim-1", - "claim_type": "symptom", - "claim_text": "连接池 active 达到上限", - "support_level": "direct", - "evidence_bindings": [ - { - "source_type": "tool_trace", - "tool_name": "query_metrics", - "source_invocation_id": 999, - "raw_path": "$.alerts[0]", - "evidence_excerpt": "active=50 max=50" - } - ] - } - ], - "hypotheses": [], - "recommended_actions": [], - "missing_info": [] - } - """; - - AgentCommand command = hook.beforeModel( - List.of(new AssistantMessage(executorOutput)), - RunnableConfig.builder().addMetadata("sessionId", "fabricated-invocation-session").build() - ); - - JsonNode payload = readPayload(command); - assertEquals("fail", payload.path("gatekeeper_result").path("status").asText()); - assertEquals("reject", payload.path("gatekeeper_result").path("severity").asText()); - assertEquals("evidence.invocation_ref", - payload.path("gatekeeper_result").path("failed_rules").get(0).asText()); - } - - @Test - void beforeModelExtractsStructuredOutputFromPrefixedJsonFence() throws Exception { - ToolTraceSummaryService traceSummaryService = mock(ToolTraceSummaryService.class); - when(traceSummaryService.buildVerifierTraceSummary(anyString(), anyString())).thenReturn(List.of()); - VerifierInputHook hook = new VerifierInputHook(traceSummaryService); - - String executorOutput = """ - 现在我已经收集了足够的数据,最终输出如下。 - - ```json - { - "answer_version": "executor_evidence_v1", - "diagnosis_summary": "已确认连接池 active 达到上限。", - "claims": [ - { - "claim_id": "claim-1", - "claim_type": "symptom", - "claim_text": "连接池 active 达到上限", - "support_level": "direct", - "evidence_bindings": [ - { - "source_type": "tool_trace", - "source_id": "trace-1", - "tool_name": "query_metrics", - "source_invocation_ids": [101], - "evidence_excerpt": "active=50 max=50" - } - ] - } - ], - "hypotheses": [], - "recommended_actions": [], - "missing_info": [], - "user_facing_answer": "已确认连接池 active 达到上限。" - } - ``` - """; - - AgentCommand command = hook.beforeModel( - List.of(new AssistantMessage(executorOutput)), - RunnableConfig.builder().addMetadata("sessionId", "fenced-session").build() - ); - - JsonNode payload = readPayload(command); - assertEquals("valid", payload.path("executor_output_parse_status").path("status").asText()); - assertEquals("claim-1", - payload.path("executor_structured_output").path("claims").get(0).path("claim_id").asText()); - } - - @Test - void beforeModelMarksMalformedJsonAndKeepsRawAnswerFallback() throws Exception { - ToolTraceSummaryService traceSummaryService = mock(ToolTraceSummaryService.class); - when(traceSummaryService.buildVerifierTraceSummary(anyString(), anyString())).thenReturn(List.of()); - VerifierInputHook hook = new VerifierInputHook(traceSummaryService); - - AgentCommand command = hook.beforeModel( - List.of(new AssistantMessage("{\"diagnosis_summary\":\"缺少 claims\"}")), - RunnableConfig.builder().addMetadata("sessionId", "malformed-session").build() - ); - - JsonNode payload = readPayload(command); - assertEquals("malformed", payload.path("executor_output_parse_status").path("status").asText()); - assertTrue(payload.path("executor_structured_output").isNull()); - assertEquals("{\"diagnosis_summary\":\"缺少 claims\"}", payload.path("executor_final_answer").asText()); - assertEquals("malformed", VerifierContextHolder.getExecutorOutputParseStatus().get("status")); - } - - @Test - void beforeModelMarksPlainTextAsMissingStructuredOutput() throws Exception { - ToolTraceSummaryService traceSummaryService = mock(ToolTraceSummaryService.class); - when(traceSummaryService.buildVerifierTraceSummary(anyString(), anyString())).thenReturn(List.of()); - VerifierInputHook hook = new VerifierInputHook(traceSummaryService); - - AgentCommand command = hook.beforeModel( - List.of(new AssistantMessage("普通自然语言答案")), - RunnableConfig.builder().addMetadata("sessionId", "plain-session").build() - ); - - JsonNode payload = readPayload(command); - assertEquals("missing", payload.path("executor_output_parse_status").path("status").asText()); - assertTrue(payload.path("executor_structured_output").isNull()); - assertFalse(payload.path("executor_final_answer").asText().isBlank()); - } - - private JsonNode readPayload(AgentCommand command) throws Exception { - var field = AgentCommand.class.getDeclaredField("messages"); - field.setAccessible(true); - @SuppressWarnings("unchecked") - List messages = (List) field.get(command); - assertEquals(1, messages.size()); - Message message = messages.get(0); - assertTrue(message instanceof UserMessage); - return objectMapper.readTree(((UserMessage) message).getText()); - } - - private ToolInvocation invocation(Long id, String sessionId, String toolName, String rawPath, String text) { - return ToolInvocation.builder() - .id(id) - .sessionId(sessionId) - .toolName(toolName) - .retrievalDetails("{\"evidence_refs\":[{\"raw_path\":\"" + rawPath - + "\",\"text\":\"" + text + "\"}]}") - .build(); - } -} diff --git a/src/test/java/com/superbiz/agent/service/ChatServiceGraphIntegrationTest.java b/src/test/java/com/superbiz/agent/service/ChatServiceGraphIntegrationTest.java index 05edb69..c806bec 100644 --- a/src/test/java/com/superbiz/agent/service/ChatServiceGraphIntegrationTest.java +++ b/src/test/java/com/superbiz/agent/service/ChatServiceGraphIntegrationTest.java @@ -19,6 +19,7 @@ import com.superbiz.agent.repository.DiagnosisRunRepository; import com.superbiz.agent.repository.ToolInvocationRepository; import com.superbiz.agent.tool.LookupKnowledgeTool; import com.superbiz.agent.tool.RetrievedDocTracker; +import com.superbiz.agent.util.SessionContextHolder; import org.junit.jupiter.api.Test; import org.mockito.ArgumentCaptor; import org.springframework.ai.chat.model.ChatModel; @@ -33,6 +34,7 @@ import static org.junit.jupiter.api.Assertions.assertEquals; import static org.junit.jupiter.api.Assertions.assertFalse; import static org.junit.jupiter.api.Assertions.assertNotEquals; import static org.junit.jupiter.api.Assertions.assertNotNull; +import static org.junit.jupiter.api.Assertions.assertNull; import static org.junit.jupiter.api.Assertions.assertTrue; import static org.mockito.ArgumentMatchers.any; import static org.mockito.ArgumentMatchers.anyString; @@ -76,6 +78,8 @@ class ChatServiceGraphIntegrationTest { assertFalse(saved.getSelfEvaluation().contains("tool_trace_summary")); verify(fixture.evaluationService).evaluateRun(result.runId(), result.answer()); verify(fixture.retrievedDocTracker).clearSession(sessionId); + assertNull(SessionContextHolder.getSessionId()); + assertNull(SessionContextHolder.getRunId()); } @Test @@ -113,6 +117,8 @@ class ChatServiceGraphIntegrationTest { assertEquals(result.answer(), saved.getAnswer()); verify(fixture.evaluationService, never()).evaluateRun(anyString(), anyString()); verify(fixture.retrievedDocTracker).clearSession(sessionId); + assertNull(SessionContextHolder.getSessionId()); + assertNull(SessionContextHolder.getRunId()); } @Test