diff --git a/devflow/index.md b/devflow/index.md index afd67c2..ab601be 100644 --- a/devflow/index.md +++ b/devflow/index.md @@ -8,6 +8,7 @@ | 2026-07-07 | executor-evidence-output-contract | Chat质量门禁/证据归因 | Executor structured output, evidence bindings, Verifier structured claims, LOW_CONFID, hallucination | openspec/changes/archive/2026-07-07-executor-evidence-output-contract | archived | | 2026-07-07 | executor-v2-output-contract | Chat质量门禁/证据归因 | executor_evidence_v2, user_facing_answer removal, diagnosis_summary removal, structured renderer | openspec/changes/archive/2026-07-07-executor-v2-output-contract | archived | | 2026-07-07 | executor-gatekeeper-hook | Chat质量门禁/证据归因 | Gatekeeper, verifier payload, source_invocation_ids, tool_name match, self_evaluation | openspec/changes/archive/2026-07-07-executor-gatekeeper-hook | archived | +| 2026-07-07 | executor-verifier-claim-checks | Chat质量门禁/证据归因 | Verifier claim_checks, facts_checked compatibility, effective verdict guardrail, malformed output downgrade | openspec/changes/archive/2026-07-07-executor-verifier-claim-checks | archived | | 2026-07-06 | rag-eval-pipeline-closure | RAG/评测/回归闭环 | lookupResult fixture, LookupKnowledgeTool snapshot, evidenceBlocks, contextPack, retrievalTrace, rerankTrace, baseline diff, fallback case | devflow/projects/2026-07-06-rag-eval-pipeline-closure | archived | | 2026-07-06 | modular-rag-pipeline | RAG/Agent工具/证据链 | modular RAG, lookup_knowledge, evidenceBlocks, contextPack, rerank, retrievalTrace, L0 hint, unfiltered retry | openspec/changes/archive/2026-07-06-modular-rag-pipeline | archived | | 2026-07-05 | mvp-demo-interview-runbook | MVP Demo/Interview | Plan C, payment timeout, runbook, trace checklist, demo script | openspec/changes/archive/2026-07-05-mvp-demo-interview-runbook | archived | diff --git a/devflow/projects/2026-07-07-executor-verifier-claim-checks/acceptance.md b/devflow/projects/2026-07-07-executor-verifier-claim-checks/acceptance.md new file mode 100644 index 0000000..e663c9d --- /dev/null +++ b/devflow/projects/2026-07-07-executor-verifier-claim-checks/acceptance.md @@ -0,0 +1,58 @@ +# Acceptance + +## Implementation Result + +Implemented stage three of Executor Structured Output V2: + +- Verifier prompt now validates claim derivability rather than scanning final natural-language output. +- Verifier output supports `claim_checks`. +- `ChatService` derives compatibility `facts_checked` from `claim_checks`. +- `ChatService` persists both `claim_checks` and `facts_checked`. +- Effective verdict guardrails prevent Gatekeeper failures and malformed structured output from remaining `PASS`. +- Verifier logging summarizes `claim_checks`. + +## Static Verification + +- Reviewed `git diff --stat` and changed files are scoped to stage three implementation, tests, OpenSpec/devflow, and the issue handoff document. +- `cmd /c openspec validate executor-verifier-claim-checks` passed. +- After OpenSpec archive, `cmd /c openspec validate --specs` passed. + +## Script Verification + +Passed: + +```powershell +mvn "-Dtest=ExecutorGatekeeperServiceTest,VerifierInputHookTest,ChatServiceSequentialAgentTest" test +``` + +Coverage: + +- Verifier payload and Gatekeeper hook behavior. +- Gatekeeper schema/invocation validation. +- `claim_checks` parsing and persistence. +- `claim_checks` to `facts_checked` compatibility mapping. +- all claim verification mapping classes. +- Gatekeeper failure downgrade from model `PASS`. +- malformed Executor output downgrade from model `PASS`. + +The targeted Maven test command was re-run after OpenSpec archive and passed. + +## Browser / Manual Verification + +Not run. This stage changes backend prompt, parser, audit, and tests only. + +## OpenSpec Archive Status + +Archived: + +```text +openspec/changes/archive/2026-07-07-executor-verifier-claim-checks +``` + +Archive follow-up: the generated canonical `chat-verifier-agent` spec was reviewed and amended to preserve pre-existing verifier input and Gatekeeper schema scenarios while adding the new claim-check scenarios. + +## Remaining Risks + +- Composer is not implemented in this stage; final PASS rendering still uses the temporary V2 renderer until stage four. +- Full eval fixture expansion is deferred to stage five. +- Existing Maven warnings about duplicate test dependency and Lombok builder defaults remain outside this stage. diff --git a/devflow/projects/2026-07-07-executor-verifier-claim-checks/brief.md b/devflow/projects/2026-07-07-executor-verifier-claim-checks/brief.md new file mode 100644 index 0000000..9cd18ab --- /dev/null +++ b/devflow/projects/2026-07-07-executor-verifier-claim-checks/brief.md @@ -0,0 +1,41 @@ +# Executor Verifier Claim Checks + +## Background + +Stage one moved Chat Executor to `executor_evidence_v2`, and stage two added deterministic Gatekeeper checks before Verifier. After those stages, Verifier still primarily used the legacy `facts_checked` contract and could still treat `executor_final_answer` as a fact source. + +That left two risks: + +- Verifier could still extract extra confirmed facts from natural-language Executor output. +- Downstream audit and retry consumers could not distinguish V2 claim-level verification from legacy fact checks. + +## Goal + +Make Verifier V2 claim-oriented: + +- verify `executor_structured_output.claims` as the primary target; +- emit `claim_checks` as the authoritative V2 result; +- keep `facts_checked` only as a compatibility projection; +- enforce code-side guardrails so Gatekeeper failures or malformed structured output cannot remain effective `PASS`. + +## Scope + +- Updated `chat-verifier-prompt.md` to frame verification as claim derivability. +- Extended `ChatService` to parse, normalize, map, and persist `claim_checks`. +- Added effective verdict guardrails for Gatekeeper failure and malformed/missing Executor structured output. +- Updated verifier logging summaries to count `claim_checks`. +- Updated sequential workflow tests to cover claim mapping and downgrade behavior. + +## Non-Goals + +- No Composer integration in this phase. +- No final-answer material filtering beyond existing templates and temporary V2 renderer. +- No Executor retry behavior change. +- No Gatekeeper rule expansion. +- No database schema migration. + +## OpenSpec + +- Active change before archive: `openspec/changes/executor-verifier-claim-checks` +- Capability: `chat-verifier-agent` +- Scale: standard diff --git a/devflow/projects/2026-07-07-executor-verifier-claim-checks/decisions.md b/devflow/projects/2026-07-07-executor-verifier-claim-checks/decisions.md new file mode 100644 index 0000000..c220cc1 --- /dev/null +++ b/devflow/projects/2026-07-07-executor-verifier-claim-checks/decisions.md @@ -0,0 +1,57 @@ +# Decisions + +## Scope Decision + +Stage three is limited to Verifier V2 claim checks. Composer is explicitly deferred to stage four. + +Reason: Composer requires stable verifier output and allowed-material filtering; mixing it into this stage would make rollback and acceptance unclear. + +## Contract Decision + +`claim_checks` is the authoritative V2 verifier output. + +`facts_checked` remains as a compatibility projection generated from `claim_checks` when present. + +Reason: existing low-confidence rendering, retry context, trace output, and evaluation code still depend on `facts_checked`. + +## Mapping Decision + +Claim verification maps to legacy facts as follows: + +| claim verification | legacy facts_checked verification | +|---|---| +| `direct_observation` | `direct_evidence` | +| `reasonable_inference` | `indirect_support` | +| `overstated` | `indirect_support` | +| `unsupported` | `no_evidence` | +| `external_unknown` | `no_evidence` | +| `contradicted` | `contradicted` | + +## Guardrail Decision + +Effective verdict is enforced in code: + +- `gatekeeper_result.status=fail` cannot remain `PASS`. +- `evidence.invocation_ref` failure downgrades to `REJECT`. +- Other Gatekeeper failures downgrade at least to `LOW_CONFID`. +- missing/malformed Executor structured output cannot remain `PASS`. + +Reason: prompt compliance is not deterministic enough for safety-critical evidence attribution. + +## Apply Fix Record + +Initial targeted Maven verification failed because older tests expected PASS to return Executor natural-language output or V1 `user_facing_answer`. + +Classification: test drift from the committed OpenSpec, not a design blocker. + +Resolution: update tests to use valid Executor V2 output for PASS paths and assert downgrade behavior for malformed or Gatekeeper-failed outputs. + +## Interface Impact + +L2 internal contract extension: + +- Verifier output gains `claim_checks`. +- Existing `facts_checked` remains available. +- Persistence JSON gains `claim_checks` under existing `self_evaluation`. + +No external API or database schema changes. diff --git a/devflow/projects/2026-07-07-executor-verifier-claim-checks/evidence.md b/devflow/projects/2026-07-07-executor-verifier-claim-checks/evidence.md new file mode 100644 index 0000000..0673a23 --- /dev/null +++ b/devflow/projects/2026-07-07-executor-verifier-claim-checks/evidence.md @@ -0,0 +1,35 @@ +# Evidence + +## Relevant History + +- `executor-v2-output-contract`: Executor emits `executor_evidence_v2` and no longer emits final-expression fields. +- `executor-gatekeeper-hook`: Gatekeeper validates schema and invocation references before Verifier and persists `gatekeeper_result`. +- `chat-verifier-agent`: Existing Verifier used `facts_checked`, low-confidence rendering, retry context, and verifier audit. + +## Code Evidence + +- `src/main/resources/prompts/chat-verifier-prompt.md`: Verifier prompt now makes `executor_structured_output.claims` primary and treats `executor_final_answer` as debug/fallback only. +- `src/main/java/com/superbiz/agent/service/ChatService.java`: parses `claim_checks`, maps them to compatibility `facts_checked`, persists both, and applies effective verdict guardrails. +- `src/main/java/com/superbiz/agent/hook/AgentLoggingHook.java`: verifier thought summaries now include `claim_checks`. +- `src/test/java/com/superbiz/agent/service/ChatServiceSequentialAgentTest.java`: covers claim-check mapping, Gatekeeper downgrade, malformed output downgrade, and V2 PASS paths. + +## Evidence-Driven Conclusions + +- `facts_checked` cannot be removed yet because existing low-confidence templates, retry context, trace tooling, and eval paths still consume it. +- Prompt-only prevention is insufficient for Gatekeeper failures; `ChatService` must enforce effective verdict downgrades in code. +- No database schema migration is needed because `claim_checks` is persisted inside existing `diagnosis_session.self_evaluation`. +- Composer remains stage four and must not be mixed into this stage. + +## Verification Evidence + +Script verification passed: + +```powershell +mvn "-Dtest=ExecutorGatekeeperServiceTest,VerifierInputHookTest,ChatServiceSequentialAgentTest" test +cmd /c openspec validate executor-verifier-claim-checks +``` + +Known existing warnings: + +- Maven reports duplicate `spring-boot-starter-test` dependency in `pom.xml`. +- Existing Lombok `@Builder` default warnings remain. diff --git a/mvp/issues/executor-structured-output-v2.md b/mvp/issues/executor-structured-output-v2.md index 7c113f3..f4d2d22 100644 --- a/mvp/issues/executor-structured-output-v2.md +++ b/mvp/issues/executor-structured-output-v2.md @@ -1,10 +1,10 @@ -# Executor Structured Output V2 实施 Issue +# Executor Structured Output V2 可执行设计与实施 Issue -**状态**:分阶段实施中 +**状态**:阶段三实施中 **严重程度**:高 **创建日期**:2026-07-07 **最后更新**:2026-07-08 -**文档类型**:可执行 issue / 分阶段实施说明 +**文档类型**:可执行设计 issue / 分阶段实施说明 **范围**:Chat Executor 输出结构、Gatekeeper、Verifier、Composer 数据契约调整 **关联问题**: @@ -17,7 +17,25 @@ ## 0. 给后续实现 Agent 的执行入口 -这份文档不是单纯的数据结构草案,而是 `Executor Structured Output V2` 的分阶段实施 issue。后续 agent 接手时,应先理解问题背景和阶段边界,再进入具体字段定义。 +这份文档是 `Executor Structured Output V2` 的可执行设计 issue,不是单纯的数据结构草案。后续 agent 接手时,先按本节确认背景、当前阶段、执行纪律和验收边界,再进入后面的字段契约。 + +### 0.0 当前接手快照 + +截至 2026-07-08,阶段状态如下: + +| 阶段 | 状态 | 记录 | +|---|---|---| +| 阶段一:Executor V2 输出契约 | 已完成、已归档、已提交 | `050cbc8 feat(agent): add executor evidence v2 contract` | +| 阶段二:Gatekeeper 接入 VerifierInputHook | 已完成、已归档、已提交 | `c5e496e feat(agent): add executor gatekeeper hook` | +| 阶段三:Verifier V2 可推导性校验 | 实施中 | active change:`openspec/changes/executor-verifier-claim-checks` | +| 阶段四:Composer 输出最终答案 | 未开始 | 等阶段三归档并提交后再启动 | +| 阶段五:回归评测与审计闭环 | 未开始 | 等 Composer 链路稳定后补齐 | + +接手时的第一动作: + +1. 先运行 `git status --short`,确认是否已有阶段三未提交改动。 +2. 如果 `openspec/changes/executor-verifier-claim-checks` 仍存在,继续阶段三,不要跳到 Composer。 +3. 阶段三完成后,必须先验证、归档 OpenSpec、回填 devflow、单独 commit,再进入阶段四。 推荐阅读顺序: @@ -42,6 +60,13 @@ - 验收失败如果是代码问题,agent 自行修复;如果是设计决策不明确,停下来问用户。 - 本 issue 不要求一次性完成所有阶段;后续 agent 应从当前 git/OpenSpec/devflow 状态继续推进。 +交付纪律: + +- 每个阶段只解决该阶段的问题,不顺手做下一阶段。 +- 每个阶段至少要有主成功路径和新增失败路径测试。 +- 提交前必须确认 diff 只包含当前阶段内容。 +- 不提交、不推送,除非用户明确要求;本 issue 默认只指导阶段实现和本地提交。 + --- ## 0.1 背景 @@ -187,13 +212,25 @@ Executor raw JSON | 阶段 | 名称 | 目标 | 状态 | 退出条件 | |---|---|---|---|---| | 1 | Executor V2 输出契约 | 移除 Executor 最终表达字段,只输出结构化诊断材料 | 已完成并归档 | `executor_evidence_v2` 可运行,最终用户不再看到 raw JSON | -| 2 | Gatekeeper 接入 VerifierInputHook | 在 Verifier 前加入确定性拦截与审计 | 进行中 | `gatekeeper_result` 进入 Verifier payload 和 `self_evaluation` | -| 3 | Verifier V2 可推导性校验 | 从 `facts_checked` 转向 `claim_checks`,判断 claims 是否可由证据推出 | 未开始 | Verifier 不再从自然语言答案抽取额外事实 | +| 2 | Gatekeeper 接入 VerifierInputHook | 在 Verifier 前加入确定性拦截与审计 | 已完成并归档 | `gatekeeper_result` 进入 Verifier payload 和 `self_evaluation` | +| 3 | Verifier V2 可推导性校验 | 从 `facts_checked` 转向 `claim_checks`,判断 claims 是否可由证据推出 | 实施中 | Verifier 不再从自然语言答案抽取额外事实 | | 4 | Composer 最终表达 | 由 Composer 根据 Verifier 允许材料生成最终用户答案 | 未开始 | PASS/LOW_CONFID/REJECT 都不读取 Executor `user_facing_answer` | | 5 | 回归评测与审计闭环 | 用测试和 eval fixtures 证明幻觉拦截链路有效 | 未开始 | 伪造 ID、张冠李戴、hypothesis 写成事实等场景都有回归覆盖 | 更详细的实施拆解见 `21. 实施阶段`。 +阶段推进顺序必须串行: + +```text +阶段一 archive + commit + -> 阶段二 archive + commit + -> 阶段三 archive + commit + -> 阶段四 archive + commit + -> 阶段五 archive + commit +``` + +不能在阶段三未归档提交时启动 Composer,也不能在 Composer 未稳定时把回归评测阶段提前做成大杂烩。 + --- ## 0.6 范围与非目标 @@ -279,6 +316,22 @@ cmd /c openspec validate --specs 如阶段新增专门测试,应把测试类加入 Maven `-Dtest` 列表。 +后续 agent 每个阶段的固定执行清单: + +1. 确认当前阶段:读取本 issue、`git status --short`、`openspec status`。 +2. 确认 OpenSpec:没有 active change 时先创建;已有当前阶段 active change 时继续使用。 +3. 实现阶段内代码:只改该阶段要求的 prompt/service/hook/test,不夹带下一阶段。 +4. 验证:运行该阶段建议测试和 `cmd /c openspec validate ...`。 +5. 回填任务:更新 OpenSpec `tasks.md`、`decisions.md`,记录实际验证命令和结果。 +6. 归档:创建或更新 `devflow/projects/-/`,然后 `cmd /c openspec archive -y`。 +7. 复验:归档后再次运行最小测试和 `cmd /c openspec validate --specs`。 +8. 提交:检查 `git diff --cached --stat`,确认只包含当前阶段,再单独 commit。 + +若步骤 4 或 7 失败: + +- 代码或测试问题:自行修复并重新验证。 +- 阶段边界或协议设计问题:停止,向用户说明阻塞点,不进入下一阶段。 + --- ## 0.10 推荐实现切片 @@ -297,6 +350,27 @@ cmd /c openspec validate --specs --- +## 0.11 为什么这样拆阶段 + +本 issue 的改造目标不是一次性重写 Chat 诊断链路,而是逐步切断“证据不足但自然语言先下结论”的通道。阶段拆分按风险来源分层: + +| 阶段 | 切断的风险 | 为什么不能合并 | +|---|---|---| +| 阶段一 | Executor 提前输出最终诊断话术 | 先移除污染源,否则后面 Verifier/Composer 仍会被旧答案影响 | +| 阶段二 | 伪造 ID、工具名不匹配、非法 schema 等物理级幻觉 | 这类问题不需要 LLM 判断,先用代码低成本拦截 | +| 阶段三 | claim 与证据之间是否可推导 | 这是 Verifier 的语义职责,必须在 Gatekeeper 之后处理 | +| 阶段四 | 最终用户答案夹带未验证事实 | 只有 Verifier 输出稳定后,Composer 才知道哪些材料可用 | +| 阶段五 | 新链路是否真的防住历史幻觉场景 | 等链路完整后再做系统化回归,避免测试绑死临时实现 | + +每个阶段的设计都必须满足两个条件: + +- 可以独立验证,不依赖后续阶段已经完成。 +- 失败时可以独立回滚,不破坏已经归档提交的前置阶段。 + +因此后续实现时不要把“顺手移除临时 renderer”“顺手扩展 Gatekeeper 规则”“顺手接 Composer”混进阶段三。阶段三只解决 Verifier `claim_checks` 和有效 verdict guardrail。 + +--- + ## 1. 设计目标 当前 `executor_evidence_v1` 同时包含结构化诊断材料和自然语言表达字段: @@ -1440,13 +1514,13 @@ Composer 输出严格 JSON。 ### 阶段二:Gatekeeper 接入 VerifierInputHook -状态:进行中。 +状态:已完成并归档。 -当前 OpenSpec change: +已完成记录: -```text -openspec/changes/executor-gatekeeper-hook -``` +- OpenSpec archive:`openspec/changes/archive/2026-07-07-executor-gatekeeper-hook` +- devflow:`devflow/projects/2026-07-07-executor-gatekeeper-hook` +- commit:`c5e496e feat(agent): add executor gatekeeper hook` 目标:在 Verifier 之前用确定性规则拦截物理级幻觉。 @@ -1487,7 +1561,6 @@ openspec/changes/executor-gatekeeper-hook ```powershell mvn "-Dtest=ExecutorGatekeeperServiceTest,VerifierInputHookTest,ChatServiceSequentialAgentTest" test -cmd /c openspec validate executor-gatekeeper-hook cmd /c openspec validate --specs ``` @@ -1499,7 +1572,19 @@ cmd /c openspec validate --specs ### 阶段三:Verifier V2 可推导性校验 -状态:未开始。 +状态:实施中。 + +当前 OpenSpec change: + +```text +openspec/changes/executor-verifier-claim-checks +``` + +接手说明: + +- 如果该 active change 仍存在,继续完成阶段三,不要启动阶段四。 +- 阶段三已有 OpenSpec proposal/design/spec/tasks,后续实现应先对照 `tasks.md` 逐项完成。 +- 阶段三完成后必须 archive、回填 devflow、单独 commit,推荐提交信息:`feat(agent): add verifier claim checks`。 目标:Verifier 不再逐字扫描自然语言,而是校验 claim 是否能由证据合理推出。 @@ -1532,11 +1617,22 @@ cmd /c openspec validate --specs - `gatekeeper_result.status=fail` + Verifier 返回 PASS 时,代码侧应降级或测试 prompt 禁止该行为。 - `unsupported` / `overstated` 能进入低置信模板需要的 missing evidence 语义。 +建议验证命令: + +```powershell +mvn "-Dtest=ExecutorGatekeeperServiceTest,VerifierInputHookTest,ChatServiceSequentialAgentTest" test +cmd /c openspec validate executor-verifier-claim-checks +cmd /c openspec validate --specs +``` + 阶段退出条件: - Verifier prompt 已不再要求逐字扫描最终自然语言答案。 - 审计中同时可见 `claim_checks` 和兼容 `facts_checked`。 - 现有低置信重试链路不回归。 +- OpenSpec change 已 archive。 +- devflow 已回填 brief、evidence、decisions、acceptance。 +- 本阶段代码和文档已单独 commit。 ### 阶段四:Composer 输出最终答案 diff --git a/openspec/changes/archive/2026-07-07-executor-verifier-claim-checks/.archive-ready b/openspec/changes/archive/2026-07-07-executor-verifier-claim-checks/.archive-ready new file mode 100644 index 0000000..395527d --- /dev/null +++ b/openspec/changes/archive/2026-07-07-executor-verifier-claim-checks/.archive-ready @@ -0,0 +1 @@ +ready diff --git a/openspec/changes/archive/2026-07-07-executor-verifier-claim-checks/.committed b/openspec/changes/archive/2026-07-07-executor-verifier-claim-checks/.committed new file mode 100644 index 0000000..d0fe822 --- /dev/null +++ b/openspec/changes/archive/2026-07-07-executor-verifier-claim-checks/.committed @@ -0,0 +1 @@ +committed diff --git a/openspec/changes/archive/2026-07-07-executor-verifier-claim-checks/.openspec.yaml b/openspec/changes/archive/2026-07-07-executor-verifier-claim-checks/.openspec.yaml new file mode 100644 index 0000000..aee4ef1 --- /dev/null +++ b/openspec/changes/archive/2026-07-07-executor-verifier-claim-checks/.openspec.yaml @@ -0,0 +1,2 @@ +schema: spec-driven +created: 2026-07-07 diff --git a/openspec/changes/archive/2026-07-07-executor-verifier-claim-checks/decisions.md b/openspec/changes/archive/2026-07-07-executor-verifier-claim-checks/decisions.md new file mode 100644 index 0000000..9ead985 --- /dev/null +++ b/openspec/changes/archive/2026-07-07-executor-verifier-claim-checks/decisions.md @@ -0,0 +1,110 @@ +# Decisions: executor-verifier-claim-checks + +## sm-flow Progress + +### Clarify + +Entry summary: implement stage three of Executor Structured Output V2 by making Verifier V2 claim-oriented. + +Slug: `executor-verifier-claim-checks` + +Scale: standard. This changes internal verifier output/audit contracts and parsing logic, but does not change external APIs or database schema. + +### Context + +Relevant history: + +- `executor-v2-output-contract`: Executor emits `executor_evidence_v2` and no longer emits final-expression fields. +- `executor-gatekeeper-hook`: Gatekeeper validates schema and invocation references before Verifier and persists `gatekeeper_result`. +- `chat-verifier-agent`: Current Verifier still primarily uses `facts_checked`. + +Current code shape: + +- `chat-verifier-prompt.md` still frames the task around `executor_final_answer` and `facts_checked`. +- `ChatService.parseVerifierDecision(...)` only parses `facts_checked`. +- `buildLowConfidenceOutput(...)` and `buildRetryContext(...)` consume `VerifierDecision.factsChecked()`. +- `persistVerifierEvaluation(...)` already persists parse status, structured output, trace summary, and gatekeeper result. +- `AgentLoggingHook` summarizes verifier output using `facts_checked`. + +### Grill + +Question pool: + +| Question | Mode | Resolution | +|---|---|---| +| Should Verifier still scan `executor_final_answer` when structured output is valid? | evidence-driven | No. Stage three explicitly makes claims the primary target and raw output debug/fallback only. | +| Should `facts_checked` be removed now? | evidence-driven | No. It remains a compatibility projection for low-confidence templates, retry context, eval, and trace tooling. | +| Should Gatekeeper fail prevention be prompt-only? | evidence-driven | No. Stage two noted prompt compliance is not deterministic; stage three adds code-side effective verdict guard. | +| Does this require database migration? | evidence-driven | No. `claim_checks` is stored under existing JSON self_evaluation. | +| Does this introduce Composer? | evidence-driven | No. Composer is stage four. | + +No user-interview questions are open for this stage. + +### Specify + +OpenSpec artifacts: + +- `proposal.md`: why and scope. +- `design.md`: Verifier V2 output, compatibility mapping, verdict guardrails, risks. +- `specs/chat-verifier-agent/spec.md`: observable requirements for claim checks, compatibility facts, persistence, and guardrails. +- `tasks.md`: executable implementation and verification checklist. + +### Audit + +Architecture risk summary: + +- This is an L2 internal verifier contract extension. +- Existing consumers continue using `facts_checked`, which is now generated from `claim_checks` when present. +- Gatekeeper PASS prevention becomes deterministic in code, reducing reliance on prompt compliance. +- No external API or database schema changes are introduced. + +Cross-artifact alignment: + +| Source | Target | Status | +|---|---|---| +| issue stage three | proposal | aligned | +| proposal scope / non-goals | design | aligned | +| design output contract and guardrails | specs | aligned | +| specs observable behavior | tasks | aligned | + +Interface impact: + +- Verifier output: L2 internal extension with `claim_checks`. +- Persistence JSON: L2 internal audit extension in existing `self_evaluation`. +- External HTTP/API behavior: unchanged. + +### Commit + +Commit gate result: passed. + +- `proposal.md`, `design.md`, `specs/chat-verifier-agent/spec.md`, and `tasks.md` exist. +- `cmd /c openspec validate executor-verifier-claim-checks` passed. +- `cmd /c openspec status --change executor-verifier-claim-checks` reports 4/4 artifacts complete. +- No unresolved user-interview questions remain for this stage. + +### Apply + +Capability source: OpenSpec CLI + `openspec-apply-change` protocol, executed through the local shell tool. No separate semantic/LSP tools are available in this session, so implementation evidence used OpenSpec artifacts, `rg`/diff inspection, and targeted tests. + +Implemented changes: + +- Updated `chat-verifier-prompt.md` so `executor_structured_output.claims` is the primary verification target. +- Added `claim_checks` parsing, normalization, persistence, and compatibility mapping to legacy `facts_checked` in `ChatService`. +- Added code-side effective verdict guardrails: + - missing/malformed structured output cannot remain `PASS`; + - `gatekeeper_result.status=fail` cannot remain `PASS`; + - `evidence.invocation_ref` failures downgrade to `REJECT`; + - other Gatekeeper failures downgrade at least to `LOW_CONFID`. +- Updated verifier logging summaries to account for `claim_checks`. +- Updated sequential workflow tests to use valid Executor V2 output for PASS paths and to verify downgrade paths for Gatekeeper failure and malformed output. + +Conflict / fix record: + +- Initial targeted Maven verification failed because legacy tests still expected PASS to return Executor natural-language output or V1 `user_facing_answer`. +- Classification: code/test drift from the committed OpenSpec, not a design blocker. +- Resolution: updated tests to assert the stage-three contract: valid V2 structured output may PASS through the temporary renderer, while missing/malformed/V1-style output cannot produce effective PASS through natural-language fallback. + +Verification: + +- `mvn "-Dtest=ExecutorGatekeeperServiceTest,VerifierInputHookTest,ChatServiceSequentialAgentTest" test` passed. +- `cmd /c openspec validate executor-verifier-claim-checks` passed. diff --git a/openspec/changes/archive/2026-07-07-executor-verifier-claim-checks/design.md b/openspec/changes/archive/2026-07-07-executor-verifier-claim-checks/design.md new file mode 100644 index 0000000..52e5a72 --- /dev/null +++ b/openspec/changes/archive/2026-07-07-executor-verifier-claim-checks/design.md @@ -0,0 +1,140 @@ +## Context + +Current state after stage two: + +```text +chat_planner + -> chat_executor + -> VerifierInputHook + Gatekeeper + -> chat_verifier + -> ChatService final rendering +``` + +Verifier receives explicit inputs: + +- `original_query` +- `executor_final_answer` +- `executor_structured_output` +- `executor_output_parse_status` +- `tool_trace_summary` +- `gatekeeper_result` +- `retry_context` + +However, Verifier output is still primarily: + +```json +{ + "verdict": "PASS", + "groundedness_score": 0.8, + "critical_fact_count": 1, + "facts_checked": [], + "rationale": "..." +} +``` + +Stage three introduces V2 output while preserving the old compatibility field. + +## Verifier V2 Output + +Verifier should output: + +```json +{ + "verdict": "LOW_CONFID", + "groundedness_score": 0.62, + "critical_fact_count": 1, + "claim_checks": [ + { + "claim_id": "claim-1", + "claim_text": "payment-service CPU usage is high", + "claim_type": "symptom", + "verification": "direct_observation", + "detail": "query_metrics shows CPU=92%", + "evidence_refs": [ + { + "trace_ref": "trace-1", + "tool_name": "query_metrics", + "source_invocation_ids": [394], + "note": "metrics summary contains CPU=92%" + } + ] + } + ], + "hypothesis_checks": [], + "facts_checked": [], + "rationale": "..." +} +``` + +`facts_checked` remains for compatibility. If Verifier does not emit it, `ChatService` must derive it from `claim_checks`. + +## Claim Verification Set + +`claim_checks[].verification` is limited to: + +| Value | Meaning | Legacy mapping | +|---|---|---| +| `direct_observation` | Evidence directly observes the claim | `direct_evidence` | +| `reasonable_inference` | Evidence can reasonably support the claim, but not as direct observation | `indirect_support` | +| `overstated` | Evidence partially supports the claim, but the claim says too much | `indirect_support` | +| `unsupported` | Evidence is insufficient | `no_evidence` | +| `external_unknown` | Claim introduces evidence-external entity/value/root cause | `no_evidence` | +| `contradicted` | Claim conflicts with evidence | `contradicted` | + +## Compatibility Mapping + +`ChatService` must keep old downstream behavior alive by producing `facts_checked`. + +Suggested mapping: + +```text +facts_checked[].fact = "{claim_id}: {claim_text}" +facts_checked[].is_critical = claim_type in ["root_cause", "symptom", "impact", "risk"] +facts_checked[].verification = mapped legacy verification +facts_checked[].detail = claim_checks[].detail +facts_checked[].evidence_refs = claim_checks[].evidence_refs +``` + +If Verifier emits both `claim_checks` and `facts_checked`, `claim_checks` is authoritative. `facts_checked` may be replaced by the deterministic compatibility projection to avoid inconsistent audit data. + +If Verifier emits only old `facts_checked`, ChatService keeps the old path. + +## Verdict Guardrails + +Gatekeeper fail: + +- If `gatekeeper_result.status = fail`, effective verdict must not be `PASS`. +- If the model returns `PASS`, ChatService should downgrade the effective verdict to `LOW_CONFID` or `REJECT`. +- For this phase, `evidence.invocation_ref` failure should downgrade to `REJECT`; other Gatekeeper failures should downgrade to `LOW_CONFID`. + +Malformed or missing structured output: + +- If `executor_output_parse_status.status` is `missing` or `malformed`, Verifier should not use natural-language extraction to produce PASS. +- Effective verdict should be `LOW_CONFID`. + +## Prompt Boundary + +The prompt should say: + +- Primary target is `executor_structured_output.claims`. +- Do not extract additional confirmed facts from `executor_final_answer` when structured output is valid. +- `executor_final_answer` is debug/fallback only. +- `claim_checks` is the primary output. +- `facts_checked` is compatibility output. + +## Interface Impact + +- L2 internal contract extension. +- No external API change. +- No database schema change. +- Audit JSON gains `claim_checks`. + +## Risks / Mitigations + +- Risk: old low-confidence templates rely on `facts_checked`. + - Mitigation: derive `facts_checked` from `claim_checks`. +- Risk: prompt-only Gatekeeper PASS prevention is insufficient. + - Mitigation: add code-side effective verdict guard. +- Risk: Agent logging only summarizes `facts_checked`. + - Mitigation: update logging to understand `claim_checks` while keeping old summary compatibility. + diff --git a/openspec/changes/archive/2026-07-07-executor-verifier-claim-checks/proposal.md b/openspec/changes/archive/2026-07-07-executor-verifier-claim-checks/proposal.md new file mode 100644 index 0000000..ac35d47 --- /dev/null +++ b/openspec/changes/archive/2026-07-07-executor-verifier-claim-checks/proposal.md @@ -0,0 +1,47 @@ +## Why + +Stage one moved Executor to `executor_evidence_v2`, and stage two added deterministic Gatekeeper checks before Verifier. The Verifier still mainly operates through the legacy `facts_checked` contract and the prompt still allows fallback extraction from `executor_final_answer`. + +That keeps two problems alive: + +- Verifier can still treat natural-language Executor output as a fact source. +- Downstream code cannot distinguish claim-level verification results from legacy natural-language fact checks. + +This phase makes Verifier V2 claim-oriented: Verifier evaluates `executor_structured_output.claims` for whether each claim can be reasonably derived from evidence, emits `claim_checks`, and keeps `facts_checked` only as a compatibility projection. + +## What Changes + +- Update `chat-verifier-prompt.md` so the primary verification target is `executor_structured_output.claims`. +- Add Verifier V2 output field `claim_checks`. +- Preserve compatibility by mapping `claim_checks` into legacy `facts_checked`. +- Parse and persist `claim_checks` in `ChatService`. +- Ensure `gatekeeper_result.status=fail` cannot result in an effective `PASS`. +- Ensure missing or malformed Executor structured output does not fall back to natural-language fact extraction for PASS. + +## Capabilities + +### New Capabilities + +None. + +### Modified Capabilities + +- `chat-verifier-agent`: Verifier output now includes claim-level checks and uses claim derivability as the primary groundedness contract. + +## Impact + +- Affected prompt: `src/main/resources/prompts/chat-verifier-prompt.md`. +- Affected service: `ChatService.parseVerifierDecision(...)`, retry context generation, verifier persistence. +- Affected audit: `diagnosis_session.self_evaluation.verifier_evaluation` gains `claim_checks` and keeps `facts_checked`. +- Affected logging: verifier thought summaries may count `claim_checks`. +- Affected tests: `ChatServiceSequentialAgentTest` and focused verifier parsing tests. +- Database schema: no table or column change. + +## Non-Goals + +- No Composer in this phase. +- No final-answer material filtering in this phase beyond existing LOW_CONFID/REJECT templates and temporary V2 renderer. +- No Executor retry behavior change. +- No Gatekeeper rule expansion. +- No database schema migration. + diff --git a/openspec/changes/archive/2026-07-07-executor-verifier-claim-checks/specs/chat-verifier-agent/spec.md b/openspec/changes/archive/2026-07-07-executor-verifier-claim-checks/specs/chat-verifier-agent/spec.md new file mode 100644 index 0000000..748af4c --- /dev/null +++ b/openspec/changes/archive/2026-07-07-executor-verifier-claim-checks/specs/chat-verifier-agent/spec.md @@ -0,0 +1,109 @@ +## MODIFIED Requirements + +### Requirement: Verifier SHALL fact-check Executor answers +The system SHALL have a Verifier Agent that reads structured Executor claims and the tool call history, then produces a structured verdict based on claim derivability. + +#### Scenario: PASS verdict when all claims have evidence +- **WHEN** all critical claims in `executor_structured_output.claims` have direct observation or reasonable inference support in tool call results +- **AND** at least one critical claim has direct observation +- **AND** no critical claim is contradicted, unsupported, external unknown, or overstated +- **AND** `gatekeeper_result.status` is not `fail` +- **THEN** the Verifier MAY output verdict="PASS" with groundedness_score ≥ 0.5 + +#### Scenario: LOW_CONFID verdict with partial evidence +- **WHEN** no critical claim contradicts the tool results +- **AND** some critical claims are `unsupported`, `external_unknown`, or `overstated` +- **THEN** the Verifier SHALL output verdict="LOW_CONFID" + +#### Scenario: LOW_CONFID verdict with only inference support +- **WHEN** no critical claim contradicts the tool results +- **AND** all critical claims are only `reasonable_inference` +- **THEN** the Verifier SHALL output verdict="LOW_CONFID" + +#### Scenario: REJECT verdict when claims contradict evidence +- **WHEN** any critical claim in `executor_structured_output.claims` contradicts tool call results +- **OR** the claim fabricates a key entity, error code, or conclusion that does not exist in the tool evidence +- **THEN** the Verifier SHALL output verdict="REJECT" + +#### Scenario: Structured Executor claims are verified first +- **WHEN** `executor_structured_output.claims` is present and valid +- **THEN** Verifier SHALL verify each structured claim against `tool_trace_summary` through `claim_checks` +- **AND** each claim's evidence bindings SHALL reference existing trace or invocation identifiers when those identifiers are available +- **AND** a claim with fabricated or missing evidence references SHALL NOT be classified as `direct_observation` +- **AND** Verifier SHALL NOT add extra confirmed facts from `executor_final_answer` that are absent from `executor_structured_output.claims` + +#### Scenario: Malformed structured output cannot pass through natural language fallback +- **WHEN** Executor does not return parseable structured output +- **THEN** Verifier SHALL NOT produce an effective `PASS` by extracting facts from `executor_final_answer` +- **AND** the effective verdict SHALL be `LOW_CONFID` + +### Requirement: Verifier SHALL output structured JSON +The Verifier SHALL output a JSON object with verdict, groundedness_score, claim_checks array, compatibility facts_checked array, and rationale. + +#### Scenario: claim-level verifier output is accepted +- **WHEN** the Verifier checks Executor V2 structured output +- **THEN** the output SHALL contain `verdict`, `groundedness_score`, `critical_fact_count`, `claim_checks`, `facts_checked`, and `rationale` +- **AND** `claim_checks` SHALL be the primary V2 verification result +- **AND** `facts_checked` SHALL remain available for compatibility + +### Requirement: facts_checked SHALL use a fixed classification set +The system SHALL continue to expose legacy `facts_checked` using its fixed verification classification set. + +#### Scenario: claim checks are mapped to legacy facts +- **WHEN** Verifier output contains `claim_checks` +- **THEN** ChatService SHALL derive compatibility `facts_checked` +- **AND** `direct_observation` SHALL map to `direct_evidence` +- **AND** `reasonable_inference` and `overstated` SHALL map to `indirect_support` +- **AND** `unsupported` and `external_unknown` SHALL map to `no_evidence` +- **AND** `contradicted` SHALL map to `contradicted` + +### Requirement: Verifier SHALL be observable +The Verifier's verdict SHALL be persisted for observability. + +#### Scenario: claim checks written to self_evaluation +- **WHEN** the Verifier evaluation is persisted +- **THEN** `diagnosis_session.self_evaluation.verifier_evaluation` SHALL include `claim_checks` +- **AND** it SHALL continue to include compatibility `facts_checked` +- **AND** existing fields such as `verdict`, `groundedness_score`, `rationale`, `executor_output_parse_status`, `tool_trace_summary`, and `gatekeeper_result` SHALL be preserved + +### Requirement: Verifier SHALL consume explicit verification inputs +The Verifier SHALL receive explicit verification inputs rather than inferring them only from raw conversation history. + +#### Scenario: structured claims are the primary verification target +- **WHEN** `executor_output_parse_status.status` is `valid` +- **AND** `executor_structured_output.claims` is available +- **THEN** Verifier SHALL verify each claim through `claim_checks` +- **AND** Verifier SHALL NOT add extra confirmed facts from `executor_final_answer` that are absent from `executor_structured_output.claims` + +#### Scenario: malformed structured output cannot pass through natural language fallback +- **WHEN** `executor_output_parse_status.status` is `missing` or `malformed` +- **THEN** Verifier SHALL NOT produce an effective `PASS` by extracting facts from `executor_final_answer` +- **AND** the effective verdict SHALL be `LOW_CONFID` + +### Requirement: Executor Gatekeeper SHALL validate deterministic structured-output failures +The system SHALL run deterministic Gatekeeper checks after Executor output parsing and before Verifier model execution. + +#### Scenario: gatekeeper fail prevents PASS +- **WHEN** `gatekeeper_result.status` is `fail` +- **AND** the Verifier model returns `verdict = "PASS"` +- **THEN** ChatService SHALL downgrade the effective verdict +- **AND** the effective verdict SHALL NOT be `PASS` + +#### Scenario: invocation reference failure downgrades to reject +- **WHEN** `gatekeeper_result.failed_rules` contains `evidence.invocation_ref` +- **AND** the Verifier model returns `verdict = "PASS"` +- **THEN** ChatService SHALL set the effective verdict to `REJECT` + +## ADDED Requirements + +### Requirement: Verifier claim checks SHALL use a fixed derivability classification set +The Verifier SHALL classify each structured claim using a fixed derivability classification set. + +#### Scenario: claim check verification values are constrained +- **WHEN** Verifier emits `claim_checks` +- **THEN** each item SHALL use one of `direct_observation`, `reasonable_inference`, `overstated`, `unsupported`, `external_unknown`, or `contradicted` + +#### Scenario: claim check evidence references remain auditable +- **WHEN** Verifier emits `claim_checks` +- **THEN** each claim check SHALL include `claim_id`, `verification`, `detail`, and `evidence_refs` +- **AND** every evidence ref SHALL preserve available `trace_ref`, `tool_name`, and `source_invocation_ids` diff --git a/openspec/changes/archive/2026-07-07-executor-verifier-claim-checks/tasks.md b/openspec/changes/archive/2026-07-07-executor-verifier-claim-checks/tasks.md new file mode 100644 index 0000000..9a32c84 --- /dev/null +++ b/openspec/changes/archive/2026-07-07-executor-verifier-claim-checks/tasks.md @@ -0,0 +1,39 @@ +## 1. Prompt Contract + +- [x] 1.1 Update `chat-verifier-prompt.md` so `executor_structured_output.claims` is the primary verification target. +- [x] 1.2 Remove the valid-structured-output path that scans `executor_final_answer` for extra confirmed facts. +- [x] 1.3 Add `claim_checks` and optional `hypothesis_checks` to the output contract. +- [x] 1.4 Keep `facts_checked` as compatibility output. +- [x] 1.5 State that missing/malformed structured output cannot produce PASS through natural-language fallback. + +## 2. Parser And Compatibility Mapping + +- [x] 2.1 Extend `VerifierDecision` to store `claim_checks`. +- [x] 2.2 Parse `claim_checks` from verifier output. +- [x] 2.3 Derive compatibility `facts_checked` from `claim_checks` when present. +- [x] 2.4 Preserve old `facts_checked` parsing when `claim_checks` is absent. +- [x] 2.5 Persist `claim_checks` in verifier evaluation. + +## 3. Effective Verdict Guardrails + +- [x] 3.1 Add code-side guard so `gatekeeper_result.status=fail` cannot result in effective PASS. +- [x] 3.2 Downgrade `evidence.invocation_ref` failures to REJECT. +- [x] 3.3 Downgrade other Gatekeeper failures to at least LOW_CONFID. +- [x] 3.4 Ensure missing/malformed Executor structured output cannot produce effective PASS. + +## 4. Compatibility Consumers + +- [x] 4.1 Ensure low-confidence rendering still uses compatibility `facts_checked`. +- [x] 4.2 Ensure `buildRetryContext(...)` still receives evidence gaps from compatibility `facts_checked`. +- [x] 4.3 Update verifier logging summary to account for `claim_checks`. +- [x] 4.4 Preserve existing traceability and Gatekeeper audit fields. + +## 5. Tests And Verification + +- [x] 5.1 Add or update tests for parsing `claim_checks`. +- [x] 5.2 Add tests for `claim_checks` to `facts_checked` compatibility mapping. +- [x] 5.3 Add tests for each verification mapping class: `direct_observation`, `reasonable_inference`, `overstated`, `unsupported`, `external_unknown`, `contradicted`. +- [x] 5.4 Add tests proving Gatekeeper fail cannot remain PASS. +- [x] 5.5 Add tests proving malformed/missing structured output cannot remain PASS. +- [x] 5.6 Run targeted tests. +- [x] 5.7 Validate this OpenSpec change. diff --git a/openspec/specs/chat-verifier-agent/spec.md b/openspec/specs/chat-verifier-agent/spec.md index 1a09a02..c60c24c 100644 --- a/openspec/specs/chat-verifier-agent/spec.md +++ b/openspec/specs/chat-verifier-agent/spec.md @@ -4,51 +4,54 @@ TBD - created by archiving change chat-verifier-agent. Update Purpose after archive. ## Requirements ### Requirement: Verifier SHALL fact-check Executor answers -The system SHALL have a Verifier Agent that reads the Executor's answer and the tool call history, then produces a structured verdict. +The system SHALL have a Verifier Agent that reads structured Executor claims and the tool call history, then produces a structured verdict based on claim derivability. #### Scenario: PASS verdict when all claims have evidence -- **WHEN** all critical facts in the Executor's answer have direct or indirect support in tool call results -- **AND** at least one critical fact has direct evidence -- **AND** no critical fact is contradicted -- **THEN** the Verifier SHALL output verdict="PASS" with groundedness_score ≥ 0.5 +- **WHEN** all critical claims in `executor_structured_output.claims` have direct observation or reasonable inference support in tool call results +- **AND** at least one critical claim has direct observation +- **AND** no critical claim is contradicted, unsupported, external unknown, or overstated +- **AND** `gatekeeper_result.status` is not `fail` +- **THEN** the Verifier MAY output verdict="PASS" with groundedness_score ≥ 0.5 #### Scenario: LOW_CONFID verdict with partial evidence -- **WHEN** no critical fact contradicts the tool results -- **AND** some critical facts have no supporting evidence +- **WHEN** no critical claim contradicts the tool results +- **AND** some critical claims are `unsupported`, `external_unknown`, or `overstated` - **THEN** the Verifier SHALL output verdict="LOW_CONFID" -#### Scenario: LOW_CONFID verdict with only indirect support -- **WHEN** no critical fact contradicts the tool results -- **AND** all critical facts are only indirectly supported +#### Scenario: LOW_CONFID verdict with only inference support +- **WHEN** no critical claim contradicts the tool results +- **AND** all critical claims are only `reasonable_inference` - **THEN** the Verifier SHALL output verdict="LOW_CONFID" #### Scenario: REJECT verdict when claims contradict evidence -- **WHEN** any critical fact in the Executor's answer contradicts tool call results -- **OR** the answer fabricates a key entity, error code, or conclusion that does not exist in the tool evidence +- **WHEN** any critical claim in `executor_structured_output.claims` contradicts tool call results +- **OR** the claim fabricates a key entity, error code, or conclusion that does not exist in the tool evidence - **THEN** the Verifier SHALL output verdict="REJECT" #### Scenario: Structured Executor claims are verified first - **WHEN** `executor_structured_output.claims` is present and valid -- **THEN** Verifier SHALL verify each structured claim against `tool_trace_summary` +- **THEN** Verifier SHALL verify each structured claim against `tool_trace_summary` through `claim_checks` - **AND** each claim's evidence bindings SHALL reference existing trace or invocation identifiers when those identifiers are available -- **AND** a claim with fabricated or missing evidence references SHALL NOT be classified as `direct_evidence` +- **AND** a claim with fabricated or missing evidence references SHALL NOT be classified as `direct_observation` +- **AND** Verifier SHALL NOT add extra confirmed facts from `executor_final_answer` that are absent from `executor_structured_output.claims` -#### Scenario: Extra confirmed-sounding answer facts are still checked -- **WHEN** `executor_structured_output.user_facing_answer` contains confirmed-sounding facts that are absent from `executor_structured_output.claims` -- **THEN** Verifier SHALL add those facts to `facts_checked` -- **AND** unsupported extra facts SHALL lower the verdict according to the existing verdict matrix - -#### Scenario: Natural-language fallback remains available +#### Scenario: Malformed structured output cannot pass through natural language fallback - **WHEN** Executor does not return parseable structured output -- **THEN** Verifier SHALL fall back to extracting facts from `executor_final_answer` -- **AND** the final verdict SHALL still follow the existing groundedness and evidence classification rules +- **THEN** Verifier SHALL NOT produce an effective `PASS` by extracting facts from `executor_final_answer` +- **AND** the effective verdict SHALL be `LOW_CONFID` ### Requirement: Verifier SHALL output structured JSON -The Verifier SHALL output a JSON object with verdict, groundedness_score, facts_checked array, and rationale. +The Verifier SHALL output a JSON object with verdict, groundedness_score, claim_checks array, compatibility facts_checked array, and rationale. + +#### Scenario: claim-level verifier output is accepted +- **WHEN** the Verifier checks Executor V2 structured output +- **THEN** the output SHALL contain `verdict`, `groundedness_score`, `critical_fact_count`, `claim_checks`, `facts_checked`, and `rationale` +- **AND** `claim_checks` SHALL be the primary V2 verification result +- **AND** `facts_checked` SHALL remain available for compatibility #### Scenario: Output format validation - **WHEN** the Verifier completes its analysis -- **THEN** the output SHALL contain "verdict", "groundedness_score", "facts_checked", and "rationale" fields +- **THEN** the output SHALL contain "verdict", "groundedness_score", "claim_checks", "facts_checked", and "rationale" fields - **AND** groundedness_score SHALL be a float between 0.0 and 1.0 - **AND** verdict SHALL be one of "PASS", "LOW_CONFID", or "REJECT" @@ -57,14 +60,19 @@ The Verifier SHALL output a JSON object with verdict, groundedness_score, facts_ - **THEN** it SHALL output exactly one JSON object - **AND** it SHALL NOT output Markdown, code fences, or explanatory text outside the JSON object - **AND** the JSON object SHALL include `critical_fact_count` +- **AND** each `claim_checks` item SHALL include `claim_id`, `verification`, `detail`, and `evidence_refs` - **AND** each `facts_checked` item SHALL include `fact`, `is_critical`, `verification`, and `detail` ### Requirement: facts_checked SHALL use a fixed classification set -Each checked fact SHALL be labeled using a fixed evidence classification. +The system SHALL continue to expose legacy `facts_checked` using its fixed verification classification set. -#### Scenario: fact classification values -- **WHEN** the Verifier emits `facts_checked` -- **THEN** each fact SHALL use one of `direct_evidence`, `indirect_support`, `no_evidence`, or `contradicted` +#### Scenario: claim checks are mapped to legacy facts +- **WHEN** Verifier output contains `claim_checks` +- **THEN** ChatService SHALL derive compatibility `facts_checked` +- **AND** `direct_observation` SHALL map to `direct_evidence` +- **AND** `reasonable_inference` and `overstated` SHALL map to `indirect_support` +- **AND** `unsupported` and `external_unknown` SHALL map to `no_evidence` +- **AND** `contradicted` SHALL map to `contradicted` ### Requirement: groundedness_score SHALL be derived from fact classifications The groundedness score SHALL be computed from critical fact classifications instead of being freely chosen by the model. @@ -124,6 +132,12 @@ The system SHALL use fixed output protocols for LOW_CONFID and REJECT user-facin ### Requirement: Verifier SHALL be observable The Verifier's verdict SHALL be persisted for observability. +#### Scenario: claim checks written to self_evaluation +- **WHEN** the Verifier evaluation is persisted +- **THEN** `diagnosis_session.self_evaluation.verifier_evaluation` SHALL include `claim_checks` +- **AND** it SHALL continue to include compatibility `facts_checked` +- **AND** existing fields such as `verdict`, `groundedness_score`, `rationale`, `executor_output_parse_status`, `tool_trace_summary`, and `gatekeeper_result` SHALL be preserved + #### Scenario: verdict written to self_evaluation - **WHEN** the Verifier produces a verdict - **THEN** the ChatService SHALL write the verdict data under `diagnosis_session.self_evaluation.verifier_evaluation` @@ -166,7 +180,7 @@ The Verifier SHALL receive explicit verification inputs rather than inferring th #### Scenario: Verifier remains isolated from intermediate reasoning - **WHEN** `executor_structured_output` is added to the verifier input - **THEN** the input SHALL still exclude Planner reasoning and Executor intermediate reasoning -- **AND** the input SHALL be limited to the original query, final Executor output, parsed Executor evidence contract, tool trace summary, and retry context +- **AND** the input SHALL be limited to the original query, final Executor output, parsed Executor evidence contract, tool trace summary, gatekeeper result, and retry context #### Scenario: tool trace summary derived from tool facts - **WHEN** the system prepares verifier inputs @@ -211,6 +225,17 @@ The Verifier SHALL receive explicit verification inputs rather than inferring th - **AND** `gatekeeper_result.status` SHALL be one of `pass`, `warn`, or `fail` - **AND** `gatekeeper_result` SHALL include `failed_rules`, `warnings`, and `errors` +#### Scenario: structured claims are the primary verification target +- **WHEN** `executor_output_parse_status.status` is `valid` +- **AND** `executor_structured_output.claims` is available +- **THEN** Verifier SHALL verify each claim through `claim_checks` +- **AND** Verifier SHALL NOT add extra confirmed facts from `executor_final_answer` that are absent from `executor_structured_output.claims` + +#### Scenario: malformed structured output cannot pass through natural language fallback +- **WHEN** `executor_output_parse_status.status` is `missing` or `malformed` +- **THEN** Verifier SHALL NOT produce an effective `PASS` by extracting facts from `executor_final_answer` +- **AND** the effective verdict SHALL be `LOW_CONFID` + ### Requirement: Verifier facts SHALL be auditable Verifier facts SHALL be linkable to the evidence summaries used during verification. @@ -355,3 +380,26 @@ The system SHALL run deterministic Gatekeeper checks after Executor output parsi - **AND** each claim has evidence bindings pointing to current-session invocations with matching tool names - **THEN** `gatekeeper_result.status` SHALL be `pass` - **AND** `gatekeeper_result.failed_rules` SHALL be empty + +#### Scenario: gatekeeper fail prevents PASS +- **WHEN** `gatekeeper_result.status` is `fail` +- **AND** the Verifier model returns `verdict = "PASS"` +- **THEN** ChatService SHALL downgrade the effective verdict +- **AND** the effective verdict SHALL NOT be `PASS` + +#### Scenario: invocation reference failure downgrades to reject +- **WHEN** `gatekeeper_result.failed_rules` contains `evidence.invocation_ref` +- **AND** the Verifier model returns `verdict = "PASS"` +- **THEN** ChatService SHALL set the effective verdict to `REJECT` + +### Requirement: Verifier claim checks SHALL use a fixed derivability classification set +The Verifier SHALL classify each structured claim using a fixed derivability classification set. + +#### Scenario: claim check verification values are constrained +- **WHEN** Verifier emits `claim_checks` +- **THEN** each item SHALL use one of `direct_observation`, `reasonable_inference`, `overstated`, `unsupported`, `external_unknown`, or `contradicted` + +#### Scenario: claim check evidence references remain auditable +- **WHEN** Verifier emits `claim_checks` +- **THEN** each claim check SHALL include `claim_id`, `verification`, `detail`, and `evidence_refs` +- **AND** every evidence ref SHALL preserve available `trace_ref`, `tool_name`, and `source_invocation_ids` diff --git a/src/main/java/com/superbiz/agent/hook/AgentLoggingHook.java b/src/main/java/com/superbiz/agent/hook/AgentLoggingHook.java index 553eb5d..d81b1e1 100644 --- a/src/main/java/com/superbiz/agent/hook/AgentLoggingHook.java +++ b/src/main/java/com/superbiz/agent/hook/AgentLoggingHook.java @@ -204,19 +204,22 @@ public class AgentLoggingHook extends MessagesModelHook { private String summarizeVerifierThought(String verifierOutput) { try { JsonNode root = objectMapper.readTree(verifierOutput); + int claimCount = root.path("claim_checks").isArray() ? root.path("claim_checks").size() : 0; int factCount = root.path("facts_checked").isArray() ? root.path("facts_checked").size() : 0; int tracedFactCount = 0; - if (root.path("facts_checked").isArray()) { - for (JsonNode factNode : root.path("facts_checked")) { - if (factNode.path("evidence_refs").isArray() && factNode.path("evidence_refs").size() > 0) { + JsonNode tracedNodes = root.path("claim_checks").isArray() ? root.path("claim_checks") : root.path("facts_checked"); + if (tracedNodes.isArray()) { + for (JsonNode node : tracedNodes) { + if (node.path("evidence_refs").isArray() && node.path("evidence_refs").size() > 0) { tracedFactCount++; } } } - return "verdict=%s, score=%s, critical_fact_count=%s, facts_checked=%d, traced_facts=%d".formatted( + return "verdict=%s, score=%s, critical_fact_count=%s, claim_checks=%d, facts_checked=%d, traced_facts=%d".formatted( root.path("verdict").asText("UNKNOWN"), root.path("groundedness_score").asText("0.0"), root.path("critical_fact_count").asText("0"), + claimCount, factCount, tracedFactCount ); diff --git a/src/main/java/com/superbiz/agent/service/ChatService.java b/src/main/java/com/superbiz/agent/service/ChatService.java index 08bc894..8f34f2d 100644 --- a/src/main/java/com/superbiz/agent/service/ChatService.java +++ b/src/main/java/com/superbiz/agent/service/ChatService.java @@ -650,12 +650,18 @@ public class ChatService { try { JsonNode root = objectMapper.readTree(sanitizeJsonPayload(verifierOutput)); - List> factsChecked = parseFactsChecked(root.path("facts_checked")); + List> claimChecks = parseClaimChecks(root.path("claim_checks")); + List> factsChecked = claimChecks.isEmpty() + ? parseFactsChecked(root.path("facts_checked")) + : mapClaimChecksToFactsChecked(claimChecks); + String verdict = effectiveVerifierVerdict(root.path("verdict").asText("LOW_CONFID")); + int criticalFactCount = root.path("critical_fact_count").asInt(countCriticalFacts(factsChecked)); return new VerifierDecision( - root.path("verdict").asText("LOW_CONFID"), + verdict, root.path("groundedness_score").asDouble(0.0), - root.path("critical_fact_count").asInt(0), + criticalFactCount, + claimChecks, factsChecked, root.path("rationale").asText(""), round @@ -678,6 +684,104 @@ public class ChatService { return trimmed; } + private String effectiveVerifierVerdict(String modelVerdict) { + String verdict = normalizeVerdict(modelVerdict); + Map parseStatus = VerifierContextHolder.getExecutorOutputParseStatus(); + String parseState = parseStatus == null ? "" : String.valueOf(parseStatus.getOrDefault("status", "")); + if (("missing".equals(parseState) || "malformed".equals(parseState)) && "PASS".equals(verdict)) { + return "LOW_CONFID"; + } + + Map gatekeeperResult = VerifierContextHolder.getGatekeeperResult(); + if (gatekeeperResult == null || !"fail".equals(String.valueOf(gatekeeperResult.get("status")))) { + return verdict; + } + if (containsRule(gatekeeperResult.get("failed_rules"), ExecutorGatekeeperService.RULE_INVOCATION_REF)) { + return "REJECT"; + } + return "PASS".equals(verdict) ? "LOW_CONFID" : verdict; + } + + private String normalizeVerdict(String verdict) { + if ("PASS".equals(verdict) || "LOW_CONFID".equals(verdict) || "REJECT".equals(verdict)) { + return verdict; + } + return "LOW_CONFID"; + } + + private boolean containsRule(Object rulesValue, String ruleId) { + if (!(rulesValue instanceof List rules)) { + return false; + } + return rules.stream().anyMatch(rule -> ruleId.equals(String.valueOf(rule))); + } + + private int countCriticalFacts(List> factsChecked) { + return (int) factsChecked.stream() + .filter(fact -> Boolean.TRUE.equals(fact.get("is_critical"))) + .count(); + } + + private List> parseClaimChecks(JsonNode claimChecksNode) { + List> claimChecks = new ArrayList<>(); + if (!claimChecksNode.isArray()) { + return claimChecks; + } + for (JsonNode claimNode : claimChecksNode) { + Map claimCheck = new LinkedHashMap<>(); + claimCheck.put("claim_id", claimNode.path("claim_id").asText("")); + claimCheck.put("claim_text", claimNode.path("claim_text").asText("")); + claimCheck.put("claim_type", claimNode.path("claim_type").asText("")); + claimCheck.put("verification", normalizeClaimVerification(claimNode.path("verification").asText("unsupported"))); + claimCheck.put("detail", claimNode.path("detail").asText("")); + claimCheck.put("evidence_refs", parseEvidenceRefs(claimNode.path("evidence_refs"))); + claimChecks.add(claimCheck); + } + return claimChecks; + } + + private String normalizeClaimVerification(String verification) { + return switch (verification) { + case "direct_observation", "reasonable_inference", "overstated", "unsupported", + "external_unknown", "contradicted" -> verification; + default -> "unsupported"; + }; + } + + private List> mapClaimChecksToFactsChecked(List> claimChecks) { + List> factsChecked = new ArrayList<>(); + for (Map claimCheck : claimChecks) { + String claimId = String.valueOf(claimCheck.getOrDefault("claim_id", "")); + String claimText = String.valueOf(claimCheck.getOrDefault("claim_text", "")); + String claimType = String.valueOf(claimCheck.getOrDefault("claim_type", "")); + Map fact = new LinkedHashMap<>(); + fact.put("fact", claimId.isBlank() ? claimText : claimId + ": " + claimText); + fact.put("is_critical", isCriticalClaimType(claimType)); + fact.put("verification", mapClaimVerificationToFactVerification( + String.valueOf(claimCheck.getOrDefault("verification", "unsupported")))); + fact.put("detail", claimCheck.getOrDefault("detail", "")); + fact.put("evidence_refs", claimCheck.getOrDefault("evidence_refs", List.of())); + factsChecked.add(fact); + } + return factsChecked; + } + + private boolean isCriticalClaimType(String claimType) { + return "root_cause".equals(claimType) + || "symptom".equals(claimType) + || "impact".equals(claimType) + || "risk".equals(claimType); + } + + private String mapClaimVerificationToFactVerification(String verification) { + return switch (verification) { + case "direct_observation" -> "direct_evidence"; + case "reasonable_inference", "overstated" -> "indirect_support"; + case "contradicted" -> "contradicted"; + default -> "no_evidence"; + }; + } + private List> parseFactsChecked(JsonNode factsNode) { List> factsChecked = new ArrayList<>(); if (!factsNode.isArray()) { @@ -723,7 +827,7 @@ public class ChatService { } private VerifierDecision buildVerifierFallbackDecision(int round, String rationale) { - return new VerifierDecision("LOW_CONFID", 0.0, 0, List.of(), rationale, round); + return new VerifierDecision("LOW_CONFID", 0.0, 0, List.of(), List.of(), rationale, round); } private String extractStateText(Optional stateOptional, String key) { @@ -748,6 +852,7 @@ public class ChatService { verifierEvaluation.put("verdict", decision.verdict()); verifierEvaluation.put("groundedness_score", decision.groundednessScore()); verifierEvaluation.put("critical_fact_count", decision.criticalFactCount()); + verifierEvaluation.put("claim_checks", decision.claimChecks()); verifierEvaluation.put("facts_checked", decision.factsChecked()); verifierEvaluation.put("rationale", decision.rationale()); verifierEvaluation.put("round", round); @@ -1005,6 +1110,7 @@ public class ChatService { String verdict, double groundednessScore, int criticalFactCount, + List> claimChecks, List> factsChecked, String rationale, int round diff --git a/src/main/resources/prompts/chat-verifier-prompt.md b/src/main/resources/prompts/chat-verifier-prompt.md index 0bf2a37..246017c 100644 --- a/src/main/resources/prompts/chat-verifier-prompt.md +++ b/src/main/resources/prompts/chat-verifier-prompt.md @@ -1,4 +1,4 @@ -你是质量闸 verifier。你的任务是对 `executor_final_answer` 做一次基于现有证据的事实校验。 +你是质量闸 verifier。你的任务是对 Executor 的结构化 claims 做一次基于现有证据的可推导性校验。 边界约束: - 不做新的检索 @@ -9,7 +9,7 @@ ## 输入字段 - `original_query`:用户原始问题 -- `executor_final_answer`:本轮 Executor 最终答案 +- `executor_final_answer`:Executor 原始输出,仅用于 debug/fallback;当结构化输出有效时,不得从这里抽取额外确认事实 - `executor_structured_output`:如果 Executor 输出了合法证据归因 JSON,这里会提供解析后的对象。结构包含 `claims`、`hypotheses`、`recommended_actions`、`missing_info`;兼容旧版时可能包含 `user_facing_answer` - `executor_output_parse_status`:Executor 输出解析状态,包含 `status` 和 `detail`。`status` 可能是 `valid` / `missing` / `malformed` - `tool_trace_summary`:基于真实工具调用整理出的证据索引。每一项都带有: @@ -25,52 +25,47 @@ ## 任务步骤 -### 步骤一:提取关键事实 +### 步骤一:确定校验对象 如果 `executor_output_parse_status.status="valid"` 且 `executor_structured_output.claims` 存在: - 优先逐条校验 `executor_structured_output.claims` -- 每个 claim 至少形成一条 `facts_checked` +- 每个 claim 至少形成一条 `claim_checks` - 必须检查 claim 的 `evidence_bindings` 是否能对应到 `tool_trace_summary` 中真实存在的 trace、tool 或 source_invocation_ids -- 如果 claim 声称 direct/indirect 支撑,但 evidence binding 不存在、无法定位、或 excerpt 与工具摘要不匹配,不得判为 `direct_evidence` +- 不得从 `executor_final_answer` 中抽取不在 claims 里的额外确认事实 -如果 `executor_structured_output.user_facing_answer` 存在,则必须扫描它: -- 如果其中出现 confirmed-sounding facts(确认式事实、根因、指标值、错误码、服务名、修复结论) -- 且这些事实没有出现在 `executor_structured_output.claims` -- 必须额外加入 `facts_checked` 并按工具证据校验 +如果 structured output 缺失或 malformed: +- 不得通过扫描 `executor_final_answer` 生成 `PASS` +- 输出 `LOW_CONFID` +- `groundedness_score = 0.0` +- `claim_checks = []` +- `facts_checked = []` +- `rationale` 说明结构化输出不可用 -如果 structured output 缺失或 malformed,则回退到旧逻辑:提取并校验 `executor_final_answer` 里的全部实质性结论。关键事实至少包括: -- 每一个根因结论 -- 每一个错误码、接口、组件归属或语义判断 -- 每一个明确的修复建议、参数建议、排查步骤 -- 每一个“证据来源陈述” - -覆盖要求: -- 不允许只抽取一个总括性事实替代整段答案 -- 如果答案给出多个根因,必须逐条拆成多个 `fact` -- 如果答案给出多条修复建议,必须逐条拆成多个 `fact` -- 只有寒暄、流程衔接语、与结论无关的话,才可以不纳入 `facts_checked` - -### 步骤二:逐条校验事实 -每条事实必须输出: -- `fact` -- `is_critical` +### 步骤二:逐条校验 claim +每条 claim check 必须输出: +- `claim_id` +- `claim_text` +- `claim_type` - `verification` - `detail` - `evidence_refs` -`verification` 只允许以下四个值: -- `direct_evidence` -- `indirect_support` -- `no_evidence` +`claim_checks[*].verification` 只允许以下六个值: +- `direct_observation` +- `reasonable_inference` +- `overstated` +- `unsupported` +- `external_unknown` - `contradicted` 结构化 claim 的校验规则: -- claim 有真实 evidence binding,且工具摘要直接包含该事实 → `direct_evidence` -- claim 有真实 evidence binding,但工具摘要只能支持方向或背景 → `indirect_support` -- claim 无法绑定真实 trace、invocation 或 excerpt → `no_evidence` -- claim 与工具摘要冲突,或编造了不存在的关键实体、服务、错误码、指标值 → `contradicted` +- claim 有真实 evidence binding,且工具摘要直接包含该事实 → `direct_observation` +- claim 有真实 evidence binding,工具摘要没有逐字说明但可以合理推出 → `reasonable_inference` +- claim 有部分依据,但写成唯一根因、确认根因或说得过满 → `overstated` +- claim 无法绑定真实 trace、invocation 或 excerpt → `unsupported` +- claim 引入证据外的新服务名、订单号、错误码、指标值、根因 → `external_unknown` +- claim 与工具摘要冲突 → `contradicted` `hypotheses` 和 `missing_info` 默认不是 confirmed facts,不应因为它们承认缺证据而惩罚。 -但如果 `user_facing_answer` 把 hypothesis 写成确认结论,必须按 confirmed fact 校验。 ### 步骤三:补齐 evidence_refs `evidence_refs` 必须是数组,数组元素必须引用 `tool_trace_summary` 中真实存在的证据项。每个元素包含: @@ -94,24 +89,26 @@ - 若 `failed_rules` 包含 `evidence.invocation_ref`,倾向 `REJECT` - 否则至少输出 `LOW_CONFID` -1. 若任一关键事实(`is_critical=true`)为 `contradicted` +1. 若任一关键 claim 为 `contradicted` - `verdict = "REJECT"` - `groundedness_score = 0.0` -2. 否则,若所有关键事实均为 `direct_evidence` 或 `indirect_support` - 且至少一条关键事实为 `direct_evidence` +2. 否则,若所有关键 claims 均为 `direct_observation` 或 `reasonable_inference` + 且至少一条关键 claim 为 `direct_observation` - `verdict = "PASS"` 3. 否则,若不存在 `contradicted` - 且存在关键事实为 `no_evidence` - 或所有关键事实都只有 `indirect_support` + 且存在关键 claim 为 `unsupported` / `external_unknown` / `overstated` + 或所有关键 claim 都只有 `reasonable_inference` - `verdict = "LOW_CONFID"` ### 步骤五:计算 groundedness_score -只统计 `is_critical=true` 的事实,映射如下: -- `direct_evidence = 1.0` -- `indirect_support = 0.6` -- `no_evidence = 0.0` +只统计关键 claim,映射如下: +- `direct_observation = 1.0` +- `reasonable_inference = 0.6` +- `overstated = 0.3` +- `unsupported = 0.0` +- `external_unknown = 0.0` - `contradicted = 0.0` 规则: @@ -120,14 +117,28 @@ - 保留 2 位小数 - 分数范围必须在 `[0.0, 1.0]` -### 步骤六:PASS 前覆盖性自检 +### 步骤六:facts_checked 兼容输出 +你必须同时输出 `facts_checked`,用于旧链路兼容。 + +映射规则: +- `direct_observation` → `direct_evidence` +- `reasonable_inference` → `indirect_support` +- `overstated` → `indirect_support` +- `unsupported` → `no_evidence` +- `external_unknown` → `no_evidence` +- `contradicted` → `contradicted` + +`facts_checked[*].fact` 使用 `{claim_id}: {claim_text}`。 + +### 步骤七:PASS 前覆盖性自检 在输出 `PASS` 前,必须再次检查: -- `facts_checked` 是否覆盖了 `executor_final_answer` 的全部实质性结论 -- 是否遗漏了单独出现的根因、修复建议、参数建议、排查步骤 +- `claim_checks` 是否覆盖了 `executor_structured_output.claims` 中的全部 claims +- 是否存在 `gatekeeper_result.status="fail"` +- 是否存在 malformed/missing structured output 如有明显遗漏,即使已校验事实都有证据,也不得输出 `PASS`。 -### 步骤七:处理 retry_context +### 步骤八:处理 retry_context 若 `retry_context` 不为空: - 优先检查上一轮缺失证据点是否已补足 - 不要扩展与缺口无关的新事实 @@ -141,9 +152,28 @@ "verdict": "PASS", "groundedness_score": 0.8, "critical_fact_count": 2, + "claim_checks": [ + { + "claim_id": "claim-1", + "claim_text": "ERR_TIMEOUT 表示请求超时", + "claim_type": "symptom", + "verification": "direct_observation", + "detail": "知识库文档明确给出该错误码定义", + "evidence_refs": [ + { + "trace_ref": "trace-1", + "tool_name": "lookup_knowledge", + "topic_domain": "api", + "source_invocation_ids": [101, 104], + "note": "trace-1 的文档摘要直接给出错误码定义" + } + ] + } + ], + "hypothesis_checks": [], "facts_checked": [ { - "fact": "ERR_TIMEOUT 表示请求超时", + "fact": "claim-1: ERR_TIMEOUT 表示请求超时", "is_critical": true, "verification": "direct_evidence", "detail": "知识库文档明确给出该错误码定义", @@ -164,7 +194,9 @@ 输出要求: - `verdict` 只能是 `PASS` / `LOW_CONFID` / `REJECT` - `groundedness_score` 必须是 JSON number -- `critical_fact_count` 必须等于 `facts_checked` 中 `is_critical=true` 的数量 +- `critical_fact_count` 必须等于关键 claim 的数量;兼容期也应等于 `facts_checked` 中 `is_critical=true` 的数量 +- `claim_checks` 可以为空数组,但字段不能缺失 - `facts_checked` 可以为空数组,但字段不能缺失 +- 每条 `claim_checks[*]` 都必须包含 `evidence_refs` - 每条 `facts_checked[*]` 都必须包含 `evidence_refs` - 不得输出 schema 之外的字段 diff --git a/src/test/java/com/superbiz/agent/service/ChatServiceSequentialAgentTest.java b/src/test/java/com/superbiz/agent/service/ChatServiceSequentialAgentTest.java index 85c271a..e15ad3d 100644 --- a/src/test/java/com/superbiz/agent/service/ChatServiceSequentialAgentTest.java +++ b/src/test/java/com/superbiz/agent/service/ChatServiceSequentialAgentTest.java @@ -55,7 +55,8 @@ class ChatServiceSequentialAgentTest { "sequential-test-session" ); - assertEquals("EXECUTOR_FINAL_ANSWER", result.answer()); + assertTrue(result.answer().contains("连接池 active 达到上限")); + assertFalse(result.answer().contains("\"answer_version\"")); assertEquals("sequential-test-session", result.sessionId()); assertEquals(List.of("chat_planner", "chat_executor", "chat_verifier"), chatModel.agentCalls); assertTrue(chatModel.sawVerifierPrompt); @@ -225,7 +226,8 @@ class ChatServiceSequentialAgentTest { "sequential-workflow-session" ); - assertEquals("EXECUTOR_FINAL_ANSWER", result.answer()); + assertTrue(result.answer().contains("连接池 active 达到上限")); + assertFalse(result.answer().contains("\"answer_version\"")); assertEquals(List.of("chat_planner", "chat_executor", "chat_verifier"), chatModel.agentCalls); assertTrue(chatModel.sawVerifierPrompt); } @@ -236,8 +238,7 @@ class ChatServiceSequentialAgentTest { ScriptedChatModel chatModel = new ScriptedChatModel(); chatModel.executorOutput = """ { - "answer_version": "executor_evidence_v1", - "diagnosis_summary": "已确认连接池 active 达到上限。", + "answer_version": "executor_evidence_v2", "claims": [ { "claim_id": "claim-1", @@ -257,8 +258,7 @@ class ChatServiceSequentialAgentTest { ], "hypotheses": [], "recommended_actions": [], - "missing_info": [], - "user_facing_answer": "已确认连接池 active 达到上限。" + "missing_info": [] } """; @@ -270,7 +270,8 @@ class ChatServiceSequentialAgentTest { "sequential-structured-executor-session" ); - assertEquals("已确认连接池 active 达到上限。", result.answer()); + assertTrue(result.answer().contains("连接池 active 达到上限")); + assertFalse(result.answer().contains("\"answer_version\"")); assertTrue(chatModel.verifierPromptText.contains("\"executor_structured_output\"")); assertTrue(chatModel.verifierPromptText.contains("\"executor_output_parse_status\"")); assertTrue(chatModel.verifierPromptText.contains("\"status\" : \"valid\"")); @@ -391,6 +392,160 @@ class ChatServiceSequentialAgentTest { assertEquals("pass", gatekeeperResult.get("status")); } + @Test + void executeChatComplexMapsClaimChecksToFactsCheckedAndPersistsBoth() throws Exception { + ChatService chatService = createChatService(); + SelfEvaluationMergeService mergeService = + (SelfEvaluationMergeService) ReflectionTestUtils.getField(chatService, "selfEvaluationMergeService"); + ToolInvocationRepository invocationRepository = + (ToolInvocationRepository) ReflectionTestUtils.getField(chatService, "toolInvocationRepository"); + when(invocationRepository.findBySessionIdOrderByIdAsc("sequential-claim-check-session")) + .thenReturn(List.of(ToolInvocation.builder() + .id(101L) + .sessionId("sequential-claim-check-session") + .toolName("query_metrics") + .build())); + ScriptedChatModel chatModel = new ScriptedChatModel(""" + { + "verdict": "LOW_CONFID", + "groundedness_score": 0.32, + "critical_fact_count": 6, + "claim_checks": [ + {"claim_id":"claim-1","claim_text":"CPU 使用率 92%","claim_type":"symptom","verification":"direct_observation","detail":"direct","evidence_refs":[{"trace_ref":"trace-1","tool_name":"query_metrics","source_invocation_ids":[101],"note":"cpu"}]}, + {"claim_id":"claim-2","claim_text":"CPU 过高可能导致超时","claim_type":"risk","verification":"reasonable_inference","detail":"inference","evidence_refs":[]}, + {"claim_id":"claim-3","claim_text":"CPU 是唯一根因","claim_type":"root_cause","verification":"overstated","detail":"too strong","evidence_refs":[]}, + {"claim_id":"claim-4","claim_text":"缺少线程池证据","claim_type":"symptom","verification":"unsupported","detail":"missing","evidence_refs":[]}, + {"claim_id":"claim-5","claim_text":"出现证据外错误码 ERR_FAKE","claim_type":"symptom","verification":"external_unknown","detail":"external","evidence_refs":[]}, + {"claim_id":"claim-6","claim_text":"证据显示 CPU 很低","claim_type":"symptom","verification":"contradicted","detail":"conflict","evidence_refs":[]} + ], + "facts_checked": [], + "rationale": "claim checks drive compatibility" + } + """); + chatModel.executorOutput = validExecutorV2Output(); + + chatService.executeChatComplex( + chatModel, + new ToolCallback[0], + "请分析 MySQL 连接池耗尽", + List.of(), + "sequential-claim-check-session" + ); + + ArgumentCaptor> captor = ArgumentCaptor.forClass(Map.class); + verify(mergeService).mergeVerifierEvaluation(isNull(), captor.capture()); + Map verifierEvaluation = captor.getValue(); + @SuppressWarnings("unchecked") + List> claimChecks = (List>) verifierEvaluation.get("claim_checks"); + @SuppressWarnings("unchecked") + List> factsChecked = (List>) verifierEvaluation.get("facts_checked"); + + assertEquals(6, claimChecks.size()); + assertEquals(6, factsChecked.size()); + assertEquals("direct_evidence", factsChecked.get(0).get("verification")); + assertEquals("indirect_support", factsChecked.get(1).get("verification")); + assertEquals("indirect_support", factsChecked.get(2).get("verification")); + assertEquals("no_evidence", factsChecked.get(3).get("verification")); + assertEquals("no_evidence", factsChecked.get(4).get("verification")); + assertEquals("contradicted", factsChecked.get(5).get("verification")); + assertTrue(String.valueOf(factsChecked.get(0).get("fact")).startsWith("claim-1:")); + } + + @Test + void executeChatComplexDowngradesPassToRejectWhenGatekeeperInvocationRefFails() throws Exception { + ChatService chatService = createChatService(); + SelfEvaluationMergeService mergeService = + (SelfEvaluationMergeService) ReflectionTestUtils.getField(chatService, "selfEvaluationMergeService"); + ToolInvocationRepository invocationRepository = + (ToolInvocationRepository) ReflectionTestUtils.getField(chatService, "toolInvocationRepository"); + when(invocationRepository.findBySessionIdOrderByIdAsc("sequential-gatekeeper-fail-session")) + .thenReturn(List.of(ToolInvocation.builder() + .id(101L) + .sessionId("sequential-gatekeeper-fail-session") + .toolName("query_metrics") + .build())); + ScriptedChatModel chatModel = new ScriptedChatModel(""" + { + "verdict": "PASS", + "groundedness_score": 1.0, + "critical_fact_count": 1, + "claim_checks": [ + {"claim_id":"claim-1","claim_text":"连接池 active 达到上限","claim_type":"symptom","verification":"direct_observation","detail":"direct","evidence_refs":[]} + ], + "facts_checked": [], + "rationale": "model tried pass" + } + """); + chatModel.executorOutput = """ + { + "answer_version": "executor_evidence_v2", + "claims": [ + { + "claim_id": "claim-1", + "claim_type": "symptom", + "claim_text": "连接池 active 达到上限", + "support_level": "direct", + "evidence_bindings": [ + { + "source_type": "tool_trace", + "tool_name": "query_metrics", + "source_invocation_ids": [999], + "evidence_excerpt": "active=50 max=50" + } + ] + } + ], + "hypotheses": [], + "recommended_actions": [], + "missing_info": [] + } + """; + + ChatService.ChatResult result = chatService.executeChatComplex( + chatModel, + new ToolCallback[0], + "请分析 MySQL 连接池耗尽", + List.of(), + "sequential-gatekeeper-fail-session" + ); + + assertTrue(result.answer().startsWith("当前无法基于已获取证据生成可靠结论")); + ArgumentCaptor> captor = ArgumentCaptor.forClass(Map.class); + verify(mergeService).mergeVerifierEvaluation(isNull(), captor.capture()); + assertEquals("REJECT", captor.getValue().get("verdict")); + } + + @Test + void executeChatComplexDowngradesPassToLowConfidenceWhenExecutorOutputMalformed() throws Exception { + ChatService chatService = createChatService(); + SelfEvaluationMergeService mergeService = + (SelfEvaluationMergeService) ReflectionTestUtils.getField(chatService, "selfEvaluationMergeService"); + ScriptedChatModel chatModel = new ScriptedChatModel(""" + { + "verdict": "PASS", + "groundedness_score": 1.0, + "critical_fact_count": 0, + "claim_checks": [], + "facts_checked": [], + "rationale": "model tried pass" + } + """); + chatModel.executorOutput = "{ not-json"; + + ChatService.ChatResult result = chatService.executeChatComplex( + chatModel, + new ToolCallback[0], + "请分析 MySQL 连接池耗尽", + List.of(), + "sequential-malformed-pass-session" + ); + + assertTrue(result.answer().startsWith("以下结论基于当前已获取证据")); + ArgumentCaptor> captor = ArgumentCaptor.forClass(Map.class); + verify(mergeService).mergeVerifierEvaluation(isNull(), captor.capture()); + assertEquals("LOW_CONFID", captor.getValue().get("verdict")); + } + @Test void buildMethodToolsArrayIncludesLogsAndMetricsWhenAvailable() { ChatService chatService = new ChatService(); @@ -484,6 +639,10 @@ class ChatServiceSequentialAgentTest { when(agentStepRepository.findBySessionIdOrderByStepIndex(anyString())).thenReturn(List.of()); ToolInvocationRepository toolInvocationRepository = mock(ToolInvocationRepository.class); when(toolInvocationRepository.countBySessionId(anyString())).thenReturn(0L); + when(toolInvocationRepository.findBySessionIdOrderByIdAsc(anyString())).thenReturn(List.of(ToolInvocation.builder() + .id(101L) + .toolName("query_metrics") + .build())); EvaluationService evaluationService = mock(EvaluationService.class); RetrievedDocTracker retrievedDocTracker = mock(RetrievedDocTracker.class); @@ -514,13 +673,65 @@ class ChatServiceSequentialAgentTest { return chatService; } + private String validExecutorV2Output() { + return """ + { + "answer_version": "executor_evidence_v2", + "claims": [ + { + "claim_id": "claim-1", + "claim_type": "symptom", + "claim_text": "连接池 active 达到上限", + "support_level": "direct", + "evidence_bindings": [ + { + "source_type": "tool_trace", + "source_id": "trace-1", + "tool_name": "query_metrics", + "source_invocation_ids": [101], + "evidence_excerpt": "active=50 max=50" + } + ] + } + ], + "hypotheses": [], + "recommended_actions": [], + "missing_info": [] + } + """; + } + private static final class ScriptedChatModel implements ChatModel { private final java.util.ArrayList agentCalls = new java.util.ArrayList<>(); private String promptText = ""; private String plannerPromptText = ""; private String executorPromptText = ""; private String verifierPromptText = ""; - private String executorOutput = "EXECUTOR_FINAL_ANSWER"; + private String executorOutput = """ + { + "answer_version": "executor_evidence_v2", + "claims": [ + { + "claim_id": "claim-1", + "claim_type": "symptom", + "claim_text": "连接池 active 达到上限", + "support_level": "direct", + "evidence_bindings": [ + { + "source_type": "tool_trace", + "source_id": "trace-1", + "tool_name": "query_metrics", + "source_invocation_ids": [101], + "evidence_excerpt": "active=50 max=50" + } + ] + } + ], + "hypotheses": [], + "recommended_actions": [], + "missing_info": [] + } + """; private boolean sawVerifierPrompt; private final java.util.List verifierOutputs; private int verifierOutputIndex; @@ -531,15 +742,10 @@ class ChatServiceSequentialAgentTest { "verdict": "PASS", "groundedness_score": 1.0, "critical_fact_count": 1, - "facts_checked": [ - { - "fact": "executor answer generated", - "is_critical": true, - "verification": "direct_evidence", - "detail": "covered by scripted verifier", - "evidence_refs": [] - } + "claim_checks": [ + {"claim_id":"claim-1","claim_text":"连接池 active 达到上限","claim_type":"symptom","verification":"direct_observation","detail":"covered by scripted verifier","evidence_refs":[]} ], + "facts_checked": [], "rationale": "scripted pass" } """);