From 7b8c75e571823964ad1c28487e90c0b84094d100 Mon Sep 17 00:00:00 2001 From: aruo <40362743+zyongxin@users.noreply.github.com> Date: Wed, 8 Jul 2026 16:12:56 +0800 Subject: [PATCH] feat(agent): harden verifier evidence references --- devflow/index.md | 1 + .../acceptance.md | 52 + .../brief.md | 44 + .../decisions.md | 54 + .../evidence.md | 33 + ...-007-verifier-evidence-summary-fidelity.md | 1535 +++++++++++++++++ mvp/issues/README.md | 1 + .../.archive-ready | 0 .../.committed | 0 .../decisions.md | 73 + .../design.md | 180 ++ .../proposal.md | 62 + .../specs/chat-verifier-agent/spec.md | 110 ++ .../specs/evidence-trace-hardening/spec.md | 42 + .../tasks.md | 79 + openspec/specs/chat-verifier-agent/spec.md | 112 +- .../specs/evidence-trace-hardening/spec.md | 41 + .../agent/agent/tool/QueryLogsTools.java | 64 +- .../agent/hook/VerifierInputHook.java | 140 +- .../superbiz/agent/service/ChatService.java | 10 +- .../service/ExecutorGatekeeperService.java | 323 +++- .../agent/service/ToolInvocationRecorder.java | 135 ++ .../resources/prompts/chat-executor-prompt.md | 14 +- .../resources/prompts/chat-verifier-prompt.md | 17 +- .../agent/agent/tool/QueryLogsToolsTest.java | 48 + .../agent/hook/VerifierInputHookTest.java | 101 +- .../ChatServiceSequentialAgentTest.java | 27 +- .../ExecutorGatekeeperServiceTest.java | 151 +- .../service/ToolInvocationRecorderTest.java | 71 + 29 files changed, 3426 insertions(+), 94 deletions(-) create mode 100644 devflow/projects/2026-07-08-verifier-evidence-reference-fidelity/acceptance.md create mode 100644 devflow/projects/2026-07-08-verifier-evidence-reference-fidelity/brief.md create mode 100644 devflow/projects/2026-07-08-verifier-evidence-reference-fidelity/decisions.md create mode 100644 devflow/projects/2026-07-08-verifier-evidence-reference-fidelity/evidence.md create mode 100644 mvp/issues/ISS-007-verifier-evidence-summary-fidelity.md create mode 100644 openspec/changes/archive/2026-07-08-verifier-evidence-reference-fidelity/.archive-ready create mode 100644 openspec/changes/archive/2026-07-08-verifier-evidence-reference-fidelity/.committed create mode 100644 openspec/changes/archive/2026-07-08-verifier-evidence-reference-fidelity/decisions.md create mode 100644 openspec/changes/archive/2026-07-08-verifier-evidence-reference-fidelity/design.md create mode 100644 openspec/changes/archive/2026-07-08-verifier-evidence-reference-fidelity/proposal.md create mode 100644 openspec/changes/archive/2026-07-08-verifier-evidence-reference-fidelity/specs/chat-verifier-agent/spec.md create mode 100644 openspec/changes/archive/2026-07-08-verifier-evidence-reference-fidelity/specs/evidence-trace-hardening/spec.md create mode 100644 openspec/changes/archive/2026-07-08-verifier-evidence-reference-fidelity/tasks.md create mode 100644 src/test/java/com/superbiz/agent/agent/tool/QueryLogsToolsTest.java diff --git a/devflow/index.md b/devflow/index.md index 5d3a913..076cb63 100644 --- a/devflow/index.md +++ b/devflow/index.md @@ -10,6 +10,7 @@ | 2026-07-07 | executor-gatekeeper-hook | Chat质量门禁/证据归因 | Gatekeeper, verifier payload, source_invocation_ids, tool_name match, self_evaluation | openspec/changes/archive/2026-07-07-executor-gatekeeper-hook | archived | | 2026-07-07 | executor-verifier-claim-checks | Chat质量门禁/证据归因 | Verifier claim_checks, facts_checked compatibility, effective verdict guardrail, malformed output downgrade | openspec/changes/archive/2026-07-07-executor-verifier-claim-checks | archived | | 2026-07-08 | executor-composer-final-answer | Chat quality gate/evidence attribution | chat_composer, final answer, allowed_claims, allowed_hypotheses, safe fallback, composer_output | openspec/changes/archive/2026-07-08-executor-composer-final-answer | archived | +| 2026-07-08 | verifier-evidence-reference-fidelity | Chat质量门禁/证据归因 | evidence_refs, raw_path, Gatekeeper severity, verifier evidence excerpt, HikariCP mock, no_evidence | openspec/changes/archive/2026-07-08-verifier-evidence-reference-fidelity | archived | | 2026-07-06 | rag-eval-pipeline-closure | RAG/评测/回归闭环 | lookupResult fixture, LookupKnowledgeTool snapshot, evidenceBlocks, contextPack, retrievalTrace, rerankTrace, baseline diff, fallback case | devflow/projects/2026-07-06-rag-eval-pipeline-closure | archived | | 2026-07-06 | modular-rag-pipeline | RAG/Agent工具/证据链 | modular RAG, lookup_knowledge, evidenceBlocks, contextPack, rerank, retrievalTrace, L0 hint, unfiltered retry | openspec/changes/archive/2026-07-06-modular-rag-pipeline | archived | | 2026-07-05 | mvp-demo-interview-runbook | MVP Demo/Interview | Plan C, payment timeout, runbook, trace checklist, demo script | openspec/changes/archive/2026-07-05-mvp-demo-interview-runbook | archived | diff --git a/devflow/projects/2026-07-08-verifier-evidence-reference-fidelity/acceptance.md b/devflow/projects/2026-07-08-verifier-evidence-reference-fidelity/acceptance.md new file mode 100644 index 0000000..b0d67c5 --- /dev/null +++ b/devflow/projects/2026-07-08-verifier-evidence-reference-fidelity/acceptance.md @@ -0,0 +1,52 @@ +# Acceptance + +## Static Verification + +- `openspec validate verifier-evidence-reference-fidelity --strict`: passed. +- `openspec validate --specs`: passed. + +## Script Verification + +- `mvn "-Dtest=ToolInvocationRecorderTest,ExecutorGatekeeperServiceTest,VerifierInputHookTest,QueryLogsToolsTest,ChatServiceSequentialAgentTest" test` + - Passed: 38 tests. +- `mvn "-Dtest=ExecutorGatekeeperServiceTest" test` + - Passed: 9 tests. +- `mvn test` + - Failed on unrelated environment-gated `MilvusConnectionTest.connect`: `MILVUS_TOKEN` environment variable was not set. + - Other executed tests in the run progressed until that single failure; focused tests for this change passed. + +## End-to-End Verification + +The Java service was restarted with `mvn spring-boot:run`; logs were written under `logs/`. + +| Case | Session | Result | Gatekeeper Audit | +|---|---|---|---| +| HikariCP positive | `iss007-hikari-positive-20260708-1553` | PASS; confirmed order-service HikariCP timeout and pool saturation logs | `pass / none`, checked_bindings=2 | +| HikariCP negative | `iss007-hikari-negative-20260708-1555` | LOW_CONFID; no `generic-service`; no false positive for inventory-service | `fail / low_confid` | +| HighMemoryUsage positive | `iss007-memory-positive-20260708-1558` | PASS; confirmed HighMemoryUsage 91%, did not confirm memory leak | `pass / none`, checked_bindings=1 | +| SlowResponse positive | `iss007-slow-positive-20260708-1600` | PASS; confirmed SlowResponse and slow request logs, no DB pool root cause | `pass / none`, checked_bindings=7 | +| Narrow HighCPUUsage | `iss007-narrow-highcpu-20260708-1602` | PASS; only covered payment-service HighCPUUsage | `pass / none`, checked_bindings=1 | + +## Database Audit + +`scripts/query_mysql.py` was used to verify: + +- `diagnosis_session.self_evaluation.verifier_evaluation.verdict` +- `gatekeeper_result.status` +- `gatekeeper_result.severity` +- `gatekeeper_result.checked_bindings` +- no-hit HikariCP query rows persist `evidence_status=no_evidence` + +## Remaining Risk + +- Negative no-hit claims still have incomplete precise references when Executor uses `$.logs` for empty arrays. Gatekeeper correctly downgrades to `LOW_CONFID`. +- Prompt-only scope control is improved but not a hard contract. A future `scope_contract` may still be needed. +- Full test suite requires `MILVUS_TOKEN` to pass `MilvusConnectionTest`. + +## OpenSpec Archive + +- `openspec archive verifier-evidence-reference-fidelity --yes`: succeeded. +- Main specs updated: + - `openspec/specs/chat-verifier-agent/spec.md` + - `openspec/specs/evidence-trace-hardening/spec.md` +- Non-blocking warning: proposal did not use OpenSpec's preferred `## Why` / `## What Changes` headers, but archive completed. diff --git a/devflow/projects/2026-07-08-verifier-evidence-reference-fidelity/brief.md b/devflow/projects/2026-07-08-verifier-evidence-reference-fidelity/brief.md new file mode 100644 index 0000000..730e8bb --- /dev/null +++ b/devflow/projects/2026-07-08-verifier-evidence-reference-fidelity/brief.md @@ -0,0 +1,44 @@ +# Verifier Evidence Reference Fidelity + +## Background + +ISS-007 came from end-to-end diagnosis cases where raw tool output and Executor `evidence_excerpt` contained enough facts, but Verifier still returned `LOW_CONFID` because the verifier-facing summary compressed away key details. + +The affected flow is: + +```text +chat_planner + -> chat_executor + -> VerifierInputHook / Gatekeeper + -> chat_verifier + -> chat_composer +``` + +The change hardens the evidence handoff between Executor, Gatekeeper, and Verifier. + +## Goal + +Make Executor cite concrete tool evidence, make Gatekeeper validate that citation with code, and make Verifier judge whether verified evidence can derive the claim. + +## Scope + +- Persist `tool_invocation.retrieval_details.evidence_refs`. +- Use `source_invocation_id + raw_path + evidence_excerpt` as the precise evidence binding. +- Add Gatekeeper `severity` and checked binding audit. +- Keep Gatekeeper in the Verifier hook path. +- Keep `tool_trace_summary` as navigation/audit context, not the only evidence source. +- Fix HikariCP mock positive/no-hit behavior. +- Tighten Executor/Verifier prompts for narrow-scope evidence handling. + +## Non-goals + +- No Planner `scope_contract`. +- No new database table. +- No full JSONPath engine. +- No change to external HTTP API. +- No retry rollback from Gatekeeper to Executor in this phase. + +## OpenSpec + +- Change: `openspec/changes/verifier-evidence-reference-fidelity` +- Source issue: `mvp/issues/ISS-007-verifier-evidence-summary-fidelity.md` diff --git a/devflow/projects/2026-07-08-verifier-evidence-reference-fidelity/decisions.md b/devflow/projects/2026-07-08-verifier-evidence-reference-fidelity/decisions.md new file mode 100644 index 0000000..65e530a --- /dev/null +++ b/devflow/projects/2026-07-08-verifier-evidence-reference-fidelity/decisions.md @@ -0,0 +1,54 @@ +# Decisions + +## Evidence Reference + +Use `source_invocation_id + raw_path + evidence_excerpt` as the precise evidence reference for Executor claim bindings. + +Reason: + +- Invocation ID alone only identifies a tool call, not the evidence inside it. +- `raw_path` is enough for the first version when paired with `retrieval_details.evidence_refs`. +- `evidence_excerpt` remains the text Verifier reads, but only after Gatekeeper validates it. + +## Raw Path + +Only support stable locators in the first version: + +- `$.alerts[i]` +- `$.logs[i]` +- `$.evidence_blocks[i]` + +No full JSONPath engine is introduced. + +## Gatekeeper Severity + +Gatekeeper output includes: + +- `status` +- `severity` +- `checked_bindings` +- `failed_rules` +- `warnings` +- `errors` + +Severity meaning: + +- `none`: precise references passed. +- `low_confid`: evidence is missing or incomplete, but not fabricated. +- `reject`: fabricated ID, wrong tool, unknown raw path, or mismatched excerpt. + +## Verifier Boundary + +Verifier uses verified claim-local excerpts as primary derivability evidence. `tool_trace_summary` remains available for navigation and audit, but no longer needs to carry every concrete fact. + +## Hook Placement + +Gatekeeper remains in the Verifier input hook path. This version does not retry Executor on Gatekeeper failure. + +## Planner + +Planner is not changed. `scope_contract` remains a later-stage idea. This phase uses prompt constraints to reduce narrow-scope over-expansion. + +## Database + +No new tables. Evidence refs are stored in `tool_invocation.retrieval_details.evidence_refs`; audit is stored in `diagnosis_session.self_evaluation.verifier_evaluation.gatekeeper_result`. diff --git a/devflow/projects/2026-07-08-verifier-evidence-reference-fidelity/evidence.md b/devflow/projects/2026-07-08-verifier-evidence-reference-fidelity/evidence.md new file mode 100644 index 0000000..6d1492b --- /dev/null +++ b/devflow/projects/2026-07-08-verifier-evidence-reference-fidelity/evidence.md @@ -0,0 +1,33 @@ +# Evidence + +## Existing Context + +- Existing `chat-verifier-agent` spec still used `source_invocation_ids` and `tool_trace_summary` as the main verifier evidence context. +- Existing `evidence-trace-hardening` spec already established `tool_invocation.retrieval_details` as the right place for structured tool-specific facts. +- Prior devflow projects established that runbook/skill content is guidance, not incident evidence. + +## Code Findings + +- `VerifierInputHook` previously backfilled plural `source_invocation_ids` from `tool_trace_summary` by tool name. +- `ExecutorGatekeeperService` previously validated invocation existence and tool name, but not `raw_path` or excerpt authenticity. +- `ToolInvocationRecorder` persisted retrieval details but did not generate claim-addressable `evidence_refs`. +- `QueryLogsTools` could fall back to `generic-service` placeholder logs on no-hit. + +## Implementation Evidence + +- `ToolInvocationRecorder` now extracts: + - `$.alerts[i]` for `query_metrics` + - `$.logs[i]` for `query_logs` + - `$.evidence_blocks[i]` for `lookup_knowledge` +- `ExecutorGatekeeperService` now validates: + - invocation existence + - tool name + - raw path presence + - `retrieval_details.evidence_refs` + - excerpt similarity/support +- `VerifierInputHook` only auto-fills a singular `source_invocation_id` when exactly one candidate exists and never invents `raw_path`. +- `QueryLogsTools` returns HikariCP mock logs for `order-service` and returns empty no-hit results for unrelated services. + +## Residual Finding + +The HikariCP negative E2E no longer has generic-service pollution, but the model still issued an extra broad HikariCP query without the service filter and used order-service as context. This is a remaining narrow-scope behavior issue, not a mock evidence pollution issue. diff --git a/mvp/issues/ISS-007-verifier-evidence-summary-fidelity.md b/mvp/issues/ISS-007-verifier-evidence-summary-fidelity.md new file mode 100644 index 0000000..eb6285c --- /dev/null +++ b/mvp/issues/ISS-007-verifier-evidence-summary-fidelity.md @@ -0,0 +1,1535 @@ +# ISS-007 Verifier 证据摘要保真与工具命中质量问题 + +**状态**:设计已收口,待实施 +**严重程度**:高 +**发现时间**:2026-07-08 +**来源**:Executor V2 / Gatekeeper / Verifier / Composer 端到端验证 +**关联**: +- `executor-structured-output-v2` +- `executor-evidence-attribution-hallucination` +- `ISS-005-evidence-trace-hardening` +- `ISS-006-diagnosis-eval-harness` + +--- + +## 背景 + +当前复杂诊断链路已经形成: + +```text +chat_planner + -> chat_executor + -> VerifierInputHook 内 Gatekeeper + -> chat_verifier + -> chat_composer + -> final answer +``` + +最近端到端验证中,部分窄范围问题本应具备足够证据,但最终仍被 Verifier 判为 `LOW_CONFID`。进一步查看审计数据后发现,这几类失败并不完全是 Executor 幻觉,也不完全是 Verifier 代码 bug,而是暴露出一组更经典的工程问题: + +1. 工具原始返回和 Executor `evidence_excerpt` 中存在证据,但 `tool_trace_summary.output_summary` 压缩后丢失关键证据。 +2. Verifier 当前更依赖工具摘要,而不是直接消费 Executor 绑定的原文片段。 +3. mock 工具命中规则存在质量问题,某些查询返回了 `generic-service` 占位日志,导致负向结论虽然被正确识别,但无法验证正向路径。 +4. Executor 对窄范围问题仍可能输出过多扩展 claim,增加 Verifier 判定压力。 + +这个 issue 的目标不是立即确定实现方案,而是把这些问题归档为一组可讨论、可优化、可验收的工程问题。 + +--- + +## 复现样例 + +### 1. HighCPUUsage:正向 PASS + +**Session**:`e2e-pass-highcpu-raw-20260708-1106` + +**问题范围**:只确认 `payment-service` 当前是否存在 CPU 使用率过高告警。 + +**结果**: +- `verdict=PASS` +- `groundedness_score=1.0` +- `gatekeeper_result.status=pass` +- `composer_output.status=valid` + +**结论**: +- 工具 mock、日志、知识库 runbook 三者可以匹配。 +- 这是当前链路能够通过的健康样例,可作为后续回归基线。 + +--- + +### 2. HighMemoryUsage:证据存在但 LOW_CONFID + +**Session**:`e2e-pass-memory-20260708-1117-a1` + +**结果**: +- `verdict=LOW_CONFID` +- `groundedness_score=0.2` +- `gatekeeper_result.status=pass` + +**审计发现**: +- Executor 的 `evidence_excerpt` 中包含完整证据: + - `HighMemoryUsage` firing + - 当前内存使用率 `91%` + - JVM 堆内存 `3.8GB/4GB` + - 持续 `15m` + - 多个采样点:`86.2 -> 87.4 -> 88.6 -> 89.8 -> 91.0` + - Full GC `15` 次,平均 `850ms` +- 但 `tool_trace_summary.output_summary` 只保留了一条内存日志: + - `[11:17][WARN][order-service] 内存使用率过高: 91.0%, JVM堆内存: 3.8GB/4GB, GC次数: 128` + +**初步归类**: +- 主要问题:`ToolTraceSummaryService` 摘要压缩丢失趋势和 GC 证据。 +- 次要问题:Verifier 未直接使用 Executor 的 `evidence_excerpt` 做可推导性判断。 +- 不是主要幻觉问题:Executor 引用的片段里确实包含关键事实。 + +--- + +### 3. SlowResponse:证据存在但 LOW_CONFID + +**Session**:`e2e-pass-slowresponse-20260708-1117-a1` + +**结果**: +- `verdict=LOW_CONFID` +- `groundedness_score=0.2` +- `gatekeeper_result.status=pass` + +**审计发现**: +- Executor 的 `evidence_excerpt` 中包含: + - `SlowResponse` alert:`user-service P99=4.2s`,`firing`,持续 `10m` + - 慢请求日志: + - `/api/v1/users/profile`: `3600 / 3900 / 4200ms` + - `/api/v1/users/orders`: `3750 / 4050ms` +- 但 `tool_trace_summary.output_summary` 只保留了一条 `/api/v1/users/profile 4200ms`。 +- `query_metrics` summary 没有完整展示 `SlowResponse` firing alert。 + +**初步归类**: +- 主要问题:工具摘要没有保留足够多的关键事件和指标。 +- 次要问题:Verifier 过度依赖 summary,导致它看不到 Executor 已经绑定的证据。 +- 不是主要幻觉问题:证据片段本身存在。 + +--- + +### 4. HikariCP:负向 PASS,但正向路径缺失 + +**Session**:`e2e-pass-hikari-20260708-1117-a1` + +**结果**: +- `verdict=PASS` +- `groundedness_score=1.0` +- `gatekeeper_result.status=pass` + +**实际结论**: + +```text +application-logs 中未检索到 order-service 的真实 HikariCP 连接池耗尽日志 +``` + +**审计发现**: +- mock 工具返回的是 `generic-service` 占位日志。 +- `evidence_level=none` +- `no_hit_invocation_count=6` +- Verifier 正确识别“没查到真实 HikariCP 日志”。 + +**初步归类**: +- Verifier 的负向 PASS 是合理的。 +- 真正问题在 mock 工具命中质量:`HikariCP`、`active=50/50`、`order-service` 等查询没有命中正向 mock 分支。 + +--- + +## 问题归类 + +| 问题 | 类型 | 严重程度 | 当前判断 | +|---|---|---:|---| +| `tool_trace_summary.output_summary` 只保留少量日志,丢失趋势、采样点、关键 alert | 工具摘要质量 | 高 | 需要优化 | +| Verifier 主要依赖 summary,未充分利用 Executor `evidence_bindings[].evidence_excerpt` | 架构 / 输入设计 | 高 | 需要讨论 | +| `query_metrics` summary 没有完整展示所有 firing alert | 工具摘要质量 | 高 | 需要优化 | +| Executor 对窄范围问题输出多条扩展 claim | Agent 约束 | 中 | 需要讨论 | +| mock 日志返回 `generic-service` 占位日志 | 工具 / mock 质量 | 中 | 需要优化 | +| Gatekeeper pass 但 Verifier LOW_CONFID | 分层行为 | 低 | 不是问题,属于正常分工 | + +--- + +## 当前结论 + +### Memory / SlowResponse + +主要不是 LLM 编造事实,也不是 Gatekeeper 漏拦,而是证据在链路中被摘要层压缩丢失。 + +当前证据链路存在一个断点: + +```text +tool raw output / executor evidence_excerpt 有证据 + -> tool_trace_summary.output_summary 丢失部分证据 + -> verifier 看不到足够材料 + -> LOW_CONFID +``` + +### HikariCP + +当前负向 PASS 是合理的,因为系统确实没有检索到真实 `order-service` HikariCP 连接池耗尽证据。 + +真正需要优化的是工具 mock 的命中规则和 fixture 数据质量,否则后续很难验证 HikariCP 的正向路径。 + +--- + +## 已确认设计方向:Executor 引用证据,Gatekeeper 核对引用 + +针对“摘要失真”问题,优先采用以下主线,而不是继续让 Executor 或 `ToolTraceSummaryService` 承担更重的自然语言压缩职责: + +```text +tool raw output + -> Executor 产出 claim + 引用证据片段 + -> Gatekeeper 用代码核对引用是否真实 + -> Verifier 判断 claim 是否能由已核对证据推出 + -> Composer 只表达 Verifier 允许输出的内容 +``` + +### 分工边界 + +#### Executor:证据引用器,而不是可信摘要器 + +Executor 不再把工具结果压缩成一段“可信诊断摘要”后交给 Verifier 判断,而是负责: + +- 产出结构化 claim。 +- 为每个 claim / recommended action 绑定证据引用。 +- 引用真实 `source_invocation_id + raw_path`。 +- 给出与 claim 直接相关的 `evidence_excerpt`。 +- 不编造 invocation id。 +- 不把 runbook 通用经验升级成当前已确认事实。 + +目标是让 Executor “少总结,多引用”。 + +#### Gatekeeper:引用真实性校验 + +Gatekeeper 负责用代码检查 Executor 的引用是否真实,先拦截物理级幻觉: + +- `source_invocation_id` 是否真实存在于本轮 session 的 tool invocation 历史中。 +- `source_invocation_id` 是否属于允许被引用的 evidence tool。 +- `evidence_excerpt` 是否能在对应 tool invocation 的原始输出中找到,或达到可接受的相似度。 +- 是否存在“真实 invocation id + 编造 excerpt”的张冠李戴问题。 +- 是否存在空证据、伪造证据、引用无关工具输出等问题。 + +Gatekeeper 只判断“引用是否真实”,不判断“claim 是否成立”。 + +#### Verifier:判断可推导性 + +Verifier 在 Gatekeeper 通过后,再判断: + +- 这些已核对的证据是否足以支持 claim。 +- claim 是否存在过度推断。 +- claim 是否只是合理怀疑,而不是已确认事实。 +- recommended action 是否能由当前证据和缺口合理推出。 + +Verifier 不再把主要精力放在逐字核对引用真伪上,而是做“证据能否推出结论”的判断。 + +#### ToolTraceSummary:全局导航摘要,不再作为唯一证据源 + +`tool_trace_summary.output_summary` 仍然保留,但定位调整为: + +- 帮助 Verifier / 审计理解本轮工具调用全貌。 +- 提供全局 evidence overview。 +- 作为辅助证据视图,而不是唯一判定依据。 + +也就是说,summary 可以继续优化,但不应该再承担“唯一证据源”的职责。 + +### 这个方向解决的问题 + +该设计直接针对当前 Memory / SlowResponse 的失败模式: + +```text +工具原始返回和 Executor evidence_excerpt 有证据 + -> summary 压缩时丢失关键证据 + -> Verifier 看不到足够材料 + -> LOW_CONFID +``` + +调整后变为: + +```text +工具原始返回有证据 + -> Executor 引用关键片段 + -> Gatekeeper 校验片段来自真实工具输出 + -> Verifier 基于已核对片段判断 claim 是否可推出 +``` + +这样即使 `output_summary` 较短,Verifier 仍能看到 claim-local evidence。 + +### 设计约束 + +- Verifier 不能无条件相信 Executor 的 `evidence_excerpt`。 +- Gatekeeper 的 excerpt 原文回溯是 Verifier 消费 excerpt 的前置安全条件。 +- Gatekeeper 校验通过不代表 claim 通过,只代表“引用是真的”。 +- Verifier 判为 `PASS` 必须基于“真实引用 + 可推导结论”两个条件同时成立。 +- 不能通过简单放宽 Verifier 阈值来掩盖证据传递问题。 + +--- + +## Verifier 输入数据结构定义(当前设计) + +本节定义“Executor 引用证据,Gatekeeper 核对引用”方案下,进入 Verifier 的数据结构。核心原则: + +```text +证据定位:source_invocation_id + raw_path +证据文本:evidence_excerpt +证据真实性:由 Gatekeeper 在 Verifier 前校验 +Verifier 职责:只判断 claim_text 是否能由已核验 evidence_excerpt 推出 +``` + +### 1. 工具调用侧最小证据引用 + +当前工具尚未统一返回 `evidence_items`。第一版不引入复杂 metadata,也不要求新增数据库表,建议在 `tool_invocation.retrieval_details` 中补充最小 `evidence_refs`: + +```json +{ + "evidence_status": "supported", + "evidence_refs": [ + { + "raw_path": "$.alerts[1]", + "text": "HighMemoryUsage firing, service=order-service, current=91%, duration=15m; JVM heap usage 3.8GB/4GB" + } + ] +} +``` + +字段定义: + +| 字段 | 类型 | 必填 | 说明 | +|---|---|---:|---| +| `raw_path` | string | 是 | 证据在该工具返回中的稳定定位符。第一版只支持 `$.alerts[i]`、`$.logs[i]`、`$.evidence_blocks[i]`。 | +| `text` | string | 是 | 系统从该位置提取出的最小证据文本。Gatekeeper 用它比对 Executor 的 `evidence_excerpt`,Verifier 用 `evidence_excerpt` 做推导判断。 | + +设计约束: + +- 第一版不需要 `metadata`。 +- 第一版不需要 `evidence_id`。 +- `raw_path` 只在单次 `tool_invocation` 内有意义,必须和 `source_invocation_id` 搭配使用。 +- 如果工具返回会被 `output_preview` 截断,`evidence_refs[].text` 必须保存在 `retrieval_details` 中,不能只依赖 `output_preview`。 + +### 2. Executor 结构化输出 + +Executor 输出不再把证据压缩成可信摘要,而是输出 claim 与证据引用: + +```json +{ + "claims": [ + { + "claim_id": "claim-1", + "claim_type": "observation", + "claim_text": "order-service 当前存在 HighMemoryUsage 告警,内存使用率达到 91%。", + "evidence_bindings": [ + { + "tool_name": "query_metrics", + "source_invocation_id": 12345, + "raw_path": "$.alerts[1]", + "evidence_excerpt": "HighMemoryUsage firing, service=order-service, current=91%, duration=15m" + } + ] + } + ], + "hypotheses": [ + { + "hypothesis_id": "hyp-1", + "hypothesis_text": "order-service 可能存在内存泄漏风险。", + "basis_claim_ids": ["claim-1"], + "missing_info": "缺少堆 dump、对象分配统计或更长时间窗口的内存曲线,不能确认内存泄漏。" + } + ], + "missing_info": [ + { + "info_id": "missing-1", + "description": "缺少堆 dump 或对象分配统计,无法确认内存泄漏根因。" + } + ], + "recommended_actions": [ + { + "action_id": "action-1", + "action_text": "继续查看 order-service 的 GC 日志、堆 dump 或对象分配统计。", + "reason": "当前证据可以确认内存使用率过高,但不足以确认是否存在内存泄漏。", + "evidence_bindings": [ + { + "tool_name": "query_metrics", + "source_invocation_id": 12345, + "raw_path": "$.alerts[1]", + "evidence_excerpt": "HighMemoryUsage firing, service=order-service, current=91%, duration=15m" + } + ] + } + ] +} +``` + +字段定义: + +| 字段 | 类型 | 必填 | 说明 | +|---|---|---:|---| +| `claims` | array | 是 | Executor 提出的待验证事实断言。 | +| `claims[].claim_id` | string | 是 | claim 唯一标识,供 Verifier / Composer 引用。 | +| `claims[].claim_type` | string | 是 | claim 类型,例如 `observation`、`root_cause`、`risk`、`negative_observation`。 | +| `claims[].claim_text` | string | 是 | Executor 提出的事实断言,不是证据本体,也不是已验证结论。 | +| `claims[].evidence_bindings` | array | 是 | 支撑该 claim 的证据引用列表。 | +| `evidence_bindings[].tool_name` | string | 是 | 证据来源工具,例如 `query_metrics`、`query_logs`、`lookup_knowledge`。 | +| `evidence_bindings[].source_invocation_id` | number | 是 | 单次真实工具调用 ID。第一版使用单数;如果一个 claim 需要多个工具调用,拆成多条 binding。 | +| `evidence_bindings[].raw_path` | string | 是 | 该证据在工具返回中的路径,和 `source_invocation_id` 共同定位唯一证据位置。 | +| `evidence_bindings[].evidence_excerpt` | string | 是 | Executor 引用出来给 Verifier 阅读的证据片段。Gatekeeper 必须先核对其真实性。 | +| `hypotheses` | array | 否 | 假设,不是 confirmed fact。不能因为写在这里就进入最终确认结论。 | +| `missing_info` | array | 否 | 证据缺口。 | +| `recommended_actions` | array | 否 | 建议动作。本期只允许证据收集动作,不允许修复动作;动作可以绑定证据,但不能把缺证据假设写成已确认事实。 | + +### 3. Gatekeeper 输出 + +Gatekeeper 在 Verifier 前校验所有 `evidence_bindings`: + +```json +{ + "status": "pass", + "severity": "none", + "checked_bindings": [ + { + "claim_id": "claim-1", + "tool_name": "query_metrics", + "source_invocation_id": 12345, + "raw_path": "$.alerts[1]", + "status": "pass", + "matched_text": "HighMemoryUsage firing, service=order-service, current=91%, duration=15m" + } + ], + "failed_rules": [], + "warnings": [], + "errors": [] +} +``` + +字段定义: + +| 字段 | 类型 | 必填 | 说明 | +|---|---|---:|---| +| `status` | string | 是 | `pass` 或 `fail`。失败时 Verifier 不得输出 `PASS`。 | +| `severity` | string | 是 | `none`、`low_confid` 或 `reject`。`status=pass` 时必须为 `none`。 | +| `checked_bindings` | array | 是 | 已校验的证据绑定明细。 | +| `checked_bindings[].claim_id` | string | 否 | 对应 claim。recommended action 的 binding 可使用 `action_id`。 | +| `checked_bindings[].tool_name` | string | 是 | Executor 声称的工具名,必须和真实 invocation 对齐。 | +| `checked_bindings[].source_invocation_id` | number | 是 | 被核验的真实工具调用 ID。 | +| `checked_bindings[].raw_path` | string | 是 | 被核验的工具返回路径。 | +| `checked_bindings[].status` | string | 是 | 单条 binding 的校验结果:`pass` 或 `fail`。 | +| `checked_bindings[].matched_text` | string | 否 | Gatekeeper 从 `retrieval_details.evidence_refs` 或工具返回中找到的系统侧证据文本。 | +| `failed_rules` | array | 是 | 失败规则列表,例如 `evidence.invocation_ref`、`evidence.raw_path`、`evidence.excerpt_mismatch`。 | +| `warnings` | array | 是 | 非阻断风险。 | +| `errors` | array | 是 | 阻断错误。 | + +Gatekeeper 校验规则: + +1. `source_invocation_id` 必须存在于当前 session。 +2. `tool_name` 必须与该 invocation 的真实工具名一致。 +3. `raw_path` 必须能定位到该 invocation 的真实证据引用。 +4. `evidence_excerpt` 必须与系统侧 `text` 一致或高度相似。 +5. Gatekeeper 只判断“引用是否真实”,不判断 claim 是否成立。 + +### 4. Verifier 输入完整结构 + +Verifier 最终接收的输入结构如下: + +```json +{ + "original_query": "请排查 order-service 当前是否存在 HighMemoryUsage 告警,只确认内存使用率过高这一件事。", + "executor_structured_output": { + "claims": [ + { + "claim_id": "claim-1", + "claim_type": "observation", + "claim_text": "order-service 当前存在 HighMemoryUsage 告警,内存使用率达到 91%。", + "evidence_bindings": [ + { + "tool_name": "query_metrics", + "source_invocation_id": 12345, + "raw_path": "$.alerts[1]", + "evidence_excerpt": "HighMemoryUsage firing, service=order-service, current=91%, duration=15m" + } + ] + }, + { + "claim_id": "claim-2", + "claim_type": "observation", + "claim_text": "order-service JVM 堆内存使用接近上限,当前为 3.8GB/4GB。", + "evidence_bindings": [ + { + "tool_name": "query_metrics", + "source_invocation_id": 12345, + "raw_path": "$.alerts[1]", + "evidence_excerpt": "JVM heap usage 3.8GB/4GB" + } + ] + }, + { + "claim_id": "claim-3", + "claim_type": "observation", + "claim_text": "order-service 日志中也出现了内存使用率过高记录。", + "evidence_bindings": [ + { + "tool_name": "query_logs", + "source_invocation_id": 12346, + "raw_path": "$.logs[0]", + "evidence_excerpt": "[11:17][WARN][order-service] 内存使用率过高: 91.0%, JVM堆内存: 3.8GB/4GB, GC次数: 128" + } + ] + } + ], + "hypotheses": [ + { + "hypothesis_id": "hyp-1", + "hypothesis_text": "order-service 可能存在内存泄漏风险。", + "basis_claim_ids": ["claim-1", "claim-2"], + "missing_info": "缺少堆 dump、对象分配统计或更长时间窗口的内存曲线,不能确认内存泄漏。" + } + ], + "missing_info": [ + { + "info_id": "missing-1", + "description": "缺少堆 dump 或对象分配统计,无法确认内存泄漏根因。" + } + ], + "recommended_actions": [ + { + "action_id": "action-1", + "action_text": "继续查看 order-service 的 GC 日志、堆 dump 或对象分配统计。", + "reason": "当前证据可以确认内存使用率过高,但不足以确认是否存在内存泄漏。", + "evidence_bindings": [ + { + "tool_name": "query_metrics", + "source_invocation_id": 12345, + "raw_path": "$.alerts[1]", + "evidence_excerpt": "HighMemoryUsage firing, service=order-service, current=91%, duration=15m" + } + ] + } + ] + }, + "gatekeeper_result": { + "status": "pass", + "severity": "none", + "checked_bindings": [ + { + "claim_id": "claim-1", + "tool_name": "query_metrics", + "source_invocation_id": 12345, + "raw_path": "$.alerts[1]", + "status": "pass", + "matched_text": "HighMemoryUsage firing, service=order-service, current=91%, duration=15m" + }, + { + "claim_id": "claim-2", + "tool_name": "query_metrics", + "source_invocation_id": 12345, + "raw_path": "$.alerts[1]", + "status": "pass", + "matched_text": "JVM heap usage 3.8GB/4GB" + }, + { + "claim_id": "claim-3", + "tool_name": "query_logs", + "source_invocation_id": 12346, + "raw_path": "$.logs[0]", + "status": "pass", + "matched_text": "[11:17][WARN][order-service] 内存使用率过高: 91.0%, JVM堆内存: 3.8GB/4GB, GC次数: 128" + } + ], + "failed_rules": [], + "warnings": [], + "errors": [] + }, + "tool_trace_summary": [ + { + "trace_ref": "trace-1", + "tool_name": "query_metrics", + "source_invocation_ids": [12345], + "evidence_level": "direct", + "output_summary": "metric_evidence: HighMemoryUsage service=order-service current=91% duration=15m", + "invocation_count": 1, + "no_hit_invocation_count": 0 + }, + { + "trace_ref": "trace-2", + "tool_name": "query_logs", + "source_invocation_ids": [12346], + "evidence_level": "direct", + "output_summary": "log_evidence: [11:17][WARN][order-service] 内存使用率过高: 91.0%", + "invocation_count": 1, + "no_hit_invocation_count": 0 + } + ], + "executor_output_parse_status": { + "status": "valid", + "detail": "parsed executor evidence contract" + }, + "retry_context": null +} +``` + +Verifier 处理规则: + +- 先看 `gatekeeper_result.status`。如果为 `fail`,不得输出 `PASS`。 +- 正常校验时,只基于已通过 Gatekeeper 的 `evidence_bindings[].evidence_excerpt` 判断 `claim_text` 是否可推导。 +- `tool_trace_summary.output_summary` 只作为全局导航摘要,不作为主证据源。 +- `raw_path` 是 Gatekeeper 回查字段,不是 Verifier 推理字段。 +- Verifier 不读取 raw tool output,不重新检索,不替 Executor 补证据。 + +### 5. `tool_trace_summary` 精简字段 + +在该设计下,`tool_trace_summary` 只保留全局导航和审计索引能力。建议进入 Verifier 的最小字段为: + +```json +{ + "trace_ref": "trace-1", + "tool_name": "query_logs", + "source_invocation_ids": [12346], + "evidence_level": "direct", + "output_summary": "log_evidence: ...", + "invocation_count": 1, + "no_hit_invocation_count": 0 +} +``` + +字段定义: + +| 字段 | 类型 | 必填 | 说明 | +|---|---|---:|---| +| `trace_ref` | string | 是 | Verifier / 审计引用证据组的编号。 | +| `tool_name` | string | 是 | 证据组对应工具名。 | +| `source_invocation_ids` | array | 是 | 该证据组聚合的真实工具调用 ID。 | +| `evidence_level` | string | 是 | `direct`、`indirect` 或 `none`。只代表证据组强度,不代表 claim 通过。 | +| `output_summary` | string | 是 | 全局导航摘要。不得作为 Verifier 唯一证据源。 | +| `invocation_count` | number | 是 | 该证据组聚合的调用次数。 | +| `no_hit_invocation_count` | number | 是 | 无有效证据调用次数,用于负向场景和审计。 | + +以下字段不建议进入 Verifier 输入,只保留在审计或 trace UI: + +- `input_summary` +- `failed_invocation_count` +- `query_samples` +- `retrieval_layers` +- `relevance_levels` +- `source_documents` + +--- + +## 已确认优化方向:mock 工具命中质量先做小阶段 + +HikariCP 类问题归属工具 / mock 命中质量,不归属 Verifier。当前负向 PASS 是合理的,因为系统没有检索到真实 `order-service` HikariCP 连接池耗尽证据;不能通过放宽 Verifier 或让 runbook 代替事实证据来解决。 + +该问题可以作为独立小阶段优先落地,风险低、见效快。 + +### 阶段目标 + +让 HikariCP 相关问题能够稳定区分: + +```text +有真实 mock 日志 + -> Executor 可以引用真实 evidence + -> Gatekeeper / Verifier 可以验证 + -> 正向场景允许 PASS + +没有真实 mock 日志 + -> 明确 logs=[] + -> evidence_status=no_evidence + -> 不被 generic-service 占位日志污染 +``` + +### 改造点 + +1. `query_logs` no-hit 时不得 fallback 到 `generic-service` 占位日志。 +2. `query_logs` 应补齐 HikariCP / connection pool / active=50/50 / order-service 的正向 mock 日志。 +3. HikariCP 相关查询应支持同义表达命中: + - `HikariCP` + - `HikariPool` + - `connection pool` + - `数据库连接池` + - `连接池耗尽` + - `active=50/50` + - `waiting` + - `request timed out after 30000ms` + - `order-service` +4. no-hit 应表达为“工具调用成功但没有证据”,而不是工具失败: + +```json +{ + "success": true, + "logs": [], + "total": 0, + "message": "未找到匹配的日志" +} +``` + +并在 `tool_invocation.retrieval_details.evidence_status` 中记录: + +```json +{ + "evidence_status": "no_evidence" +} +``` + +### HikariCP 正向 mock 日志示例 + +```text +[ERROR][order-service] HikariPool-1 - Connection is not available, request timed out after 30000ms +[WARN][order-service] HikariCP pool stats: active=50/50, idle=0, waiting=32 +``` + +### 验收标准 + +1. 查询 `HikariCP` / `connection pool` / `active=50/50` / `order-service` 能命中真实 `order-service` 连接池日志。 +2. 查询不存在的服务或不相关关键词时,返回 `logs=[]`,并记录 `evidence_status=no_evidence`。 +3. 不再返回 `generic-service` 占位日志作为假证据。 +4. HikariCP positive case 能稳定走到 `PASS`。 +5. HikariCP negative case 仍然表达“未检索到真实证据”,不能误判 `PASS`。 + +### 边界 + +- 不通过 Verifier 放宽解决工具未命中问题。 +- 不允许 runbook 通用排查建议升级成“当前已经发生 HikariCP 连接池耗尽”的事实结论。 +- mock 数据应尽量和 `knowledge_base` 中的连接池排查 runbook 对齐,便于 eval 和 demo 形成闭环。 + +--- + +## 已确认优化方向:本期先用 Executor Prompt 控制过度展开 + +Planner 当前不会产出 `scope_contract`。本期先不改 Planner 输出契约,也不新增 scope contract 解析逻辑;先通过 Executor prompt 控制窄范围问题的过度展开。 + +### 本期边界 + +```text +不改 Planner +不新增 scope_contract +不新增 Controller +不做复杂 scope Gatekeeper +只调整 Executor Prompt 的单一职责和输出边界 +``` + +### Executor 单一职责 + +Executor 在本期应被约束为: + +```text +证据收集 + 微观事实提炼 +``` + +也就是: + +- 调用工具收集当前任务范围内的证据。 +- 输出工具证据直接支持的 observation / negative_observation claim。 +- 为 claim 绑定证据引用。 +- 输出 missing_info 表达证据缺口。 + +Executor 不应负责: + +- 生成最终用户答案。 +- 输出修复方案。 +- 扩展到用户未要求的服务、订单、告警、数据库、连接池或下游依赖。 +- 将 Runbook、Skill 或知识库中的通用经验写成当前环境已发生的事实。 +- 在证据不足时用“可能是、一般来说、根据经验”等话术补 claim。 + +### 窄范围问题识别 + +如果用户问题包含以下表达,Executor 应视为窄范围确认任务: + +- `只确认` +- `只排查` +- `只看` +- `不要分析` +- `不要扩展` +- `只回答` +- `是否真实存在` +- `是否存在某告警 / 某日志 / 某错误` + +窄范围确认任务下,Executor 必须遵守: + +1. 只输出 `observation` / `negative_observation` 类型 claim。 +2. claim 数量应为最少必要数量,通常 1 条,最多 2 条。 +3. 只能围绕用户明确要求的目标对象和主题输出 claim。 +4. 用户明确排除的对象、告警、服务、订单、数据库、连接池等,禁止出现在 claim 中。 +5. Runbook / Skill / 知识库只能用于指导要查什么,不能作为当前事实 claim。 +6. 如果证据不足,只输出 `missing_info`,不要补充合理化解释。 +7. `recommended_actions` 如果保留,只能是继续收集证据的动作,不允许是修复动作。 + +### 限制 claim 数量不是限制证据数量 + +这里限制的是: + +```text +claim 数量 +``` + +不是限制: + +```text +工具调用数量 +证据数量 +evidence_bindings 数量 +missing_info 数量 +``` + +窄范围任务的理想输出是: + +```text +1 条核心 claim +多条直接相关 evidence_bindings +必要的 missing_info +``` + +示例: + +```json +{ + "claim_id": "claim-1", + "claim_type": "observation", + "claim_text": "order-service 当前存在 HighMemoryUsage 告警,内存使用率达到 91%。", + "evidence_bindings": [ + { + "tool_name": "query_metrics", + "source_invocation_id": 12345, + "raw_path": "$.alerts[1]", + "evidence_excerpt": "HighMemoryUsage firing, service=order-service, current=91%, duration=15m" + }, + { + "tool_name": "query_metrics", + "source_invocation_id": 12345, + "raw_path": "$.alerts[1]", + "evidence_excerpt": "JVM heap usage 3.8GB/4GB" + }, + { + "tool_name": "query_logs", + "source_invocation_id": 12346, + "raw_path": "$.logs[0]", + "evidence_excerpt": "[11:17][WARN][order-service] 内存使用率过高: 91.0%, JVM堆内存: 3.8GB/4GB, GC次数: 128" + } + ] +} +``` + +不建议拆成: + +```text +claim-1: 存在 HighMemoryUsage +claim-2: 当前 91% +claim-3: JVM 3.8GB/4GB +claim-4: Full GC 增多 +claim-5: 可能内存泄漏 +``` + +原因是这些大多属于同一观察事实或证据细节,应合并到一条 claim 的 `evidence_bindings` 中。证据不足时应进入 `missing_info`,而不是生成更多弱 claim。 + +### Prompt 草案 + +可加入 `chat-executor-prompt.md` 的约束草案: + +```text +## 单一职责与输出边界 + +你是证据收集与微观事实提炼专家。 +你只负责调用工具收集证据,并输出工具证据直接支持的微观事实断言。 + +你不得输出最终用户答案。 +你不得生成修复方案。 +你不得扩展到用户未要求的服务、订单、告警、数据库、连接池或下游依赖。 +你不得将 Runbook、Skill 或知识库中的通用经验写成当前环境已发生的事实。 + +## 窄范围问题 HARD-GATE + +如果用户问题包含“只确认、只排查、只看、不要分析、不要扩展、只回答、是否真实存在、是否存在某告警/日志/错误”等表达,视为窄范围确认任务。 + +窄范围确认任务必须遵守: +1. 只输出 observation / negative_observation 类型 claim。 +2. claim 数量使用最少必要数量,通常 1 条,最多 2 条。 +3. 限制 claim 数量不限制 evidence_bindings 数量;每条 claim 应绑定所有直接相关证据。 +4. 不得把同一观察事实拆成多条 claim。 +5. 只能围绕用户明确要求的目标对象和主题输出 claim。 +6. 用户明确排除的对象、告警、服务、订单、数据库、连接池等,禁止出现在 claim 中。 +7. Runbook / Skill / 知识库只能用于指导要查什么,不能作为当前事实 claim。 +8. 如果证据不足,只输出 missing_info,不要补充合理化解释。 +9. recommended_actions 只能是继续收集证据的动作,不能是修复动作。 +``` + +### 验收标准 + +1. 用户说“只确认 HighCPUUsage”时,Executor 不输出订单、OOM、DB、HikariCP。 +2. 用户说“只确认 HighMemoryUsage”时,Executor 不输出“内存泄漏已确认”。 +3. 用户说“不要分析连接池”时,claim 中不出现 `HikariCP` / `connection pool`。 +4. runbook 内容只能进入 `missing_info` 或证据收集动作,不能变成 confirmed claim。 +5. 窄范围正向场景仍能绑定多条证据,不因 claim 数量限制导致证据不足。 + +### 后续可选增强 + +如果仅靠 Executor prompt 不稳定,再考虑后续引入: + +- Planner `scope_contract` +- Gatekeeper scope 校验 +- eval fixture 中的 forbidden claim 断言 + +--- + +## 当前剩余实施问题 + +当前真正剩余的设计问题主要是实施细节,而不是方向选择: + +1. 已决:`evidence_refs` 第一版由 `ToolInvocationRecorder` 根据工具返回统一抽取生成。 +2. 已决:`raw_path` 第一版只支持 `$.alerts[i]` / `$.logs[i]` / `$.evidence_blocks[i]` 三类稳定定位符。 +3. 已决:Gatekeeper 校验失败分级处理;伪造 / 张冠李戴类 `REJECT`,证据缺失 / 过渡期兼容类 `LOW_CONFID`。 +4. 已决:`VerifierInputHook` 自动回填必须收紧;本期只作兼容,不作为 `PASS` 依据,后续废弃。 +5. 已决:本期采用 Prompt-first 策略,只实现 Executor Prompt 约束;`scope_contract` 作为后续增强。 +6. 已决:补最小 eval / E2E 覆盖矩阵,覆盖 Memory、SlowResponse、HikariCP positive、HikariCP negative、Gatekeeper reject、窄范围 forbidden claim。 + +--- + +## 已确认实施决策:`evidence_refs` 第一版由 Recorder 统一抽取 + +第一版 `evidence_refs` 由 `ToolInvocationRecorder` 在工具调用入库时,根据 `toolName` 和工具返回 JSON 统一抽取生成,并写入 `tool_invocation.retrieval_details.evidence_refs`。 + +采用该方案的原因: + +- 当前 `query_logs` / `query_metrics` 返回已经是结构化 JSON。 +- 第一版只需要 `raw_path + text`,不需要复杂 metadata。 +- 抽取逻辑集中在 recorder 附近,更容易保证 `evidence_refs` 格式一致。 +- 不需要一次性改造所有工具返回协议。 + +第一版抽取规则: + +```text +query_metrics: + $.alerts[i] -> text = alert_name + state + description/duration + +query_logs: + $.logs[i] -> text = timestamp + level + service + message + +lookup_knowledge: + retrieval_details.evidence_blocks[i] -> text = content/title/source +``` + +实现边界: + +- Recorder 只做结构化抽取,不做诊断推理。 +- Recorder 不得把 `HighMemoryUsage`、`JVM 3.8GB/4GB` 等证据推理成“内存泄漏已确认”。 +- 如果 output JSON 解析失败或工具无可抽取结构,`evidence_refs` 可以为空,但必须保留原有 `evidence_status`。 +- 后续如果某个工具返回特别复杂,可以允许工具显式传入 `evidence_refs` 覆盖默认抽取,但第一版不做。 + +--- + +## 已确认实施决策:`raw_path` 第一版使用稳定定位符 + +第一版 `raw_path` 不实现完整 JSONPath,也不支持任意深层字段路径。它只是 `retrieval_details.evidence_refs` 内的稳定定位符,用来和 `source_invocation_id` 共同定位一条证据。 + +支持的路径格式仅包括: + +```text +$.alerts[i] +$.logs[i] +$.evidence_blocks[i] +``` + +对应工具: + +| 工具 | raw_path 格式 | 说明 | +|---|---|---| +| `query_metrics` | `$.alerts[i]` | 第 i 条告警证据 | +| `query_logs` | `$.logs[i]` | 第 i 条日志证据 | +| `lookup_knowledge` | `$.evidence_blocks[i]` | 第 i 个知识库证据块 | + +第一版不支持: + +```text +$.alerts[1].description +$.logs[0].message +$.data.alerts[0] +$.retrieval_details.evidence_blocks[0] +过滤表达式或复杂 JSONPath +``` + +原因: + +- Gatekeeper 不需要实现通用 JSONPath 引擎。 +- Executor 不需要理解工具返回内部字段细节。 +- `evidence_refs[].text` 已经保存该 raw path 对应的最小证据文本。 +- Gatekeeper 只需要在 `retrieval_details.evidence_refs` 中按 `raw_path` 精确匹配,然后比对 `evidence_excerpt` 与系统侧 `text`。 + +示例: + +```json +{ + "evidence_refs": [ + { + "raw_path": "$.alerts[1]", + "text": "HighMemoryUsage firing, service=order-service, current=91%, duration=15m; JVM heap usage 3.8GB/4GB" + } + ] +} +``` + +Executor 引用: + +```json +{ + "tool_name": "query_metrics", + "source_invocation_id": 12345, + "raw_path": "$.alerts[1]", + "evidence_excerpt": "HighMemoryUsage firing, service=order-service, current=91%, duration=15m" +} +``` + +Gatekeeper 校验: + +1. 查 `source_invocation_id=12345` 是否属于当前 session。 +2. 查该 invocation 的 `retrieval_details.evidence_refs` 是否存在 `raw_path="$.alerts[1]"`。 +3. 比对 `evidence_excerpt` 是否被该 `text` 支撑。 + +--- + +## 已确认实施决策:Gatekeeper 失败分级并入库审计 + +Gatekeeper 校验失败不一律 `REJECT`。第一版按失败性质分级: + +```text +物理级幻觉 / 证据污染 -> REJECT +证据缺失 / 格式不完整 / 过渡期兼容问题 -> LOW_CONFID +``` + +### Gatekeeper 输出结构 + +建议 Gatekeeper 输出增加 `severity` 字段: + +```json +{ + "status": "fail", + "severity": "reject", + "checked_bindings": [ + { + "claim_id": "claim-1", + "tool_name": "query_metrics", + "source_invocation_id": 12345, + "raw_path": "$.alerts[9]", + "status": "fail", + "rule": "evidence.raw_path", + "message": "raw_path not found in retrieval_details.evidence_refs" + } + ], + "failed_rules": ["evidence.raw_path"], + "warnings": [], + "errors": [ + { + "rule": "evidence.raw_path", + "message": "raw_path not found in retrieval_details.evidence_refs" + } + ] +} +``` + +字段定义: + +| 字段 | 类型 | 必填 | 说明 | +|---|---|---:|---| +| `status` | string | 是 | `pass` 或 `fail`。 | +| `severity` | string | 是 | `none`、`low_confid`、`reject`。`status=pass` 时为 `none`。 | +| `checked_bindings` | array | 是 | 每条 evidence binding 的校验结果。 | +| `failed_rules` | array | 是 | 命中的失败规则名。 | +| `warnings` | array | 是 | 非阻断风险。 | +| `errors` | array | 是 | 阻断错误,必须可审计。 | + +### `REJECT` 级失败 + +以下属于物理级幻觉或证据污染,应标记: + +```json +{ + "status": "fail", + "severity": "reject" +} +``` + +规则: + +1. `source_invocation_id` 不存在。 +2. `source_invocation_id` 不属于当前 session。 +3. `source_invocation_id` 存在,但 `tool_name` 与真实 invocation 不一致。 +4. `raw_path` 不存在于该 invocation 的 `retrieval_details.evidence_refs`。 +5. `evidence_excerpt` 与 `evidence_refs[].text` 明显不匹配。 + +这些情况表示 Executor 声称引用了某条证据,但系统无法核实,或核实结果与声称内容冲突。该类结果不得进入正常 Verifier 推导链路。 + +### `LOW_CONFID` 级失败 + +以下属于证据不足、格式不完整或过渡期兼容问题,应标记: + +```json +{ + "status": "fail", + "severity": "low_confid" +} +``` + +规则: + +1. `evidence_bindings` 为空。 +2. `raw_path` 缺失,但 `source_invocation_id` 存在。 +3. 旧工具调用未生成 `retrieval_details.evidence_refs`。 +4. `evidence_excerpt` 太短或太泛,无法稳定比对。 +5. Executor 输出格式不完整,但没有伪造具体 invocation / raw_path / excerpt。 + +这些情况不得 `PASS`,但不一定构成模型造假。 + +### 下游处理 + +- `severity=reject`:Verifier 不得输出 `PASS`,最终倾向 `REJECT`。 +- `severity=low_confid`:Verifier 不得输出 `PASS`,最终输出 `LOW_CONFID`。 +- `severity=none`:Verifier 正常判断 claim 是否可由已核验证据推出。 + +Gatekeeper 不直接生成最终用户答案,也不替代 Verifier。它只输出确定性校验结果和处理严重程度。 + +### 审计入库 + +Gatekeeper 的完整结果必须记录到数据库审计数据中。第一版不新增复杂表,复用当前 `DiagnosisSession.selfEvaluation` 中的 `verifier_evaluation.gatekeeper_result`。 + +需要入库的最小字段: + +```json +{ + "gatekeeper_result": { + "status": "fail", + "severity": "reject", + "checked_bindings": [], + "failed_rules": [], + "warnings": [], + "errors": [] + } +} +``` + +入库要求: + +- 每次 Verifier 前置 Gatekeeper 执行结果都必须保存。 +- `checked_bindings` 至少记录失败 binding;通过 binding 可按体积控制保留摘要。 +- `failed_rules` / `errors` 必须保留,便于离线定位是伪造 ID、raw_path 不存在,还是 excerpt 不匹配。 +- 审计数据必须能回答:Executor 引用了什么、Gatekeeper 查到了什么、为什么拦截或降级。 + +--- + +## 已确认实施决策:收紧 `VerifierInputHook` 自动回填 + +当前 `VerifierInputHook` 会在 Executor 未填写 `source_invocation_id` 时,根据 `tool_name` 从 `tool_trace_summary` 自动回填 invocation id。旧版实现里该字段可能表现为 `source_invocation_ids`。该逻辑是旧链路中的兼容补救,但与新设计的“精确证据引用”存在冲突。 + +新设计下,证据引用必须由: + +```text +source_invocation_id + raw_path + evidence_excerpt +``` + +共同构成。只按 `tool_name` 自动回填 invocation id,会把宽泛工具调用误包装成精确证据引用。 + +### 本期策略:兼容但收紧 + +本期暂不完全删除自动回填,但必须收紧: + +1. 不再根据 `tool_name` 批量回填多个 invocation id。 +2. 只有同一 `tool_name` 下存在唯一候选 invocation 时,才允许回填 `source_invocation_id`。 +3. 不允许自动回填 `raw_path`。 +4. 自动回填必须写入 `gatekeeper_result.warnings`。 +5. 自动回填后的 binding 如果缺少 `raw_path`,Gatekeeper 必须标记为 `severity=low_confid`,不得作为 `PASS` 依据。 +6. 伪造 `source_invocation_id`、`raw_path` 或 `evidence_excerpt` 仍然是 `severity=reject`。 + +warning 示例: + +```json +{ + "rule": "evidence.invocation_auto_backfill", + "message": "source_invocation_id was auto-filled from the unique tool invocation candidate; raw_path remains missing" +} +``` + +### 后续策略:废弃自动回填 + +等 Executor prompt 和输出结构稳定后,废弃自动回填: + +```text +Executor 必须自己输出 source_invocation_id + raw_path + evidence_excerpt。 +缺少精确引用时,不能 PASS。 +``` + +最终规则: + +```text +缺失引用 -> LOW_CONFID +伪造引用 -> REJECT +真实引用 + 可推导 claim -> PASS 候选 +``` + +--- + +## 已确认实施决策:Prompt-first,Contract-later + +本期不实现 Planner `scope_contract`,只通过 Executor Prompt 控制窄范围问题的过度展开。 + +原因: + +```text +scope_contract 不只是加一个字段。 +它会牵动 Planner 输出契约、PlannerSkillMetadataHook、Executor 对 planner_plan 的解析、可能的 Gatekeeper scope 校验和 eval fixture。 +``` + +本期真正要先验证的是: + +```text +Executor 是否能在 prompt 约束下减少无关 claim、根因推断和修复建议。 +``` + +### 本期做 + +1. 调整 `chat-executor-prompt.md`。 +2. 强化 Executor 单一职责:证据收集 + 微观事实提炼。 +3. 窄范围问题只输出 `observation` / `negative_observation`。 +4. 输出最少必要 claim,通常 1 条,最多 2 条。 +5. 不限制 `evidence_bindings` 数量。 +6. Runbook / Skill / 知识库不能作为当前事实。 +7. `recommended_actions` 如果保留,只能是证据收集动作,不能是修复动作。 +8. 增加 eval / E2E 验证 forbidden claim。 + +### 本期不做 + +1. 不改 Planner。 +2. 不新增 `scope_contract`。 +3. 不解析 `planner_plan.scope_contract`。 +4. 不做 Gatekeeper scope 校验。 +5. 不改多 Agent 编排。 + +### 何时再引入 `scope_contract` + +如果出现以下情况,再进入下一阶段: + +1. Prompt 调整后,Executor 仍频繁输出用户明确排除的服务或主题。 +2. 窄范围问题仍然生成 `root_cause` / `risk` / 修复建议。 +3. eval 中 forbidden claim 仍不稳定。 +4. Planner 已经能稳定识别用户 scope,但 Executor 不遵守。 + +### 本期验收 + +1. HighCPUUsage 窄范围 case:不出现 HighMemoryUsage、SlowResponse、order-123、HikariCP、DB。 +2. HighMemoryUsage 窄范围 case:不出现“内存泄漏已确认”。 +3. SlowResponse 窄范围 case:不出现数据库连接池耗尽根因。 +4. 用户明确排除 HikariCP 时,`claim_text` 和最终答案不出现 HikariCP 确认结论。 + +--- + +## 已确认实施决策:最小 eval / E2E 覆盖矩阵 + +本期 eval / E2E 不铺太大,围绕已确认的改造点做最小闭环: + +```text +evidence_refs 抽取 +raw_path / excerpt Gatekeeper 校验 +Verifier 基于已核验 evidence_excerpt 推导 +mock 工具命中质量 +Executor 窄范围 prompt 约束 +``` + +### 1. HighMemoryUsage positive + +目标:验证 `evidence_refs -> Executor evidence_binding -> Gatekeeper -> Verifier` 能让 HighMemoryUsage 从证据存在但 `LOW_CONFID` 变成可通过。 + +场景: + +```text +只确认 order-service 是否存在 HighMemoryUsage 告警。 +``` + +期望: + +- Gatekeeper `pass`。 +- Verifier `PASS`。 +- claim 只包含 HighMemoryUsage 观察事实。 +- 不确认内存泄漏。 + +覆盖点: + +- `query_metrics` 生成 `evidence_refs`。 +- `raw_path=$.alerts[i]`。 +- `evidence_excerpt` 可核验。 +- Verifier 基于已核验 excerpt 做推导。 + +### 2. SlowResponse positive + +目标:验证一个 claim 可以绑定多条证据,不靠拆出多条 claim 凑信息。 + +场景: + +```text +只确认 user-service 是否存在 SlowResponse 告警和慢请求日志。 +``` + +期望: + +- Verifier `PASS`。 +- claim 数量 1-2 条。 +- `evidence_bindings` 同时包含 alert 和 logs。 +- 不推断数据库连接池耗尽。 + +覆盖点: + +- `query_metrics` + `query_logs`。 +- `raw_path=$.alerts[i]`。 +- `raw_path=$.logs[i]`。 +- 多 `evidence_bindings`。 +- 窄范围不扩展根因。 + +### 3. HikariCP positive + +目标:验证 HikariCP 相关查询能命中真实 mock 日志。 + +场景: + +```text +确认 order-service 是否存在 HikariCP 连接池耗尽。 +``` + +期望: + +- `query_logs` 命中 `order-service` HikariCP 日志。 +- 不返回 `generic-service`。 +- Gatekeeper `pass`。 +- Verifier `PASS`。 +- 最终答案可以确认连接池耗尽日志存在。 + +覆盖点: + +- `HikariCP` / `HikariPool` / `connection pool` / `active=50/50` 同义匹配。 +- `raw_path=$.logs[i]`。 +- HikariCP 正向 mock 数据。 + +### 4. HikariCP negative + +目标:验证 no-hit 不污染证据链。 + +场景: + +```text +确认 inventory-service 是否存在 HikariCP 连接池耗尽。 +``` + +期望: + +- `logs=[]`。 +- `evidence_status=no_evidence`。 +- 不返回 `generic-service`。 +- Verifier 不输出正向 `PASS` 结论。 +- 最终表达“未检索到真实证据”。 + +覆盖点: + +- no-hit 语义。 +- `negative_observation`。 +- `generic-service` 占位日志清理。 + +### 5. Gatekeeper reject + +目标:验证伪造 `raw_path` 或 “真实 invocation + 编造 excerpt” 会被拦截。 + +该场景优先做单元测试,不一定需要 E2E。 + +输入示例: + +```json +{ + "source_invocation_id": 12345, + "raw_path": "$.alerts[99]", + "evidence_excerpt": "HikariCP active=50/50" +} +``` + +期望: + +- `gatekeeper_result.status=fail`。 +- `gatekeeper_result.severity=reject`。 +- `failed_rules` 包含 `evidence.raw_path` 或 `evidence.excerpt_mismatch`。 +- 结果不得 `PASS`。 +- `gatekeeper_result` 入库审计。 + +覆盖点: + +- `raw_path` 不存在。 +- `excerpt` 不匹配。 +- `REJECT` 分级。 +- 审计入库。 + +### 6. Executor narrow forbidden claim + +目标:验证 Executor Prompt 对窄范围问题的约束有效。 + +场景: + +```text +只确认 HighCPUUsage,不要分析订单123、OOM、数据库慢查询、连接池或 user-service。 +``` + +期望: + +- claims 只围绕 HighCPUUsage / payment-service。 +- 不出现 `order-123`。 +- 不出现 `OOM`。 +- 不出现 `DB` / `database`。 +- 不出现 `HikariCP` / `connection pool`。 +- 不出现 `user-service`。 + +覆盖点: + +- Executor Prompt。 +- forbidden claim keywords。 +- Composer 不泄漏 blocked 内容。 + +### 推荐测试层级 + +单元测试: + +- `evidence_refs` 抽取。 +- `raw_path` 校验。 +- Gatekeeper `severity` 分级。 +- `VerifierInputHook` 自动回填收紧。 + +离线 eval fixture: + +- Gatekeeper reject。 +- unsupported / forbidden claim。 +- Memory / SlowResponse 的结构化输入。 + +E2E: + +- HighMemoryUsage positive。 +- SlowResponse positive。 +- HikariCP positive。 +- HikariCP negative。 +- HighCPUUsage narrow forbidden。 + +### 本期最小可验收 + +如果时间紧,至少完成: + +1. Gatekeeper raw_path reject 单测。 +2. `evidence_refs` 抽取单测。 +3. HikariCP positive E2E。 +4. HikariCP negative E2E。 +5. HighMemoryUsage positive E2E。 +6. narrow forbidden E2E。 + +--- + +## 历史候选优化方向 + +说明:本节保留早期候选方向,便于理解设计演进。当前已确认的决策以上文“已确认设计方向 / 已确认优化方向”为准: + +- Verifier 主证据源改为 Gatekeeper 已核验的 `evidence_excerpt`。 +- Gatekeeper 需要核对 `source_invocation_id + raw_path + evidence_excerpt`。 +- `tool_trace_summary` 降级为全局导航摘要,不再作为唯一证据源。 +- HikariCP / generic-service 问题先作为 mock 工具命中质量小阶段处理。 +- 本期暂不做 Planner `scope_contract`,先通过 Executor Prompt 控制窄范围过度展开。 + +### 方向 A:增强 `ToolTraceSummaryService` 的证据摘要保真度 + +让 summary 不再只是人类可读摘要,而是保留 Verifier 可用的最小机器证据。 + +候选策略: +- 每个 tool invocation 至少保留 top 3-5 条关键 evidence excerpt。 +- 每个 firing alert 保留结构化摘要:告警名、服务、状态、当前值、阈值、持续时间。 +- 日志类工具按错误码、服务名、时间、endpoint、耗时等字段保留关键片段。 +- 对趋势型证据保留首尾值和采样数量。 + +需要讨论: +- `output_summary` 继续做人类摘要,还是升级为机器证据索引? +- 是否需要新增字段,而不是继续塞进 `output_summary`? + +--- + +### 方向 B:Verifier 直接消费 Executor 的 `evidence_excerpt` + +Verifier 判断 claim 是否可推导时,不只看 `tool_trace_summary.output_summary`,也看 Executor 已经绑定到 claim / action 的证据片段。 + +优点: +- 证据离 claim 更近。 +- 对 Memory / SlowResponse 这类 case 更容易 PASS。 +- 不要求 summary 承担全部证据承载职责。 + +风险: +- 如果 Executor 编造 excerpt,Verifier 会被污染。 +- 需要 Gatekeeper 先保证 excerpt 能回溯到真实 tool output。 + +需要讨论: +- Verifier 消费 excerpt 前,Gatekeeper 是否必须加入原文回溯 / 相似度校验? +- Verifier 应该把 excerpt 当作主证据,还是辅助证据? + +--- + +### 方向 C:Gatekeeper 增加 excerpt 原文回溯校验 + +在 Verifier 前,用代码验证 Executor 声称的 `evidence_excerpt` 是否能在对应 tool invocation 的原始输出中找到或高度相似。 + +候选规则: +- 证据引用必须能回溯到真实 tool invocation。 +- `evidence_excerpt` 与系统侧证据文本明显不匹配时拒绝。 +- 找不到引用片段时,不进入 Verifier,避免张冠李戴。 + +优点: +- 为 Verifier 直接消费 excerpt 建立前置安全条件。 +- 能拦住“真实 ID + 编造文本”的物理级幻觉。 + +风险: +- 原始输出和 excerpt 经过格式化后可能不完全一致,需要容忍截断、空白、标点差异。 +- 相似度阈值需要通过 fixture 校准。 + +--- + +### 方向 D:窄范围问题限制 Executor claim 数量 + +对于用户明确要求“只确认某一件事”的问题,Executor 输出应更克制。 + +候选策略: +- Planner 或 Executor prompt 生成 scope contract。 +- Executor 对窄范围问题最多输出 1-2 个核心 claim。 +- 与用户明确排除的服务、告警、错误类型无关的 claim 不输出。 + +需要讨论: +- 这个约束放在 Planner,还是 Executor 自己根据用户问题判断? +- 是否需要 Gatekeeper 增加 scope 校验? + +--- + +### 方向 E:修复 mock 工具命中质量 + +针对 HikariCP 这类 case,补齐正向 mock 分支或改进查询匹配。 + +候选策略: +- `query_logs` 支持 `HikariCP`、`connection pool`、`active=50/50`、`order-service` 等同义匹配。 +- 避免正向场景返回 `generic-service` 占位日志。 +- no-hit 时明确标记为无证据,减少污染。 + +需要讨论: +- mock 工具是只服务 eval,还是也作为 demo 数据源长期维护? +- mock 数据是否应该和 `knowledge_base` fixture 建立对应关系? + +--- + +## 初步验收标准 + +优化后至少需要覆盖以下回归: + +1. `HighCPUUsage` 正向场景稳定 `PASS`。 +2. `HighMemoryUsage` 在证据存在时不再因为 summary 丢证据而 `LOW_CONFID`。 +3. `SlowResponse` 在 alert 与慢请求日志都存在时不再因为 summary 丢证据而 `LOW_CONFID`。 +4. `HikariCP` 负向场景仍能正确表达“未检索到真实证据”。 +5. `HikariCP` 正向场景能够命中真实 mock 数据并被 Verifier 正确识别。 +6. Gatekeeper 审计字段完整记录校验结果。 +7. 最终 `user_facing_answer` 不泄漏 raw JSON、tool raw output 或 Executor 未通过 Verifier 的 claim。 + +--- + +## 历史问题记录 + +以下问题是早期讨论时提出的候选方向,当前均已被上文“已确认设计方向 / 已确认实施决策”覆盖: + +| 历史问题 | 当前决策 | +|---|---| +| Verifier 后续应该以 `tool_trace_summary` 为唯一证据源,还是同时消费 Executor `evidence_excerpt`? | Verifier 主证据源改为 Gatekeeper 已核验的 `evidence_excerpt`,`tool_trace_summary` 降级为全局导航摘要。 | +| 如果 Verifier 消费 `evidence_excerpt`,Gatekeeper 是否必须先做 excerpt 原文回溯? | 必须。Gatekeeper 校验 `source_invocation_id + raw_path + evidence_excerpt` 后,Verifier 才能使用 excerpt。 | +| `ToolTraceSummaryService` 应该增强现有 `output_summary`,还是新增更结构化的 evidence excerpt 字段? | 本期不把 `output_summary` 作为主证据源;结构化证据引用进入 `tool_invocation.retrieval_details.evidence_refs`。 | +| summary 保留多少条 excerpt 才够,不会重新变成超长上下文? | 不靠 summary 承载主证据;claim-local evidence 由 Executor 引用并经 Gatekeeper 校验。 | +| `generic-service` 占位日志是否应该彻底视为 `success=false` 或 `evidence_level=none`? | no-hit 应返回 `logs=[]` 且 `evidence_status=no_evidence`,不再返回 `generic-service` 占位日志作为证据。 | +| 窄范围问题是否需要显式 scope contract,避免 Executor 过度展开? | 本期采用 Prompt-first,只调 Executor Prompt;`scope_contract` 作为后续增强。 | diff --git a/mvp/issues/README.md b/mvp/issues/README.md index 6090f40..4019314 100644 --- a/mvp/issues/README.md +++ b/mvp/issues/README.md @@ -8,6 +8,7 @@ | ISS-004 | Executor 域级检索水位控制(Phase 2) | 低 | 待规划 | [ISS-004-executor-domain-hard-limit.md](ISS-004-executor-domain-hard-limit.md) | | ISS-005 | 证据链补齐与降级契约收敛 | 高 | 已归档 | [ISS-005-evidence-trace-hardening.md](ISS-005-evidence-trace-hardening.md) | | ISS-006 | 固定诊断评测集与回归 Harness | 高 | 已归档 | [ISS-006-diagnosis-eval-harness.md](ISS-006-diagnosis-eval-harness.md) | +| ISS-007 | Verifier 证据摘要保真与工具命中质量问题 | 高 | 已实施 | [ISS-007-verifier-evidence-summary-fidelity.md](ISS-007-verifier-evidence-summary-fidelity.md) | | executor-evidence-attribution-hallucination | Executor 证据归因幻觉 | 高 | 待规划 | [executor-evidence-attribution-hallucination.md](executor-evidence-attribution-hallucination.md) | | executor-self-evidence-loop-design-note | Executor 自证循环与证据摘要链路设计记录 | 高 | 已形成方向 | [executor-self-evidence-loop-design-note.md](executor-self-evidence-loop-design-note.md) | | expand-diagnosis-eval-fixtures | 补齐固定诊断评测 fixture 与 baseline | 中 | 已归档 | [expand-diagnosis-eval-fixtures.md](expand-diagnosis-eval-fixtures.md) | diff --git a/openspec/changes/archive/2026-07-08-verifier-evidence-reference-fidelity/.archive-ready b/openspec/changes/archive/2026-07-08-verifier-evidence-reference-fidelity/.archive-ready new file mode 100644 index 0000000..e69de29 diff --git a/openspec/changes/archive/2026-07-08-verifier-evidence-reference-fidelity/.committed b/openspec/changes/archive/2026-07-08-verifier-evidence-reference-fidelity/.committed new file mode 100644 index 0000000..e69de29 diff --git a/openspec/changes/archive/2026-07-08-verifier-evidence-reference-fidelity/decisions.md b/openspec/changes/archive/2026-07-08-verifier-evidence-reference-fidelity/decisions.md new file mode 100644 index 0000000..6fb68c2 --- /dev/null +++ b/openspec/changes/archive/2026-07-08-verifier-evidence-reference-fidelity/decisions.md @@ -0,0 +1,73 @@ +# Decisions + +## Discover Context + +- Source issue: `mvp/issues/ISS-007-verifier-evidence-summary-fidelity.md`. +- Related existing specs: `chat-verifier-agent`, `evidence-trace-hardening`. +- Related devflow records: `executor-gatekeeper-hook`, `executor-verifier-claim-checks`, `executor-composer-final-answer`, `evidence-trace-hardening`. +- Current repo instruction requested semantic code search and LSP confirmation before code changes; those tools are not exposed in this environment, so implementation will use `rg`, direct code reading, and focused tests as fallback evidence. + +## Classification + +- sm-flow scale: `standard`. +- Reason: internal contract change across recorder, hook, Gatekeeper, verifier prompt, mock tools, tests, and E2E validation. + +## Question Pool + +| Question | Mode | Resolution | +|---|---|---| +| Should Planner change or add `scope_contract`? | user-interview, already resolved in issue | No. This change is prompt-first and does not alter Planner. | +| Should Gatekeeper move to Executor hook for retry? | user-interview, already resolved in issue | No. Gatekeeper remains in Verifier hook path for this version. | +| Should new DB tables be added for evidence refs or audit? | user-interview, already resolved in issue | No. Use `tool_invocation.retrieval_details.evidence_refs` and existing `self_evaluation.verifier_evaluation.gatekeeper_result`. | +| Should `tool_trace_summary` remain the primary evidence source? | user-interview, already resolved in issue | No. It becomes navigation/audit context; verified claim-local excerpts become primary evidence for derivability. | +| Should full JSONPath be supported? | user-interview, already resolved in issue | No. Only stable raw paths listed in design are supported. | + +## Cross-artifact Alignment + +| Check | Status | +|---|---| +| Issue background and target -> proposal | aligned | +| Proposal scope and non-goals -> design | aligned | +| Design protocol and risks -> specs/tasks | aligned | +| Specs observable behavior -> tasks acceptance | aligned | + +## Architecture Audit + +The change keeps the existing agent orchestration and only hardens the evidence payload between Executor, Gatekeeper, and Verifier. The highest coupling risk is transition compatibility from plural `source_invocation_ids` to singular `source_invocation_id`; the implementation must accept old data but only allow precise new bindings to pass. No new DB table is introduced, reducing migration risk. Verifier prompt and effective verdict guardrails must agree on `gatekeeper_result.severity`, otherwise the model could still produce a PASS that runtime later must downgrade. + +## Commit Gate + +Passed on 2026-07-08. + +- `openspec validate verifier-evidence-reference-fidelity --strict`: passed. +- `openspec validate --specs`: passed. +- File completeness: proposal, design, specs, tasks present. +- Consistency: proposal -> design -> specs -> tasks aligned. +- Apply authorization: user requested continuing implementation and fixing autonomously; proceed to Apply. + +## Apply Verification + +Focused tests: + +- `mvn "-Dtest=ToolInvocationRecorderTest,ExecutorGatekeeperServiceTest,VerifierInputHookTest,QueryLogsToolsTest,ChatServiceSequentialAgentTest" test`: passed, 38 tests. +- `mvn "-Dtest=ExecutorGatekeeperServiceTest" test`: passed, 9 tests after adding the old-invocation-without-evidence-refs case. + +OpenSpec: + +- `openspec validate verifier-evidence-reference-fidelity --strict`: passed. +- `openspec validate --specs`: passed. + +End-to-end sessions: + +| Case | Session | Result | Gatekeeper | +|---|---|---|---| +| HikariCP positive | `iss007-hikari-positive-20260708-1553` | `PASS`; answer confirmed order-service HikariCP timeout and pool saturation logs | `pass / none`, checked_bindings=2 | +| HikariCP negative | `iss007-hikari-negative-20260708-1555` | `LOW_CONFID`; no `generic-service` pollution; no positive inventory-service confirmation | `fail / low_confid` due incomplete negative evidence references | +| HighMemoryUsage positive | `iss007-memory-positive-20260708-1558` | `PASS`; answer confirmed HighMemoryUsage 91% without confirming memory leak | `pass / none`, checked_bindings=1 | +| SlowResponse positive | `iss007-slow-positive-20260708-1600` | `PASS`; answer confirmed SlowResponse and slow request logs without DB-pool root cause | `pass / none`, checked_bindings=7 | +| Narrow HighCPUUsage | `iss007-narrow-highcpu-20260708-1602` | `PASS`; answer only covered payment-service HighCPUUsage | `pass / none`, checked_bindings=1 | + +Residual observation: + +- HikariCP negative no longer returns `generic-service`, and no-hit rows persist `evidence_status=no_evidence`. +- The model still issued an extra broad HikariCP query without the `inventory-service` filter and used order-service as a context claim. This is a remaining narrow-scope behavior issue, not a mock data pollution issue. It is acceptable for this change because the final verdict did not become a false positive for inventory-service. diff --git a/openspec/changes/archive/2026-07-08-verifier-evidence-reference-fidelity/design.md b/openspec/changes/archive/2026-07-08-verifier-evidence-reference-fidelity/design.md new file mode 100644 index 0000000..687cda1 --- /dev/null +++ b/openspec/changes/archive/2026-07-08-verifier-evidence-reference-fidelity/design.md @@ -0,0 +1,180 @@ +# Design + +## Data Flow + +```text +Evidence tool result + -> ToolInvocationRecorder + persists tool_invocation.retrieval_details.evidence_refs + -> Executor + emits claim-local evidence_bindings + -> VerifierInputHook + preserves structured payload and runs Gatekeeper + -> ExecutorGatekeeperService + validates reference authenticity + -> Verifier + checks derivability from verified excerpts + -> ChatService / Composer + enforces effective verdict and safe final answer +``` + +## Evidence Reference Contract + +`tool_invocation.retrieval_details.evidence_refs` is an array of minimal evidence references: + +```json +{ + "evidence_refs": [ + { + "raw_path": "$.alerts[0]", + "text": "HighMemoryUsage firing, service=order-service, current=91%, duration=15m" + } + ] +} +``` + +Supported `raw_path` formats in this change: + +- `$.alerts[i]` for `query_metrics` +- `$.logs[i]` for `query_logs` +- `$.evidence_blocks[i]` for `lookup_knowledge` + +Unsupported in this change: + +- Deep JSONPath such as `$.alerts[1].description` +- Nested aliases such as `$.retrieval_details.evidence_blocks[0]` +- Filter expressions +- Tool-specific metadata beyond `raw_path` and `text` + +## Executor Binding Contract + +Executor V2 claim bindings should use: + +```json +{ + "tool_name": "query_metrics", + "source_invocation_id": 12345, + "raw_path": "$.alerts[0]", + "evidence_excerpt": "HighMemoryUsage firing, service=order-service, current=91%, duration=15m" +} +``` + +Compatibility: + +- Legacy `source_invocation_ids` may still be read for transition. +- New precise validation requires singular `source_invocation_id` plus `raw_path`. +- Missing `raw_path` is `LOW_CONFID`, not `PASS`. + +## Gatekeeper Severity + +Gatekeeper output includes: + +```json +{ + "status": "fail", + "severity": "reject", + "checked_bindings": [], + "failed_rules": [], + "warnings": [], + "errors": [] +} +``` + +Severity mapping: + +- `status=pass`, `severity=none`: all checked bindings are authentic. +- `status=fail`, `severity=low_confid`: evidence is missing or incomplete but not fabricated. +- `status=fail`, `severity=reject`: Executor cites fabricated or mismatched evidence. + +`reject` cases: + +- Invocation ID does not exist. +- Invocation belongs to another session. +- Tool name mismatches persisted invocation. +- `raw_path` is not present in `retrieval_details.evidence_refs`. +- `evidence_excerpt` clearly mismatches the system-side `text`. + +`low_confid` cases: + +- Confirmed claim has empty `evidence_bindings`. +- `source_invocation_id` exists but `raw_path` is missing. +- Old invocation lacks `evidence_refs`. +- `evidence_excerpt` is too short or generic to compare. +- Output is incomplete without concrete fabricated IDs or paths. + +## Verifier Behavior + +Verifier should treat `gatekeeper_result.severity` as a hard boundary: + +- `reject`: effective result cannot be `PASS`; fabricated-reference cases should become `REJECT`. +- `low_confid`: effective result cannot be `PASS`; missing-reference cases should become `LOW_CONFID`. +- `none`: Verifier judges whether claim text is derivable from verified excerpts. + +`tool_trace_summary` remains available as navigation and audit context, but claim-local verified excerpts are the primary evidence for derivability. + +## Hook Placement + +Gatekeeper remains in the Verifier hook path for this change. Retry or rollback into Executor is not implemented in this stage. + +Existing ad hoc verifier-hook validation should be replaced by Gatekeeper output. The hook may normalize compatibility fields, but it should not independently decide pass/fail outside Gatekeeper semantics. + +## Auto-backfill + +`VerifierInputHook` may only auto-fill a missing `source_invocation_id` when exactly one invocation candidate exists for the binding's `tool_name`. + +Rules: + +- Never bulk-fill multiple IDs. +- Never auto-fill `raw_path`. +- Add a warning when auto-fill occurs. +- Auto-filled binding without `raw_path` must remain `LOW_CONFID`. + +## Prompt Constraints + +Executor prompt changes are prompt-first, contract-later: + +- Executor is an evidence collector and micro-fact extractor. +- Narrow-scope questions should usually produce one claim and at most two claims. +- Do not limit evidence binding count. +- Emit observation or negative observation, not root-cause certainty, for narrow confirmation questions. +- Runbook and skill content cannot become current incident facts. +- Recommended actions, if any, are evidence-collection next steps, not remediation actions. + +## HikariCP Mock Behavior + +`query_logs` should support positive mock hits for: + +- `HikariCP` +- `HikariPool` +- `connection pool` +- `数据库连接池` +- `连接池耗尽` +- `active=50/50` +- `waiting` +- `request timed out after 30000ms` +- `order-service` + +Positive output should include order-service HikariCP log records. No-hit should return `logs=[]` and `evidence_status=no_evidence`, not `generic-service` placeholder logs. + +## Audit + +Gatekeeper result must be persisted under: + +```text +diagnosis_session.self_evaluation.verifier_evaluation.gatekeeper_result +``` + +The minimum persisted fields are: + +- `status` +- `severity` +- `checked_bindings` +- `failed_rules` +- `warnings` +- `errors` + +## Interface Impact + +Impact level: L2 internal contract change. + +The change affects internal Agent payload JSON, persisted audit JSON, prompts, and test fixtures. It does not add public HTTP endpoints or new database tables. diff --git a/openspec/changes/archive/2026-07-08-verifier-evidence-reference-fidelity/proposal.md b/openspec/changes/archive/2026-07-08-verifier-evidence-reference-fidelity/proposal.md new file mode 100644 index 0000000..0b76e4a --- /dev/null +++ b/openspec/changes/archive/2026-07-08-verifier-evidence-reference-fidelity/proposal.md @@ -0,0 +1,62 @@ +# verifier-evidence-reference-fidelity + +## Problem + +Recent end-to-end checks show that some narrow diagnosis questions still become `LOW_CONFID` even when the raw tool output and Executor evidence excerpts contain enough concrete evidence. The failure is caused by evidence being compressed or lost before Verifier reasoning, plus mock log no-hit behavior that can return `generic-service` placeholder logs. + +Current weak points: + +- Verifier still relies too heavily on `tool_trace_summary.output_summary`. +- Executor evidence bindings identify tool invocations but do not precisely locate evidence inside the invocation output. +- Gatekeeper validates invocation IDs and tool names, but does not yet verify `raw_path` and excerpt fidelity. +- `query_logs` mock data cannot reliably produce a positive HikariCP connection-pool exhaustion path and can pollute no-hit results with placeholder logs. +- Narrow-scope Executor answers can still over-expand into unrelated claims. + +## Proposed Change + +Introduce a claim-local evidence reference protocol: + +```text +tool raw output + -> ToolInvocationRecorder stores retrieval_details.evidence_refs + -> Executor outputs claim + source_invocation_id + raw_path + evidence_excerpt + -> Gatekeeper verifies that the reference is real + -> Verifier judges whether verified evidence can derive the claim + -> Composer only expresses Verifier-allowed material +``` + +The first implementation keeps the orchestration unchanged. Gatekeeper remains in the Verifier input hook path. Planner `scope_contract` is out of scope for this change. + +## Scope + +- Add minimal `retrieval_details.evidence_refs` extraction for `query_metrics`, `query_logs`, and `lookup_knowledge`. +- Tighten Executor evidence bindings to prefer singular `source_invocation_id`, stable `raw_path`, and `evidence_excerpt`. +- Extend Gatekeeper to validate `source_invocation_id + raw_path + evidence_excerpt`. +- Add Gatekeeper severity: `none`, `low_confid`, `reject`. +- Make Verifier consume verified `evidence_excerpt` as the primary claim-local evidence. +- Tighten `VerifierInputHook` auto-backfill: only unique invocation candidate, never `raw_path`, and no `PASS` without a precise reference. +- Fix HikariCP mock log matching and no-hit behavior. +- Update Executor and Verifier prompts for narrow-scope and derivability behavior. +- Add focused tests and end-to-end checks for the minimum acceptance matrix. + +## Non-goals + +- Do not change Planner output. +- Do not implement Planner `scope_contract`. +- Do not add new database tables. +- Do not implement a general JSONPath engine. +- Do not make `tool_trace_summary` the primary evidence source again. +- Do not allow Executor `diagnosis_summary` or `user_facing_answer` to re-enter the V2 contract. + +## Context Constraints + +- `tool_invocation.retrieval_details` is the preferred place for tool-specific structured details. +- `DiagnosisSession.selfEvaluation.verifier_evaluation.gatekeeper_result` is the existing audit container and must be preserved. +- `read_skill` / runbook guidance is not incident evidence. +- Existing Composer routing must keep raw Executor JSON out of normal user answers. + +## Risks + +- Existing tests or prompts may still assume plural `source_invocation_ids`. +- Some old invocations will not have `evidence_refs`; those must downgrade to `LOW_CONFID`, not `PASS`. +- Similarity checks must tolerate formatting changes without accepting unrelated text. diff --git a/openspec/changes/archive/2026-07-08-verifier-evidence-reference-fidelity/specs/chat-verifier-agent/spec.md b/openspec/changes/archive/2026-07-08-verifier-evidence-reference-fidelity/specs/chat-verifier-agent/spec.md new file mode 100644 index 0000000..4119b6a --- /dev/null +++ b/openspec/changes/archive/2026-07-08-verifier-evidence-reference-fidelity/specs/chat-verifier-agent/spec.md @@ -0,0 +1,110 @@ +## ADDED Requirements + +### Requirement: Executor evidence bindings SHALL support precise evidence references +Executor V2 evidence bindings SHALL support precise evidence references that locate evidence inside a persisted tool invocation. + +#### Scenario: Precise evidence binding contains invocation path and excerpt +- **WHEN** Executor binds evidence to a claim +- **THEN** the binding SHOULD include singular `source_invocation_id` +- **AND** the binding SHOULD include `raw_path` +- **AND** the binding SHALL include `tool_name` and `evidence_excerpt` +- **AND** the `raw_path` SHALL be interpreted relative to the referenced tool invocation's `retrieval_details.evidence_refs` + +#### Scenario: Legacy plural invocation ids remain compatibility only +- **WHEN** Executor emits legacy `source_invocation_ids` +- **THEN** the system MAY read them for compatibility +- **AND** they SHALL NOT be sufficient for a precise Gatekeeper pass without `raw_path` + +### Requirement: Gatekeeper SHALL validate evidence reference fidelity +Gatekeeper SHALL validate that Executor evidence bindings point to real current-session evidence references before Verifier uses them as primary evidence. + +#### Scenario: Valid precise binding passes +- **WHEN** a binding's `source_invocation_id` exists in the current session +- **AND** the binding's `tool_name` matches the persisted invocation +- **AND** the binding's `raw_path` exists in `retrieval_details.evidence_refs` +- **AND** the binding's `evidence_excerpt` is supported by the matching evidence ref text +- **THEN** Gatekeeper SHALL return `status=pass` +- **AND** Gatekeeper SHALL return `severity=none` + +#### Scenario: Missing raw path is low confidence +- **WHEN** a binding references an existing invocation +- **AND** the binding omits `raw_path` +- **THEN** Gatekeeper SHALL return `status=fail` +- **AND** Gatekeeper SHALL return `severity=low_confid` +- **AND** the effective verifier result SHALL NOT be `PASS` + +#### Scenario: Old invocation without evidence refs is low confidence +- **WHEN** a binding references an existing invocation +- **AND** the invocation does not contain `retrieval_details.evidence_refs` +- **THEN** Gatekeeper SHALL return `status=fail` +- **AND** Gatekeeper SHALL return `severity=low_confid` +- **AND** the effective verifier result SHALL NOT be `PASS` + +#### Scenario: Unknown raw path is rejected +- **WHEN** a binding references an existing invocation +- **AND** the binding's `raw_path` is absent from that invocation's `retrieval_details.evidence_refs` +- **THEN** Gatekeeper SHALL return `status=fail` +- **AND** Gatekeeper SHALL return `severity=reject` +- **AND** `failed_rules` SHALL include `evidence.raw_path` + +#### Scenario: Mismatched excerpt is rejected +- **WHEN** a binding references an existing invocation and raw path +- **AND** the binding's `evidence_excerpt` is not supported by the matching system-side evidence ref text +- **THEN** Gatekeeper SHALL return `status=fail` +- **AND** Gatekeeper SHALL return `severity=reject` +- **AND** `failed_rules` SHALL include `evidence.excerpt_mismatch` + +### Requirement: Verifier SHALL use verified claim-local evidence for derivability +Verifier SHALL judge structured claims primarily against Gatekeeper-verified claim-local evidence excerpts. + +#### Scenario: Verified excerpt supports direct observation +- **WHEN** `gatekeeper_result.severity=none` +- **AND** a claim's verified evidence excerpts directly contain the claim's concrete facts +- **THEN** Verifier MAY classify that claim as `direct_observation` + +#### Scenario: Tool trace summary is navigation context +- **WHEN** `executor_structured_output.claims[].evidence_bindings` are available +- **THEN** Verifier SHALL use `tool_trace_summary` as navigation and audit context +- **AND** it SHALL NOT require `tool_trace_summary.output_summary` to contain every fact already present in verified claim-local evidence + +### Requirement: Gatekeeper severity SHALL constrain effective verdict +Runtime effective verdict calculation SHALL treat Gatekeeper severity as a hard upper bound. + +#### Scenario: Reject severity prevents PASS +- **WHEN** `gatekeeper_result.severity=reject` +- **AND** the Verifier model returns `verdict=PASS` +- **THEN** ChatService SHALL downgrade the effective verdict +- **AND** the effective verdict SHALL be `REJECT` + +#### Scenario: Low confidence severity prevents PASS +- **WHEN** `gatekeeper_result.severity=low_confid` +- **AND** the Verifier model returns `verdict=PASS` +- **THEN** ChatService SHALL downgrade the effective verdict +- **AND** the effective verdict SHALL be `LOW_CONFID` + +#### Scenario: Gatekeeper audit includes severity +- **WHEN** verifier evaluation is persisted +- **THEN** `diagnosis_session.self_evaluation.verifier_evaluation.gatekeeper_result` SHALL include `status`, `severity`, `checked_bindings`, `failed_rules`, `warnings`, and `errors` + +### Requirement: Verifier input hook SHALL only perform narrow compatibility backfill +The verifier input hook SHALL avoid converting broad tool summaries into precise evidence references. + +#### Scenario: Unique invocation candidate may be backfilled +- **WHEN** an evidence binding omits `source_invocation_id` +- **AND** exactly one current-session invocation exists for the binding's `tool_name` +- **THEN** the hook MAY backfill `source_invocation_id` +- **AND** it SHALL add a Gatekeeper warning describing the auto-backfill + +#### Scenario: Raw path is never backfilled +- **WHEN** an evidence binding omits `raw_path` +- **THEN** the hook SHALL NOT synthesize `raw_path` +- **AND** Gatekeeper SHALL treat the binding as not precise enough to pass + +### Requirement: Executor prompt SHALL constrain narrow-scope over-expansion +The Executor prompt SHALL instruct Executor to keep narrow confirmation questions focused on observation-level claims. + +#### Scenario: Narrow scope produces minimal observation claims +- **WHEN** the user asks to confirm one specific service, alert, or symptom +- **THEN** Executor SHOULD output the minimum necessary claims, normally one and at most two +- **AND** those claims SHALL be `observation` or `negative_observation` unless current-session evidence proves more +- **AND** Executor SHALL NOT emit unrelated root-cause, remediation, or excluded-topic claims as confirmed facts diff --git a/openspec/changes/archive/2026-07-08-verifier-evidence-reference-fidelity/specs/evidence-trace-hardening/spec.md b/openspec/changes/archive/2026-07-08-verifier-evidence-reference-fidelity/specs/evidence-trace-hardening/spec.md new file mode 100644 index 0000000..843432d --- /dev/null +++ b/openspec/changes/archive/2026-07-08-verifier-evidence-reference-fidelity/specs/evidence-trace-hardening/spec.md @@ -0,0 +1,42 @@ +## ADDED Requirements + +### Requirement: Evidence tools SHALL persist minimal evidence refs +Evidence-bearing tool invocations SHALL persist claim-addressable evidence references in `tool_invocation.retrieval_details.evidence_refs`. + +#### Scenario: Metrics alerts produce evidence refs +- **WHEN** a `query_metrics` invocation returns alert entries +- **THEN** the persisted retrieval details SHALL include one `evidence_refs` item per usable alert +- **AND** each item SHALL include `raw_path` formatted as `$.alerts[i]` +- **AND** each item SHALL include bounded `text` containing concrete alert facts such as alert name, state, service, current value, and duration when available + +#### Scenario: Logs produce evidence refs +- **WHEN** a `query_logs` invocation returns log entries +- **THEN** the persisted retrieval details SHALL include one `evidence_refs` item per usable log +- **AND** each item SHALL include `raw_path` formatted as `$.logs[i]` +- **AND** each item SHALL include bounded `text` containing concrete log facts such as timestamp, level, service, and message when available + +#### Scenario: Knowledge lookup produces evidence refs +- **WHEN** a `lookup_knowledge` invocation returns evidence blocks +- **THEN** the persisted retrieval details SHALL include one `evidence_refs` item per usable evidence block +- **AND** each item SHALL include `raw_path` formatted as `$.evidence_blocks[i]` +- **AND** each item SHALL include bounded `text` containing concrete block content, title, or source when available + +#### Scenario: Evidence ref extraction does not infer diagnosis +- **WHEN** the recorder creates `evidence_refs` +- **THEN** it SHALL only copy or format concrete tool output fields +- **AND** it SHALL NOT infer root cause, remediation, or diagnosis conclusions + +### Requirement: Log mock no-hit semantics SHALL avoid placeholder evidence +The log query mock SHALL distinguish positive mock evidence from no-hit results without using placeholder service logs as evidence. + +#### Scenario: HikariCP positive query returns order-service pool evidence +- **WHEN** a `query_logs` request targets `order-service` and HikariCP connection-pool exhaustion terms +- **THEN** the tool SHALL return order-service HikariCP-related log entries +- **AND** the returned evidence SHALL include concrete terms such as `HikariPool`, `active=50/50`, `waiting`, or `request timed out after 30000ms` +- **AND** it SHALL NOT return `generic-service` placeholder logs + +#### Scenario: HikariCP no-hit query returns no evidence +- **WHEN** a `query_logs` request targets a service without matching HikariCP mock evidence +- **THEN** the tool SHALL return an empty `logs` array +- **AND** it SHALL mark the output as `evidence_status=no_evidence` +- **AND** it SHALL NOT return `generic-service` placeholder logs diff --git a/openspec/changes/archive/2026-07-08-verifier-evidence-reference-fidelity/tasks.md b/openspec/changes/archive/2026-07-08-verifier-evidence-reference-fidelity/tasks.md new file mode 100644 index 0000000..50078ab --- /dev/null +++ b/openspec/changes/archive/2026-07-08-verifier-evidence-reference-fidelity/tasks.md @@ -0,0 +1,79 @@ +# Tasks + +## 1. Evidence reference extraction + +- [x] Add `evidence_refs` extraction in `ToolInvocationRecorder` for `query_metrics` alert arrays. +- [x] Add `evidence_refs` extraction in `ToolInvocationRecorder` for `query_logs` log arrays. +- [x] Add `evidence_refs` extraction in `ToolInvocationRecorder` for `lookup_knowledge` evidence blocks. +- [x] Add focused recorder tests for `$.alerts[i]`, `$.logs[i]`, and `$.evidence_blocks[i]`. + +Acceptance: + +- Persisted `retrieval_details` contains minimal `raw_path` and `text`. +- Extraction does not infer root cause or diagnosis. + +## 2. Gatekeeper reference fidelity + +- [x] Extend `ExecutorGatekeeperService` to read singular `source_invocation_id`, `raw_path`, and `evidence_excerpt`. +- [x] Keep legacy `source_invocation_ids` compatibility where needed, but require `raw_path` for precise pass. +- [x] Validate invocation existence, session ownership, tool name, `raw_path`, and excerpt similarity. +- [x] Add `severity` and checked binding details to Gatekeeper output. +- [x] Add tests for valid reference, missing `raw_path`, missing `evidence_refs`, unknown `raw_path`, mismatched excerpt, fabricated invocation ID, and tool mismatch. + +Acceptance: + +- Valid precise references pass. +- Missing precision downgrades to `LOW_CONFID`. +- Fabricated or mismatched references become `REJECT`. + +## 3. Verifier input hook and prompts + +- [x] Tighten `VerifierInputHook` auto-backfill to only unique single invocation candidates. +- [x] Prevent auto-filled bindings without `raw_path` from passing Gatekeeper. +- [x] Add auto-backfill warnings into Gatekeeper/audit output. +- [x] Update `chat-executor-prompt.md` with evidence-reference and narrow-scope constraints. +- [x] Update `chat-verifier-prompt.md` so verified excerpts are the primary derivability evidence. +- [x] Add focused hook and prompt-sensitive tests where practical. + +Acceptance: + +- Hook no longer bulk-fills invocation IDs. +- Verifier can see `gatekeeper_result.severity`. +- Executor is instructed to output precise references and avoid over-expansion. + +## 4. Effective verdict and audit persistence + +- [x] Ensure `severity=reject` prevents effective `PASS` and maps to `REJECT` when appropriate. +- [x] Ensure `severity=low_confid` prevents effective `PASS` and maps to `LOW_CONFID`. +- [x] Ensure `gatekeeper_result` with `severity` is persisted under `verifier_evaluation`. +- [x] Add ChatService or integration tests for effective verdict guardrails. + +Acceptance: + +- Gatekeeper fail cannot become final PASS. +- Audit JSON contains the minimum Gatekeeper fields. + +## 5. HikariCP mock quality + +- [x] Add positive HikariCP mock logs for `order-service`. +- [x] Support HikariCP synonym matching. +- [x] Remove `generic-service` placeholder evidence for no-hit cases. +- [x] Add tests for HikariCP positive and negative no-hit behavior. + +Acceptance: + +- Positive HikariCP query returns order-service logs. +- Negative HikariCP query returns `logs=[]` and `evidence_status=no_evidence`. + +## 6. Verification + +- [x] Run focused unit tests for recorder, Gatekeeper, hook, ChatService guardrails, and query log mock behavior. +- [x] Run `openspec validate verifier-evidence-reference-fidelity --strict`. +- [x] Run `openspec validate --specs`. +- [x] Start the Java project using `mvn spring-boot:run`. +- [x] Run end-to-end checks for HighMemoryUsage positive, SlowResponse positive, HikariCP positive, HikariCP negative, and narrow forbidden claim. +- [x] Query MySQL audit data with `scripts/query_mysql.py` to confirm persisted `gatekeeper_result`. + +Acceptance: + +- Minimum E2E matrix passes or any failure is classified as code issue, mock quality issue, retrieval/tool quality issue, or model nondeterminism with evidence. diff --git a/openspec/specs/chat-verifier-agent/spec.md b/openspec/specs/chat-verifier-agent/spec.md index ade4618..d172702 100644 --- a/openspec/specs/chat-verifier-agent/spec.md +++ b/openspec/specs/chat-verifier-agent/spec.md @@ -1,4 +1,4 @@ -# chat-verifier-agent Specification +# chat-verifier-agent Specification ## Purpose TBD - created by archiving change chat-verifier-agent. Update Purpose after archive. @@ -119,6 +119,7 @@ The system SHALL use ChatService for explicit single-round `Planner -> Executor - **THEN** the system SHALL output a degraded result indicating the answer cannot be reliably generated - **AND** it SHALL NOT pass through the raw Executor answer - **AND** it SHALL NOT include a root-cause conclusion + ### Requirement: Verifier SHALL be observable The Verifier's verdict and downstream final-answer composition SHALL be persisted for observability. @@ -421,3 +422,112 @@ The system SHALL use Composer or fixed safe templates for PASS, LOW_CONFID, and - **AND** it SHALL include only confirmed facts, evidence gaps, and next-step suggestions - **AND** it SHALL NOT include unverified raw answer content - **AND** it SHALL NOT include a root-cause conclusion + +### Requirement: Executor evidence bindings SHALL support precise evidence references +Executor V2 evidence bindings SHALL support precise evidence references that locate evidence inside a persisted tool invocation. + +#### Scenario: Precise evidence binding contains invocation path and excerpt +- **WHEN** Executor binds evidence to a claim +- **THEN** the binding SHOULD include singular `source_invocation_id` +- **AND** the binding SHOULD include `raw_path` +- **AND** the binding SHALL include `tool_name` and `evidence_excerpt` +- **AND** the `raw_path` SHALL be interpreted relative to the referenced tool invocation's `retrieval_details.evidence_refs` + +#### Scenario: Legacy plural invocation ids remain compatibility only +- **WHEN** Executor emits legacy `source_invocation_ids` +- **THEN** the system MAY read them for compatibility +- **AND** they SHALL NOT be sufficient for a precise Gatekeeper pass without `raw_path` + +### Requirement: Gatekeeper SHALL validate evidence reference fidelity +Gatekeeper SHALL validate that Executor evidence bindings point to real current-session evidence references before Verifier uses them as primary evidence. + +#### Scenario: Valid precise binding passes +- **WHEN** a binding's `source_invocation_id` exists in the current session +- **AND** the binding's `tool_name` matches the persisted invocation +- **AND** the binding's `raw_path` exists in `retrieval_details.evidence_refs` +- **AND** the binding's `evidence_excerpt` is supported by the matching evidence ref text +- **THEN** Gatekeeper SHALL return `status=pass` +- **AND** Gatekeeper SHALL return `severity=none` + +#### Scenario: Missing raw path is low confidence +- **WHEN** a binding references an existing invocation +- **AND** the binding omits `raw_path` +- **THEN** Gatekeeper SHALL return `status=fail` +- **AND** Gatekeeper SHALL return `severity=low_confid` +- **AND** the effective verifier result SHALL NOT be `PASS` + +#### Scenario: Old invocation without evidence refs is low confidence +- **WHEN** a binding references an existing invocation +- **AND** the invocation does not contain `retrieval_details.evidence_refs` +- **THEN** Gatekeeper SHALL return `status=fail` +- **AND** Gatekeeper SHALL return `severity=low_confid` +- **AND** the effective verifier result SHALL NOT be `PASS` + +#### Scenario: Unknown raw path is rejected +- **WHEN** a binding references an existing invocation +- **AND** the binding's `raw_path` is absent from that invocation's `retrieval_details.evidence_refs` +- **THEN** Gatekeeper SHALL return `status=fail` +- **AND** Gatekeeper SHALL return `severity=reject` +- **AND** `failed_rules` SHALL include `evidence.raw_path` + +#### Scenario: Mismatched excerpt is rejected +- **WHEN** a binding references an existing invocation and raw path +- **AND** the binding's `evidence_excerpt` is not supported by the matching system-side evidence ref text +- **THEN** Gatekeeper SHALL return `status=fail` +- **AND** Gatekeeper SHALL return `severity=reject` +- **AND** `failed_rules` SHALL include `evidence.excerpt_mismatch` + +### Requirement: Verifier SHALL use verified claim-local evidence for derivability +Verifier SHALL judge structured claims primarily against Gatekeeper-verified claim-local evidence excerpts. + +#### Scenario: Verified excerpt supports direct observation +- **WHEN** `gatekeeper_result.severity=none` +- **AND** a claim's verified evidence excerpts directly contain the claim's concrete facts +- **THEN** Verifier MAY classify that claim as `direct_observation` + +#### Scenario: Tool trace summary is navigation context +- **WHEN** `executor_structured_output.claims[].evidence_bindings` are available +- **THEN** Verifier SHALL use `tool_trace_summary` as navigation and audit context +- **AND** it SHALL NOT require `tool_trace_summary.output_summary` to contain every fact already present in verified claim-local evidence + +### Requirement: Gatekeeper severity SHALL constrain effective verdict +Runtime effective verdict calculation SHALL treat Gatekeeper severity as a hard upper bound. + +#### Scenario: Reject severity prevents PASS +- **WHEN** `gatekeeper_result.severity=reject` +- **AND** the Verifier model returns `verdict=PASS` +- **THEN** ChatService SHALL downgrade the effective verdict +- **AND** the effective verdict SHALL be `REJECT` + +#### Scenario: Low confidence severity prevents PASS +- **WHEN** `gatekeeper_result.severity=low_confid` +- **AND** the Verifier model returns `verdict=PASS` +- **THEN** ChatService SHALL downgrade the effective verdict +- **AND** the effective verdict SHALL be `LOW_CONFID` + +#### Scenario: Gatekeeper audit includes severity +- **WHEN** verifier evaluation is persisted +- **THEN** `diagnosis_session.self_evaluation.verifier_evaluation.gatekeeper_result` SHALL include `status`, `severity`, `checked_bindings`, `failed_rules`, `warnings`, and `errors` + +### Requirement: Verifier input hook SHALL only perform narrow compatibility backfill +The verifier input hook SHALL avoid converting broad tool summaries into precise evidence references. + +#### Scenario: Unique invocation candidate may be backfilled +- **WHEN** an evidence binding omits `source_invocation_id` +- **AND** exactly one current-session invocation exists for the binding's `tool_name` +- **THEN** the hook MAY backfill `source_invocation_id` +- **AND** it SHALL add a Gatekeeper warning describing the auto-backfill + +#### Scenario: Raw path is never backfilled +- **WHEN** an evidence binding omits `raw_path` +- **THEN** the hook SHALL NOT synthesize `raw_path` +- **AND** Gatekeeper SHALL treat the binding as not precise enough to pass + +### Requirement: Executor prompt SHALL constrain narrow-scope over-expansion +The Executor prompt SHALL instruct Executor to keep narrow confirmation questions focused on observation-level claims. + +#### Scenario: Narrow scope produces minimal observation claims +- **WHEN** the user asks to confirm one specific service, alert, or symptom +- **THEN** Executor SHOULD output the minimum necessary claims, normally one and at most two +- **AND** those claims SHALL be `observation` or `negative_observation` unless current-session evidence proves more +- **AND** Executor SHALL NOT emit unrelated root-cause, remediation, or excluded-topic claims as confirmed facts diff --git a/openspec/specs/evidence-trace-hardening/spec.md b/openspec/specs/evidence-trace-hardening/spec.md index 128714d..3f9889e 100644 --- a/openspec/specs/evidence-trace-hardening/spec.md +++ b/openspec/specs/evidence-trace-hardening/spec.md @@ -108,3 +108,44 @@ The persisted trace SHALL make it possible to audit model step counts separately - **AND** `diagnosis_session.tool_call_count` SHALL count persisted evidence-tool invocation rows - **AND** helper workflow calls that are not evidence rows SHALL be auditable from agent steps or logs without inflating `tool_invocation` +### Requirement: Evidence tools SHALL persist minimal evidence refs +Evidence-bearing tool invocations SHALL persist claim-addressable evidence references in `tool_invocation.retrieval_details.evidence_refs`. + +#### Scenario: Metrics alerts produce evidence refs +- **WHEN** a `query_metrics` invocation returns alert entries +- **THEN** the persisted retrieval details SHALL include one `evidence_refs` item per usable alert +- **AND** each item SHALL include `raw_path` formatted as `$.alerts[i]` +- **AND** each item SHALL include bounded `text` containing concrete alert facts such as alert name, state, service, current value, and duration when available + +#### Scenario: Logs produce evidence refs +- **WHEN** a `query_logs` invocation returns log entries +- **THEN** the persisted retrieval details SHALL include one `evidence_refs` item per usable log +- **AND** each item SHALL include `raw_path` formatted as `$.logs[i]` +- **AND** each item SHALL include bounded `text` containing concrete log facts such as timestamp, level, service, and message when available + +#### Scenario: Knowledge lookup produces evidence refs +- **WHEN** a `lookup_knowledge` invocation returns evidence blocks +- **THEN** the persisted retrieval details SHALL include one `evidence_refs` item per usable evidence block +- **AND** each item SHALL include `raw_path` formatted as `$.evidence_blocks[i]` +- **AND** each item SHALL include bounded `text` containing concrete block content, title, or source when available + +#### Scenario: Evidence ref extraction does not infer diagnosis +- **WHEN** the recorder creates `evidence_refs` +- **THEN** it SHALL only copy or format concrete tool output fields +- **AND** it SHALL NOT infer root cause, remediation, or diagnosis conclusions + +### Requirement: Log mock no-hit semantics SHALL avoid placeholder evidence +The log query mock SHALL distinguish positive mock evidence from no-hit results without using placeholder service logs as evidence. + +#### Scenario: HikariCP positive query returns order-service pool evidence +- **WHEN** a `query_logs` request targets `order-service` and HikariCP connection-pool exhaustion terms +- **THEN** the tool SHALL return order-service HikariCP-related log entries +- **AND** the returned evidence SHALL include concrete terms such as `HikariPool`, `active=50/50`, `waiting`, or `request timed out after 30000ms` +- **AND** it SHALL NOT return `generic-service` placeholder logs + +#### Scenario: HikariCP no-hit query returns no evidence +- **WHEN** a `query_logs` request targets a service without matching HikariCP mock evidence +- **THEN** the tool SHALL return an empty `logs` array +- **AND** it SHALL mark the output as `evidence_status=no_evidence` +- **AND** it SHALL NOT return `generic-service` placeholder logs + diff --git a/src/main/java/com/superbiz/agent/agent/tool/QueryLogsTools.java b/src/main/java/com/superbiz/agent/agent/tool/QueryLogsTools.java index d432981..6b00c78 100644 --- a/src/main/java/com/superbiz/agent/agent/tool/QueryLogsTools.java +++ b/src/main/java/com/superbiz/agent/agent/tool/QueryLogsTools.java @@ -289,13 +289,9 @@ public class QueryLogsTools { logs.addAll(buildSystemEventsLogs(now, normalizedQuery, limit)); break; default: - logs.addAll(buildGenericLogs(now, normalizedQuery, limit)); + return logs; } - - if (logs.isEmpty()) { - logs.addAll(buildGenericLogs(now, normalizedQuery, limit)); - } - + // 限制返回条数 if (logs.size() > limit) { @@ -400,6 +396,10 @@ public class QueryLogsTools { */ private List buildApplicationLogs(Instant now, String query, int limit) { List logs = new ArrayList<>(); + + if (isHikariPoolQuery(query) && targetsOrderService(query)) { + logs.addAll(buildHikariPoolLogs(now)); + } // ERROR 级别日志 if (query.contains("error") || query.contains("fatal") || query.contains("500")) { @@ -514,6 +514,58 @@ public class QueryLogsTools { return logs; } + + private boolean isHikariPoolQuery(String query) { + return query.contains("hikaricp") + || query.contains("hikaripool") + || query.contains("connection pool") + || query.contains("数据库连接池") + || query.contains("连接池耗尽") + || query.contains("active=50/50") + || query.contains("waiting") + || query.contains("request timed out after 30000ms"); + } + + private boolean targetsOrderService(String query) { + if (query.contains("inventory-service") || query.contains("payment-service") || query.contains("user-service")) { + return false; + } + return query.contains("order-service") || isHikariPoolQuery(query); + } + + private List buildHikariPoolLogs(Instant now) { + List logs = new ArrayList<>(); + + LogEntry timeout = new LogEntry(); + timeout.setTimestamp(FORMATTER.format(now.minus(2, ChronoUnit.MINUTES))); + timeout.setLevel("ERROR"); + timeout.setService("order-service"); + timeout.setInstance("pod-order-service-5c7d8e9f1-m3n2p"); + timeout.setMessage("HikariPool-1 - Connection is not available, request timed out after 30000ms"); + timeout.setMetrics(Map.of( + "pool", "HikariPool-1", + "error_type", "ConnectionPoolExhaustedException", + "timeout_ms", "30000" + )); + logs.add(timeout); + + LogEntry stats = new LogEntry(); + stats.setTimestamp(FORMATTER.format(now.minus(1, ChronoUnit.MINUTES))); + stats.setLevel("WARN"); + stats.setService("order-service"); + stats.setInstance("pod-order-service-5c7d8e9f1-m3n2p"); + stats.setMessage("HikariCP pool stats: active=50/50, idle=0, waiting=32"); + stats.setMetrics(Map.of( + "pool", "HikariPool-1", + "active", "50", + "max", "50", + "idle", "0", + "waiting", "32" + )); + logs.add(stats); + + return logs; + } /** * 构建数据库慢查询日志(与慢响应告警关联) diff --git a/src/main/java/com/superbiz/agent/hook/VerifierInputHook.java b/src/main/java/com/superbiz/agent/hook/VerifierInputHook.java index 70a93e8..3831102 100644 --- a/src/main/java/com/superbiz/agent/hook/VerifierInputHook.java +++ b/src/main/java/com/superbiz/agent/hook/VerifierInputHook.java @@ -17,9 +17,12 @@ import org.springframework.ai.chat.messages.AssistantMessage; import org.springframework.ai.chat.messages.Message; import org.springframework.ai.chat.messages.UserMessage; +import java.util.ArrayList; import java.util.LinkedHashMap; +import java.util.LinkedHashSet; import java.util.List; import java.util.Map; +import java.util.Set; /** * Replaces verifier history with an explicit structured payload. @@ -60,14 +63,18 @@ public class VerifierInputHook extends MessagesModelHook { executorFinalAnswer = extractLastAssistantText(previousMessages); } - ExecutorOutputParseResult parseResult = parseExecutorOutput(executorFinalAnswer); - VerifierContextHolder.setExecutorStructuredOutput(parseResult.structuredOutput()); - VerifierContextHolder.setExecutorOutputParseStatus(parseResult.status()); - List> toolTraceSummary = toolTraceSummaryService.buildVerifierTraceSummary(sessionId, executorFinalAnswer); VerifierContextHolder.setToolTraceSummary(toolTraceSummary); + ExecutorOutputParseResult parseResult = parseExecutorOutput(executorFinalAnswer); + parseResult = new ExecutorOutputParseResult( + enrichExecutorStructuredOutput(parseResult.structuredOutput(), toolTraceSummary), + parseResult.status() + ); + VerifierContextHolder.setExecutorStructuredOutput(parseResult.structuredOutput()); + VerifierContextHolder.setExecutorOutputParseStatus(parseResult.status()); + Map gatekeeperResult = runGatekeeper(sessionId, parseResult); VerifierContextHolder.setGatekeeperResult(gatekeeperResult); @@ -105,6 +112,8 @@ public class VerifierInputHook extends MessagesModelHook { private Map passGatekeeperResult() { Map result = new LinkedHashMap<>(); result.put("status", "pass"); + result.put("severity", "none"); + result.put("checked_bindings", List.of()); result.put("failed_rules", List.of()); result.put("warnings", List.of()); result.put("errors", List.of()); @@ -135,6 +144,129 @@ public class VerifierInputHook extends MessagesModelHook { } } + @SuppressWarnings("unchecked") + private Map enrichExecutorStructuredOutput(Map structuredOutput, + List> toolTraceSummary) { + if (structuredOutput == null) { + return null; + } + Map> invocationIdsByTool = invocationIdsByTool(toolTraceSummary); + List> warnings = new ArrayList<>(); + enrichEvidenceBindingsInSection(structuredOutput.get("claims"), invocationIdsByTool, warnings); + enrichEvidenceBindingsInSection(structuredOutput.get("recommended_actions"), invocationIdsByTool, warnings); + if (!warnings.isEmpty()) { + structuredOutput.put("_gatekeeper_warnings", warnings); + } + return structuredOutput; + } + + @SuppressWarnings("unchecked") + private void enrichEvidenceBindingsInSection(Object sectionValue, + Map> invocationIdsByTool, + List> warnings) { + if (!(sectionValue instanceof List items)) { + return; + } + for (Object itemValue : items) { + if (!(itemValue instanceof Map item)) { + continue; + } + Object bindingsValue = item.get("evidence_bindings"); + if (!(bindingsValue instanceof List bindings)) { + continue; + } + for (Object bindingValue : bindings) { + if (!(bindingValue instanceof Map rawBinding)) { + continue; + } + Map binding = (Map) rawBinding; + String normalizedToolName = normalizeToolName(binding.get("tool_name")); + if (!normalizedToolName.isBlank()) { + binding.put("tool_name", normalizedToolName); + } + if (!hasInvocationId(binding)) { + List ids = invocationIdsByTool.getOrDefault(normalizedToolName, List.of()); + if (ids.size() == 1) { + binding.put("source_invocation_id", ids.get(0)); + warnings.add(Map.of( + "rule", "evidence.invocation_auto_backfill", + "message", "source_invocation_id was auto-filled from the unique tool invocation candidate; raw_path remains missing if Executor did not provide it", + "tool_name", normalizedToolName, + "source_invocation_id", ids.get(0) + )); + } + } + } + } + } + + private Map> invocationIdsByTool(List> toolTraceSummary) { + Map> idsByTool = new LinkedHashMap<>(); + for (Map summary : toolTraceSummary == null ? List.>of() : toolTraceSummary) { + String toolName = normalizeToolName(summary.get("tool_name")); + if (toolName.isBlank()) { + continue; + } + List ids = toLongList(summary.get("source_invocation_ids")); + if (ids.isEmpty()) { + continue; + } + idsByTool.computeIfAbsent(toolName, ignored -> new LinkedHashSet<>()).addAll(ids); + } + + Map> result = new LinkedHashMap<>(); + for (Map.Entry> entry : idsByTool.entrySet()) { + result.put(entry.getKey(), new ArrayList<>(entry.getValue())); + } + return result; + } + + private boolean hasInvocationId(Map binding) { + if (asLong(binding.get("source_invocation_id")) != null) { + return true; + } + return toLongList(binding.get("source_invocation_ids")).size() == 1; + } + + private List toLongList(Object value) { + if (!(value instanceof List values)) { + return List.of(); + } + List ids = new ArrayList<>(); + for (Object item : values) { + Long id = asLong(item); + if (id != null) { + ids.add(id); + } + } + return ids; + } + + private Long asLong(Object value) { + if (value instanceof Number number) { + return number.longValue(); + } + if (value instanceof String text) { + try { + return Long.parseLong(text); + } catch (NumberFormatException ignored) { + return null; + } + } + return null; + } + + private String normalizeToolName(Object value) { + String toolName = value == null ? "" : String.valueOf(value); + return switch (toolName) { + case "lookupKnowledge" -> "lookup_knowledge"; + case "queryLogs" -> "query_logs"; + case "queryPrometheusAlerts" -> "query_metrics"; + case "getAvailableLogTopics" -> "get_available_log_topics"; + default -> toolName; + }; + } + private String sanitizeJsonPayload(String raw) { String trimmed = raw.trim(); int fenceStart = trimmed.indexOf("```"); diff --git a/src/main/java/com/superbiz/agent/service/ChatService.java b/src/main/java/com/superbiz/agent/service/ChatService.java index 8417ef1..9ae7b7e 100644 --- a/src/main/java/com/superbiz/agent/service/ChatService.java +++ b/src/main/java/com/superbiz/agent/service/ChatService.java @@ -711,6 +711,13 @@ public class ChatService { if (gatekeeperResult == null || !"fail".equals(String.valueOf(gatekeeperResult.get("status")))) { return verdict; } + String severity = String.valueOf(gatekeeperResult.getOrDefault("severity", "")); + if (ExecutorGatekeeperService.SEVERITY_REJECT.equals(severity)) { + return "REJECT"; + } + if (ExecutorGatekeeperService.SEVERITY_LOW_CONFID.equals(severity)) { + return "PASS".equals(verdict) ? "LOW_CONFID" : verdict; + } if (containsRule(gatekeeperResult.get("failed_rules"), ExecutorGatekeeperService.RULE_INVOCATION_REF)) { return "REJECT"; } @@ -885,7 +892,8 @@ public class ChatService { Optional.ofNullable(VerifierContextHolder.getToolTraceSummary()).orElse(List.of())); verifierEvaluation.put("gatekeeper_result", Optional.ofNullable(VerifierContextHolder.getGatekeeperResult()) - .orElse(Map.of("status", "pass", "failed_rules", List.of(), "warnings", List.of(), "errors", List.of()))); + .orElse(Map.of("status", "pass", "severity", "none", "checked_bindings", List.of(), + "failed_rules", List.of(), "warnings", List.of(), "errors", List.of()))); if (composerOutput != null) { verifierEvaluation.put("composer_output", composerOutput); } diff --git a/src/main/java/com/superbiz/agent/service/ExecutorGatekeeperService.java b/src/main/java/com/superbiz/agent/service/ExecutorGatekeeperService.java index ae18eff..ccbe637 100644 --- a/src/main/java/com/superbiz/agent/service/ExecutorGatekeeperService.java +++ b/src/main/java/com/superbiz/agent/service/ExecutorGatekeeperService.java @@ -1,5 +1,7 @@ package com.superbiz.agent.service; +import com.fasterxml.jackson.core.type.TypeReference; +import com.fasterxml.jackson.databind.ObjectMapper; import com.superbiz.agent.domain.entity.ToolInvocation; import com.superbiz.agent.repository.ToolInvocationRepository; import org.springframework.stereotype.Service; @@ -21,12 +23,22 @@ import java.util.stream.Collectors; public class ExecutorGatekeeperService { public static final String STATUS_PASS = "pass"; - public static final String STATUS_WARN = "warn"; public static final String STATUS_FAIL = "fail"; + public static final String SEVERITY_NONE = "none"; + public static final String SEVERITY_LOW_CONFID = "low_confid"; + public static final String SEVERITY_REJECT = "reject"; public static final String RULE_SCHEMA = "schema.executor_v2"; public static final String RULE_INVOCATION_REF = "evidence.invocation_ref"; + public static final String RULE_RAW_PATH = "evidence.raw_path"; + public static final String RULE_EXCERPT_MISMATCH = "evidence.excerpt_mismatch"; + public static final String RULE_EVIDENCE_MISSING = "evidence.missing"; + + private static final TypeReference> MAP_TYPE = new TypeReference<>() { + }; + private static final double MIN_TOKEN_OVERLAP = 0.5; private final ToolInvocationRepository toolInvocationRepository; + private final ObjectMapper objectMapper = new ObjectMapper(); public ExecutorGatekeeperService(ToolInvocationRepository toolInvocationRepository) { this.toolInvocationRepository = toolInvocationRepository; @@ -39,6 +51,7 @@ public class ExecutorGatekeeperService { validateSchema(structuredOutput, parseStatus, result); if (structuredOutput != null) { validateInvocationRefs(sessionId, structuredOutput, result); + importWarnings(structuredOutput, result); } return result.toMap(); } @@ -49,7 +62,7 @@ public class ExecutorGatekeeperService { public Map fail(String ruleId, String target, String message) { GatekeeperResult result = new GatekeeperResult(); - result.fail(ruleId, target, message); + result.fail(ruleId, target, message, SEVERITY_REJECT); return result.toMap(); } @@ -59,31 +72,35 @@ public class ExecutorGatekeeperService { String status = parseStatus == null ? "" : String.valueOf(parseStatus.getOrDefault("status", "")); if (structuredOutput == null) { if ("valid".equals(status)) { - result.fail(RULE_SCHEMA, "executor_structured_output", "structured output is missing after valid parse"); + result.fail(RULE_SCHEMA, "executor_structured_output", + "structured output is missing after valid parse", SEVERITY_LOW_CONFID); } return; } if (!"executor_evidence_v2".equals(String.valueOf(structuredOutput.get("answer_version")))) { - result.fail(RULE_SCHEMA, "answer_version", "answer_version must be executor_evidence_v2"); + result.fail(RULE_SCHEMA, "answer_version", "answer_version must be executor_evidence_v2", + SEVERITY_LOW_CONFID); } if (structuredOutput.containsKey("diagnosis_summary")) { - result.fail(RULE_SCHEMA, "diagnosis_summary", "diagnosis_summary is removed from executor_evidence_v2"); + result.fail(RULE_SCHEMA, "diagnosis_summary", "diagnosis_summary is removed from executor_evidence_v2", + SEVERITY_REJECT); } if (structuredOutput.containsKey("user_facing_answer")) { - result.fail(RULE_SCHEMA, "user_facing_answer", "user_facing_answer is removed from executor_evidence_v2"); + result.fail(RULE_SCHEMA, "user_facing_answer", "user_facing_answer is removed from executor_evidence_v2", + SEVERITY_REJECT); } Object claimsValue = structuredOutput.get("claims"); if (!(claimsValue instanceof List claims)) { - result.fail(RULE_SCHEMA, "claims", "claims must be an array"); + result.fail(RULE_SCHEMA, "claims", "claims must be an array", SEVERITY_LOW_CONFID); return; } for (int i = 0; i < claims.size(); i++) { String target = "claims[" + i + "]"; Object claimValue = claims.get(i); if (!(claimValue instanceof Map claim)) { - result.fail(RULE_SCHEMA, target, "claim must be an object"); + result.fail(RULE_SCHEMA, target, "claim must be an object", SEVERITY_LOW_CONFID); continue; } requireString(claim, "claim_id", target, result); @@ -91,11 +108,13 @@ public class ExecutorGatekeeperService { requireString(claim, "claim_text", target, result); String supportLevel = stringValue(claim.get("support_level")); if (!"direct".equals(supportLevel) && !"indirect".equals(supportLevel)) { - result.fail(RULE_SCHEMA, target + ".support_level", "support_level must be direct or indirect"); + result.fail(RULE_SCHEMA, target + ".support_level", "support_level must be direct or indirect", + SEVERITY_LOW_CONFID); } Object bindings = claim.get("evidence_bindings"); if (!(bindings instanceof List bindingList) || bindingList.isEmpty()) { - result.fail(RULE_SCHEMA, target + ".evidence_bindings", "claims must include non-empty evidence_bindings"); + result.fail(RULE_EVIDENCE_MISSING, target + ".evidence_bindings", + "claims must include non-empty evidence_bindings", SEVERITY_LOW_CONFID); } } @@ -106,7 +125,8 @@ public class ExecutorGatekeeperService { private void validateInvocationRefs(String sessionId, Map structuredOutput, GatekeeperResult result) { if (sessionId == null || sessionId.isBlank()) { - result.fail(RULE_INVOCATION_REF, "session_id", "session id is required to validate source_invocation_ids"); + result.fail(RULE_INVOCATION_REF, "session_id", "session id is required to validate source_invocation_id", + SEVERITY_LOW_CONFID); return; } @@ -132,62 +152,254 @@ public class ExecutorGatekeeperService { String target = "claims[" + claimIndex + "].evidence_bindings[" + bindingIndex + "]"; Object bindingValue = bindings.get(bindingIndex); if (!(bindingValue instanceof Map binding)) { - result.fail(RULE_INVOCATION_REF, target, "evidence binding must be an object"); + result.fail(RULE_INVOCATION_REF, target, "evidence binding must be an object", + SEVERITY_LOW_CONFID); continue; } - validateBindingInvocationIds(binding, validInvocations, target, result); + validateEvidenceBinding(binding, validInvocations, target, claim.get("claim_id"), result); + } + } + + Object actionsValue = structuredOutput.get("recommended_actions"); + if (!(actionsValue instanceof List actions)) { + return; + } + for (int actionIndex = 0; actionIndex < actions.size(); actionIndex++) { + Object actionValue = actions.get(actionIndex); + if (!(actionValue instanceof Map action)) { + continue; + } + Object bindingsValue = action.get("evidence_bindings"); + if (!(bindingsValue instanceof List bindings)) { + continue; + } + for (int bindingIndex = 0; bindingIndex < bindings.size(); bindingIndex++) { + String target = "recommended_actions[" + actionIndex + "].evidence_bindings[" + bindingIndex + "]"; + Object bindingValue = bindings.get(bindingIndex); + if (!(bindingValue instanceof Map binding)) { + result.fail(RULE_INVOCATION_REF, target, "evidence binding must be an object", + SEVERITY_LOW_CONFID); + continue; + } + validateEvidenceBinding(binding, validInvocations, target, action.get("action_id"), result); } } } - private void validateBindingInvocationIds(Map binding, - Map validInvocations, - String target, - GatekeeperResult result) { - Object idsValue = binding.get("source_invocation_ids"); - if (!(idsValue instanceof List ids) || ids.isEmpty()) { - result.fail(RULE_INVOCATION_REF, target + ".source_invocation_ids", - "source_invocation_ids must be a non-empty array"); + private void validateEvidenceBinding(Map binding, + Map validInvocations, + String target, + Object ownerId, + GatekeeperResult result) { + Map checked = new LinkedHashMap<>(); + checked.put("claim_id", ownerId == null ? "" : String.valueOf(ownerId)); + checked.put("tool_name", stringValue(binding.get("tool_name"))); + checked.put("source_invocation_id", binding.get("source_invocation_id")); + checked.put("raw_path", stringValue(binding.get("raw_path"))); + + Long id = singleInvocationId(binding); + if (id == null) { + checked.put("status", STATUS_FAIL); + checked.put("rule", RULE_INVOCATION_REF); + checked.put("message", "source_invocation_id is required"); + result.checked(checked); + result.fail(RULE_INVOCATION_REF, target + ".source_invocation_id", + "source_invocation_id is required", SEVERITY_LOW_CONFID); return; } + checked.put("source_invocation_id", id); String claimedToolName = stringValue(binding.get("tool_name")); if (claimedToolName.isBlank()) { - result.fail(RULE_INVOCATION_REF, target + ".tool_name", "tool_name is required"); + checked.put("status", STATUS_FAIL); + checked.put("rule", RULE_INVOCATION_REF); + checked.put("message", "tool_name is required"); + result.checked(checked); + result.fail(RULE_INVOCATION_REF, target + ".tool_name", "tool_name is required", SEVERITY_LOW_CONFID); + return; } - Set checkedIds = new HashSet<>(); - for (Object idValue : ids) { - Long id = asLong(idValue); - if (id == null) { - result.fail(RULE_INVOCATION_REF, target + ".source_invocation_ids", - "source_invocation_ids must contain numeric ids"); - continue; - } - if (!checkedIds.add(id)) { - continue; - } - ToolInvocation invocation = validInvocations.get(id); - if (invocation == null) { - result.fail(RULE_INVOCATION_REF, target, "source_invocation_ids not found in current session: " + id); - continue; - } - if (!claimedToolName.isBlank() && !Objects.equals(claimedToolName, invocation.getToolName())) { - result.fail(RULE_INVOCATION_REF, target + ".tool_name", - "tool_name does not match invocation " + id + ": expected " + invocation.getToolName()); + ToolInvocation invocation = validInvocations.get(id); + if (invocation == null) { + checked.put("status", STATUS_FAIL); + checked.put("rule", RULE_INVOCATION_REF); + checked.put("message", "source_invocation_id not found in current session: " + id); + result.checked(checked); + result.fail(RULE_INVOCATION_REF, target, + "source_invocation_id not found in current session: " + id, SEVERITY_REJECT); + return; + } + if (!Objects.equals(claimedToolName, invocation.getToolName())) { + checked.put("status", STATUS_FAIL); + checked.put("rule", RULE_INVOCATION_REF); + checked.put("message", "tool_name does not match invocation " + id + ": expected " + invocation.getToolName()); + result.checked(checked); + result.fail(RULE_INVOCATION_REF, target + ".tool_name", + "tool_name does not match invocation " + id + ": expected " + invocation.getToolName(), + SEVERITY_REJECT); + return; + } + + String rawPath = stringValue(binding.get("raw_path")); + if (rawPath.isBlank()) { + checked.put("status", STATUS_FAIL); + checked.put("rule", RULE_RAW_PATH); + checked.put("message", "raw_path is required for precise evidence reference"); + result.checked(checked); + result.fail(RULE_RAW_PATH, target + ".raw_path", + "raw_path is required for precise evidence reference", SEVERITY_LOW_CONFID); + return; + } + + Map refs = evidenceRefsByRawPath(invocation.getRetrievalDetails()); + if (refs.isEmpty()) { + checked.put("status", STATUS_FAIL); + checked.put("rule", RULE_EVIDENCE_MISSING); + checked.put("message", "invocation has no retrieval_details.evidence_refs"); + result.checked(checked); + result.fail(RULE_EVIDENCE_MISSING, target, + "invocation has no retrieval_details.evidence_refs", SEVERITY_LOW_CONFID); + return; + } + + String matchedText = refs.get(rawPath); + if (matchedText == null) { + checked.put("status", STATUS_FAIL); + checked.put("rule", RULE_RAW_PATH); + checked.put("message", "raw_path not found in retrieval_details.evidence_refs"); + result.checked(checked); + result.fail(RULE_RAW_PATH, target + ".raw_path", + "raw_path not found in retrieval_details.evidence_refs", SEVERITY_REJECT); + return; + } + checked.put("matched_text", matchedText); + + String excerpt = stringValue(binding.get("evidence_excerpt")); + if (normalized(excerpt).length() < 8) { + checked.put("status", STATUS_FAIL); + checked.put("rule", RULE_EXCERPT_MISMATCH); + checked.put("message", "evidence_excerpt is too short to compare"); + result.checked(checked); + result.fail(RULE_EXCERPT_MISMATCH, target + ".evidence_excerpt", + "evidence_excerpt is too short to compare", SEVERITY_LOW_CONFID); + return; + } + if (!isExcerptSupported(excerpt, matchedText)) { + checked.put("status", STATUS_FAIL); + checked.put("rule", RULE_EXCERPT_MISMATCH); + checked.put("message", "evidence_excerpt is not supported by matched evidence ref text"); + result.checked(checked); + result.fail(RULE_EXCERPT_MISMATCH, target + ".evidence_excerpt", + "evidence_excerpt is not supported by matched evidence ref text", SEVERITY_REJECT); + return; + } + + checked.put("status", STATUS_PASS); + result.checked(checked); + } + + private Long singleInvocationId(Map binding) { + Long singular = asLong(binding.get("source_invocation_id")); + if (singular != null) { + return singular; + } + Object idsValue = binding.get("source_invocation_ids"); + if (!(idsValue instanceof List ids) || ids.size() != 1) { + return null; + } + return asLong(ids.get(0)); + } + + @SuppressWarnings("unchecked") + private void importWarnings(Map structuredOutput, GatekeeperResult result) { + Object warningsValue = structuredOutput.remove("_gatekeeper_warnings"); + if (!(warningsValue instanceof List warnings)) { + return; + } + for (Object warning : warnings) { + if (warning instanceof Map map) { + result.warn((Map) map); + } else if (warning != null) { + result.warn(Map.of("message", String.valueOf(warning))); } } } + private Map evidenceRefsByRawPath(String retrievalDetails) { + if (retrievalDetails == null || retrievalDetails.isBlank()) { + return Map.of(); + } + try { + Map details = objectMapper.readValue(retrievalDetails, MAP_TYPE); + Object refsValue = details.get("evidence_refs"); + if (!(refsValue instanceof List refs)) { + return Map.of(); + } + Map result = new LinkedHashMap<>(); + for (Object refValue : refs) { + if (!(refValue instanceof Map ref)) { + continue; + } + String rawPath = stringValue(ref.get("raw_path")); + String text = stringValue(ref.get("text")); + if (!rawPath.isBlank() && !text.isBlank()) { + result.put(rawPath, text); + } + } + return result; + } catch (Exception ignored) { + return Map.of(); + } + } + + private boolean isExcerptSupported(String excerpt, String matchedText) { + String normalizedExcerpt = normalized(excerpt); + String normalizedMatched = normalized(matchedText); + if (normalizedMatched.contains(normalizedExcerpt) || normalizedExcerpt.contains(normalizedMatched)) { + return true; + } + Set excerptTokens = tokens(normalizedExcerpt); + if (excerptTokens.isEmpty()) { + return false; + } + Set matchedTokens = tokens(normalizedMatched); + int overlap = 0; + for (String token : excerptTokens) { + if (matchedTokens.contains(token)) { + overlap++; + } + } + return (double) overlap / excerptTokens.size() >= MIN_TOKEN_OVERLAP; + } + + private String normalized(String value) { + return value == null ? "" : value.toLowerCase() + .replaceAll("[\\p{Punct}\\s,。;:、()【】《》“”‘’]+", " ") + .trim(); + } + + private Set tokens(String text) { + if (text == null || text.isBlank()) { + return Set.of(); + } + Set result = new HashSet<>(); + for (String token : text.split("\\s+")) { + if (token.length() >= 2) { + result.add(token); + } + } + return result; + } + private void requireArray(Map output, String field, GatekeeperResult result) { if (!(output.get(field) instanceof List)) { - result.fail(RULE_SCHEMA, field, field + " must be an array"); + result.fail(RULE_SCHEMA, field, field + " must be an array", SEVERITY_LOW_CONFID); } } private void requireString(Map object, String field, String target, GatekeeperResult result) { if (stringValue(object.get(field)).isBlank()) { - result.fail(RULE_SCHEMA, target + "." + field, field + " is required"); + result.fail(RULE_SCHEMA, target + "." + field, field + " is required", SEVERITY_LOW_CONFID); } } @@ -211,23 +423,42 @@ public class ExecutorGatekeeperService { private static final class GatekeeperResult { private final List failedRules = new ArrayList<>(); - private final List warnings = new ArrayList<>(); + private final List> checkedBindings = new ArrayList<>(); + private final List> warnings = new ArrayList<>(); private final List> errors = new ArrayList<>(); + private String severity = SEVERITY_NONE; - void fail(String ruleId, String target, String message) { + void fail(String ruleId, String target, String message, String failureSeverity) { if (!failedRules.contains(ruleId)) { failedRules.add(ruleId); } + if (SEVERITY_REJECT.equals(failureSeverity)) { + severity = SEVERITY_REJECT; + } else if (!SEVERITY_REJECT.equals(severity)) { + severity = SEVERITY_LOW_CONFID; + } Map error = new LinkedHashMap<>(); error.put("rule_id", ruleId); + error.put("rule", ruleId); error.put("target", target); error.put("message", message); + error.put("severity", failureSeverity); errors.add(error); } + void checked(Map checked) { + checkedBindings.add(checked); + } + + void warn(Map warning) { + warnings.add(new LinkedHashMap<>(warning)); + } + Map toMap() { Map result = new LinkedHashMap<>(); - result.put("status", failedRules.isEmpty() ? (warnings.isEmpty() ? STATUS_PASS : STATUS_WARN) : STATUS_FAIL); + result.put("status", failedRules.isEmpty() ? STATUS_PASS : STATUS_FAIL); + result.put("severity", failedRules.isEmpty() ? SEVERITY_NONE : severity); + result.put("checked_bindings", checkedBindings); result.put("failed_rules", failedRules); result.put("warnings", warnings); result.put("errors", errors); diff --git a/src/main/java/com/superbiz/agent/service/ToolInvocationRecorder.java b/src/main/java/com/superbiz/agent/service/ToolInvocationRecorder.java index d1e015f..4f6c2ca 100644 --- a/src/main/java/com/superbiz/agent/service/ToolInvocationRecorder.java +++ b/src/main/java/com/superbiz/agent/service/ToolInvocationRecorder.java @@ -1,6 +1,7 @@ package com.superbiz.agent.service; import com.fasterxml.jackson.core.JsonProcessingException; +import com.fasterxml.jackson.databind.JsonNode; import com.fasterxml.jackson.databind.ObjectMapper; import com.superbiz.agent.domain.entity.ToolInvocation; import com.superbiz.agent.dto.ContextPack; @@ -87,6 +88,10 @@ public class ToolInvocationRecorder { if (extraDetails != null && !extraDetails.isEmpty()) { details.putAll(extraDetails); } + List> evidenceRefs = extractEvidenceRefs(toolName, output, details); + if (!evidenceRefs.isEmpty()) { + details.put("evidence_refs", evidenceRefs); + } ToolInvocation invocation = ToolInvocation.builder() .toolName(toolName) @@ -139,6 +144,10 @@ public class ToolInvocationRecorder { details.put("evidence_block_count", record.evidenceBlockCount()); } details.put("evidence_blocks", record.evidenceBlocks() == null ? List.of() : record.evidenceBlocks()); + List> evidenceRefs = evidenceRefsFromEvidenceBlocks(record.evidenceBlocks()); + if (!evidenceRefs.isEmpty()) { + details.put("evidence_refs", evidenceRefs); + } details.put("query_transform", record.queryTransform() == null ? Map.of() : record.queryTransform()); details.put("retrieval_trace", record.retrievalTrace() == null ? Map.of() : record.retrievalTrace()); details.put("context_pack_summary", record.contextPack() == null ? Map.of() : record.contextPack()); @@ -186,6 +195,132 @@ public class ToolInvocationRecorder { return success ? EVIDENCE_STATUS_SUPPORTED : EVIDENCE_STATUS_FAILED; } + private List> extractEvidenceRefs(String toolName, String output, Map details) { + if (output == null || output.isBlank()) { + return List.of(); + } + if ("lookup_knowledge".equals(toolName)) { + return evidenceRefsFromEvidenceBlocks(asMapList(details.get("evidence_blocks"))); + } + + try { + JsonNode root = objectMapper.readTree(output); + if ("query_metrics".equals(toolName)) { + return evidenceRefsFromArray(root.path("alerts"), "$.alerts", this::alertText); + } + if ("query_logs".equals(toolName)) { + return evidenceRefsFromArray(root.path("logs"), "$.logs", this::logText); + } + } catch (Exception e) { + log.debug("extract evidence_refs failed for tool={}", toolName, e); + } + return List.of(); + } + + private List> evidenceRefsFromArray(JsonNode arrayNode, + String pathPrefix, + java.util.function.Function textExtractor) { + if (!arrayNode.isArray()) { + return List.of(); + } + List> refs = new ArrayList<>(); + for (int i = 0; i < arrayNode.size(); i++) { + String text = textExtractor.apply(arrayNode.get(i)); + if (text == null || text.isBlank()) { + continue; + } + refs.add(Map.of( + "raw_path", pathPrefix + "[" + i + "]", + "text", bounded(text, 500) + )); + } + return refs; + } + + private String alertText(JsonNode alert) { + List parts = new ArrayList<>(); + addPart(parts, textField(alert, "alert_name")); + addPart(parts, textField(alert, "state")); + addPart(parts, textField(alert, "description")); + addPart(parts, "active_at=" + textField(alert, "active_at")); + addPart(parts, "duration=" + textField(alert, "duration")); + return String.join(", ", parts); + } + + private String logText(JsonNode log) { + List parts = new ArrayList<>(); + addPart(parts, textField(log, "timestamp")); + addPart(parts, textField(log, "level")); + addPart(parts, textField(log, "service")); + addPart(parts, textField(log, "message")); + JsonNode metrics = log.path("metrics"); + if (metrics.isObject() && !metrics.isEmpty()) { + addPart(parts, "metrics=" + metrics.toString()); + } + return String.join(" ", parts); + } + + private List> evidenceRefsFromEvidenceBlocks(List> blocks) { + if (blocks == null || blocks.isEmpty()) { + return List.of(); + } + List> refs = new ArrayList<>(); + for (int i = 0; i < blocks.size(); i++) { + Map block = blocks.get(i); + String text = firstNonBlank(block.get("content_preview"), block.get("content"), + block.get("title"), block.get("source")); + if (text.isBlank()) { + continue; + } + refs.add(Map.of( + "raw_path", "$.evidence_blocks[" + i + "]", + "text", bounded(text, 500) + )); + } + return refs; + } + + @SuppressWarnings("unchecked") + private List> asMapList(Object value) { + if (!(value instanceof List list)) { + return List.of(); + } + List> result = new ArrayList<>(); + for (Object item : list) { + if (item instanceof Map map) { + result.add((Map) map); + } + } + return result; + } + + private String textField(JsonNode node, String field) { + JsonNode value = node.path(field); + return value.isMissingNode() || value.isNull() ? "" : value.asText(""); + } + + private void addPart(List parts, String value) { + if (value != null && !value.isBlank() && !value.endsWith("=")) { + parts.add(value); + } + } + + private String firstNonBlank(Object... values) { + for (Object value : values) { + if (value != null && !String.valueOf(value).isBlank()) { + return String.valueOf(value); + } + } + return ""; + } + + private String bounded(String value, int limit) { + if (value == null) { + return ""; + } + return value.length() <= limit ? value : value.substring(0, limit) + "..."; + } + private String preview(String output) { if (output == null) { return null; diff --git a/src/main/resources/prompts/chat-executor-prompt.md b/src/main/resources/prompts/chat-executor-prompt.md index 4c94442..6aa3d20 100644 --- a/src/main/resources/prompts/chat-executor-prompt.md +++ b/src/main/resources/prompts/chat-executor-prompt.md @@ -5,6 +5,7 @@ - 需要外部信息时调用工具,但必须遵守下方的检索约束。 - 严禁凭记忆回答,必须基于本轮工具返回的真实数据。 - 执行完成后,输出严格的证据归因 JSON,供 Verifier 校验。 +- 你是证据收集与微观事实提炼器,不是最终答复生成器。 ## 规则 - 按顺序执行,不可跳过步骤。 @@ -12,6 +13,8 @@ - runbook、skill、历史案例、知识库中的通用模式只能作为排查指导或建议动作,不能直接写成本次事故的已确认事实。 - 如果检索内容不足以支撑结论,必须显式声明证据不足,严禁补全事故故事。 - 不要使用“通常情况下”“根据经验”“很可能已经发生”等无证据推断词来伪装事实。 +- 对窄范围确认问题,只输出与用户问题直接相关的 observation / negative_observation。通常 1 条 claim,最多 2 条 claim;不要限制 evidence_bindings 数量。 +- 禁止把根因、修复动作或用户明确排除的服务/主题写成 confirmed claim,除非本轮工具证据直接证明。 ## 检索约束 @@ -59,6 +62,7 @@ ### recommended_actions `recommended_actions` 用来放下一步排查或修复动作。 建议可以来自 runbook/skill,但必须说明 reason,不能写成“已确认根因”。 +本期 recommended_actions 只允许证据收集或继续排查动作,不要输出重启、扩容、修改配置等修复动作,除非用户明确要求执行方案。 ### missing_info `missing_info` 用来列出无法确认结论所缺少的具体证据。 @@ -80,9 +84,10 @@ "evidence_bindings": [ { "source_type": "tool_trace", - "source_id": "工具返回中的 evidence block id、trace_ref 或可定位标识", - "tool_name": "lookup_knowledge/query_logs/query_metrics/read_skill 等", - "source_invocation_ids": [], + "source_id": "工具返回中的 evidence block id、trace_ref 或可定位标识,可为空", + "tool_name": "lookup_knowledge/query_logs/query_metrics 等 evidence tool", + "source_invocation_id": null, + "raw_path": "$.alerts[0] / $.logs[0] / $.evidence_blocks[0]", "evidence_excerpt": "从工具返回中摘取的原话、指标值、日志片段或关键数据" } ] @@ -115,5 +120,8 @@ - `claims[*].support_level` 只能是 `direct` 或 `indirect`。 - `claims[*].evidence_bindings` 不能为空。 - `evidence_excerpt` 必须来自工具返回,不允许编造。 +- `raw_path` 必须指向工具返回数组中的具体条目:`query_metrics` 使用 `$.alerts[i]`,`query_logs` 使用 `$.logs[i]`,`lookup_knowledge` 使用 `$.evidence_blocks[i]`。 +- `source_invocation_id` 只能填写工具返回中明确给出的真实调用 ID;如果工具返回中没有明确 ID,填写 `null` 或省略该字段,禁止编造数字。系统只会在唯一候选工具调用存在时补齐 ID,但不会补齐 `raw_path`。 +- 不要再输出 `source_invocation_ids` 作为主要字段;兼容旧字段不作为精确证据引用。 - 如果没有任何可确认事实,`claims` 返回空数组,并在 `missing_info` 说明缺少什么。 - 不要把其它服务、其它历史案例、其它会话的事实迁移为当前会话事实。 diff --git a/src/main/resources/prompts/chat-verifier-prompt.md b/src/main/resources/prompts/chat-verifier-prompt.md index 246017c..fc530a8 100644 --- a/src/main/resources/prompts/chat-verifier-prompt.md +++ b/src/main/resources/prompts/chat-verifier-prompt.md @@ -12,7 +12,7 @@ - `executor_final_answer`:Executor 原始输出,仅用于 debug/fallback;当结构化输出有效时,不得从这里抽取额外确认事实 - `executor_structured_output`:如果 Executor 输出了合法证据归因 JSON,这里会提供解析后的对象。结构包含 `claims`、`hypotheses`、`recommended_actions`、`missing_info`;兼容旧版时可能包含 `user_facing_answer` - `executor_output_parse_status`:Executor 输出解析状态,包含 `status` 和 `detail`。`status` 可能是 `valid` / `missing` / `malformed` -- `tool_trace_summary`:基于真实工具调用整理出的证据索引。每一项都带有: +- `tool_trace_summary`:基于真实工具调用整理出的全局导航和审计索引。它不是唯一证据源;当 claim 有已核验的 `evidence_bindings[].evidence_excerpt` 时,应优先使用 claim-local excerpt 判断可推导性。每一项都带有: - `trace_ref` - `tool_name` - `topic_domain` @@ -20,7 +20,7 @@ - `input_summary` - `output_summary` - `evidence_level` -- `gatekeeper_result`:Executor 结构化输出的确定性校验结果,包含 `status`、`failed_rules`、`warnings`、`errors` +- `gatekeeper_result`:Executor 结构化输出的确定性校验结果,包含 `status`、`severity`、`checked_bindings`、`failed_rules`、`warnings`、`errors` - `retry_context`:第二轮可选输入;若为空,按首轮处理 ## 任务步骤 @@ -29,7 +29,8 @@ 如果 `executor_output_parse_status.status="valid"` 且 `executor_structured_output.claims` 存在: - 优先逐条校验 `executor_structured_output.claims` - 每个 claim 至少形成一条 `claim_checks` -- 必须检查 claim 的 `evidence_bindings` 是否能对应到 `tool_trace_summary` 中真实存在的 trace、tool 或 source_invocation_ids +- 如果 `gatekeeper_result.severity="none"`,将 claim 的 `evidence_bindings[].evidence_excerpt` 视为已通过代码核验的主证据,判断 `claim_text` 是否能由这些 excerpt 推出 +- `tool_trace_summary` 只用于理解工具调用全貌、补充 trace_ref、识别 no_evidence gap,不要求它逐字包含 excerpt 中已经核验过的全部事实 - 不得从 `executor_final_answer` 中抽取不在 claims 里的额外确认事实 如果 structured output 缺失或 malformed: @@ -58,8 +59,8 @@ - `contradicted` 结构化 claim 的校验规则: -- claim 有真实 evidence binding,且工具摘要直接包含该事实 → `direct_observation` -- claim 有真实 evidence binding,工具摘要没有逐字说明但可以合理推出 → `reasonable_inference` +- claim 有 Gatekeeper 核验通过的 evidence binding,且 `evidence_excerpt` 直接包含该事实 → `direct_observation` +- claim 有 Gatekeeper 核验通过的 evidence binding,excerpt 没有逐字说明但可以合理推出 → `reasonable_inference` - claim 有部分依据,但写成唯一根因、确认根因或说得过满 → `overstated` - claim 无法绑定真实 trace、invocation 或 excerpt → `unsupported` - claim 引入证据外的新服务名、订单号、错误码、指标值、根因 → `external_unknown` @@ -86,8 +87,10 @@ 严格使用以下判定矩阵: 0. 若 `gatekeeper_result.status="fail"` - 不得输出 `PASS` - - 若 `failed_rules` 包含 `evidence.invocation_ref`,倾向 `REJECT` - - 否则至少输出 `LOW_CONFID` + - 若 `gatekeeper_result.severity="reject"`,输出 `REJECT` + - 若 `gatekeeper_result.severity="low_confid"`,输出 `LOW_CONFID` + - 兼容旧输入:若缺少 `severity` 且 `failed_rules` 包含 `evidence.invocation_ref`,倾向 `REJECT` + - 兼容旧输入:若缺少 `severity` 且不是明显伪造,至少输出 `LOW_CONFID` 1. 若任一关键 claim 为 `contradicted` - `verdict = "REJECT"` diff --git a/src/test/java/com/superbiz/agent/agent/tool/QueryLogsToolsTest.java b/src/test/java/com/superbiz/agent/agent/tool/QueryLogsToolsTest.java new file mode 100644 index 0000000..7cdec82 --- /dev/null +++ b/src/test/java/com/superbiz/agent/agent/tool/QueryLogsToolsTest.java @@ -0,0 +1,48 @@ +package com.superbiz.agent.agent.tool; + +import com.fasterxml.jackson.databind.JsonNode; +import com.fasterxml.jackson.databind.ObjectMapper; +import com.superbiz.agent.service.ToolInvocationRecorder; +import org.junit.jupiter.api.Test; +import org.springframework.test.util.ReflectionTestUtils; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertFalse; +import static org.junit.jupiter.api.Assertions.assertTrue; +import static org.mockito.Mockito.mock; + +class QueryLogsToolsTest { + + private final ObjectMapper objectMapper = new ObjectMapper(); + + @Test + void queryLogsReturnsHikariPositiveMockForOrderService() throws Exception { + QueryLogsTools tools = new QueryLogsTools(mock(ToolInvocationRecorder.class)); + ReflectionTestUtils.setField(tools, "mockEnabled", true); + + String output = tools.queryLogs("ap-guangzhou", "application-logs", + "order-service HikariCP connection pool active=50/50 waiting", 10); + + JsonNode root = objectMapper.readTree(output); + assertTrue(root.path("success").asBoolean()); + assertEquals(2, root.path("logs").size()); + assertEquals("order-service", root.path("logs").get(0).path("service").asText()); + assertTrue(root.toString().contains("HikariPool-1")); + assertFalse(root.toString().contains("generic-service")); + } + + @Test + void queryLogsReturnsEmptyNoHitForOtherServiceHikariQuery() throws Exception { + QueryLogsTools tools = new QueryLogsTools(mock(ToolInvocationRecorder.class)); + ReflectionTestUtils.setField(tools, "mockEnabled", true); + + String output = tools.queryLogs("ap-guangzhou", "application-logs", + "inventory-service HikariCP connection pool active=50/50 waiting", 10); + + JsonNode root = objectMapper.readTree(output); + assertFalse(root.path("success").asBoolean()); + assertEquals(0, root.path("logs").size()); + assertEquals(0, root.path("total").asInt()); + assertFalse(root.toString().contains("generic-service")); + } +} diff --git a/src/test/java/com/superbiz/agent/hook/VerifierInputHookTest.java b/src/test/java/com/superbiz/agent/hook/VerifierInputHookTest.java index fb171eb..3100f0c 100644 --- a/src/test/java/com/superbiz/agent/hook/VerifierInputHookTest.java +++ b/src/test/java/com/superbiz/agent/hook/VerifierInputHookTest.java @@ -95,7 +95,7 @@ class VerifierInputHookTest { )); ToolInvocationRepository invocationRepository = mock(ToolInvocationRepository.class); when(invocationRepository.findBySessionIdOrderByIdAsc("structured-v2-session")).thenReturn(List.of( - ToolInvocation.builder().id(101L).sessionId("structured-v2-session").toolName("query_metrics").build() + invocation(101L, "structured-v2-session", "query_metrics", "$.alerts[0]", "active=50 max=50") )); VerifierInputHook hook = new VerifierInputHook(traceSummaryService, new ExecutorGatekeeperService(invocationRepository)); @@ -115,7 +115,8 @@ class VerifierInputHookTest { "source_type": "tool_trace", "source_id": "trace-1", "tool_name": "query_metrics", - "source_invocation_ids": [101], + "source_invocation_id": 101, + "raw_path": "$.alerts[0]", "evidence_excerpt": "active=50 max=50" } ] @@ -140,16 +141,96 @@ class VerifierInputHookTest { assertEquals("连接池 active 达到上限", payload.path("executor_structured_output").path("claims").get(0).path("claim_text").asText()); assertEquals("pass", payload.path("gatekeeper_result").path("status").asText()); + assertEquals("none", payload.path("gatekeeper_result").path("severity").asText()); assertEquals("pass", VerifierContextHolder.getGatekeeperResult().get("status")); } + @Test + void beforeModelBackfillsOnlyUniqueInvocationIdAndDoesNotPassWithoutRawPath() throws Exception { + ToolTraceSummaryService traceSummaryService = mock(ToolTraceSummaryService.class); + when(traceSummaryService.buildVerifierTraceSummary(anyString(), anyString())).thenReturn(List.of( + Map.of( + "trace_ref", "metrics-1", + "tool_name", "query_metrics", + "source_invocation_ids", List.of(101L) + ) + )); + ToolInvocationRepository invocationRepository = mock(ToolInvocationRepository.class); + when(invocationRepository.findBySessionIdOrderByIdAsc("backfill-session")).thenReturn(List.of( + invocation(101L, "backfill-session", "query_metrics", "$.alerts[0]", + "CPU 使用率持续超过 80%,当前值为 92%") + )); + VerifierInputHook hook = new VerifierInputHook(traceSummaryService, + new ExecutorGatekeeperService(invocationRepository)); + + String executorOutput = """ + { + "answer_version": "executor_evidence_v2", + "claims": [ + { + "claim_id": "claim-1", + "claim_type": "symptom", + "claim_text": "payment-service CPU 使用率超过 92%", + "support_level": "direct", + "evidence_bindings": [ + { + "source_type": "tool_trace", + "source_id": "prometheus-alert-HighCPUUsage", + "tool_name": "queryPrometheusAlerts", + "evidence_excerpt": "CPU 使用率持续超过 80%,当前值为 92%" + } + ] + } + ], + "hypotheses": [], + "recommended_actions": [ + { + "action_text": "restart payment-service", + "reason": "cpu alert is firing", + "evidence_bindings": [ + { + "source_type": "tool_trace", + "source_id": "prometheus-alert-HighCPUUsage", + "tool_name": "queryPrometheusAlerts", + "evidence_excerpt": "CPU usage is 92%" + } + ] + } + ], + "missing_info": [] + } + """; + + AgentCommand command = hook.beforeModel( + List.of(new AssistantMessage(executorOutput)), + RunnableConfig.builder().addMetadata("sessionId", "backfill-session").build() + ); + + JsonNode payload = readPayload(command); + JsonNode binding = payload.path("executor_structured_output") + .path("claims").get(0) + .path("evidence_bindings").get(0); + assertEquals("query_metrics", binding.path("tool_name").asText()); + assertEquals(101L, binding.path("source_invocation_id").asLong()); + JsonNode actionBinding = payload.path("executor_structured_output") + .path("recommended_actions").get(0) + .path("evidence_bindings").get(0); + assertEquals("query_metrics", actionBinding.path("tool_name").asText()); + assertEquals(101L, actionBinding.path("source_invocation_id").asLong()); + assertFalse(binding.has("raw_path")); + assertEquals("fail", payload.path("gatekeeper_result").path("status").asText()); + assertEquals("low_confid", payload.path("gatekeeper_result").path("severity").asText()); + assertEquals("evidence.invocation_auto_backfill", + payload.path("gatekeeper_result").path("warnings").get(0).path("rule").asText()); + } + @Test void beforeModelAddsFailingGatekeeperResultForFabricatedInvocationId() throws Exception { ToolTraceSummaryService traceSummaryService = mock(ToolTraceSummaryService.class); when(traceSummaryService.buildVerifierTraceSummary(anyString(), anyString())).thenReturn(List.of()); ToolInvocationRepository invocationRepository = mock(ToolInvocationRepository.class); when(invocationRepository.findBySessionIdOrderByIdAsc("fabricated-invocation-session")).thenReturn(List.of( - ToolInvocation.builder().id(101L).sessionId("fabricated-invocation-session").toolName("query_metrics").build() + invocation(101L, "fabricated-invocation-session", "query_metrics", "$.alerts[0]", "active=50 max=50") )); VerifierInputHook hook = new VerifierInputHook(traceSummaryService, new ExecutorGatekeeperService(invocationRepository)); @@ -167,7 +248,8 @@ class VerifierInputHookTest { { "source_type": "tool_trace", "tool_name": "query_metrics", - "source_invocation_ids": [999], + "source_invocation_id": 999, + "raw_path": "$.alerts[0]", "evidence_excerpt": "active=50 max=50" } ] @@ -186,6 +268,7 @@ class VerifierInputHookTest { JsonNode payload = readPayload(command); assertEquals("fail", payload.path("gatekeeper_result").path("status").asText()); + assertEquals("reject", payload.path("gatekeeper_result").path("severity").asText()); assertEquals("evidence.invocation_ref", payload.path("gatekeeper_result").path("failed_rules").get(0).asText()); } @@ -284,4 +367,14 @@ class VerifierInputHookTest { assertTrue(message instanceof UserMessage); return objectMapper.readTree(((UserMessage) message).getText()); } + + private ToolInvocation invocation(Long id, String sessionId, String toolName, String rawPath, String text) { + return ToolInvocation.builder() + .id(id) + .sessionId(sessionId) + .toolName(toolName) + .retrievalDetails("{\"evidence_refs\":[{\"raw_path\":\"" + rawPath + + "\",\"text\":\"" + text + "\"}]}") + .build(); + } } diff --git a/src/test/java/com/superbiz/agent/service/ChatServiceSequentialAgentTest.java b/src/test/java/com/superbiz/agent/service/ChatServiceSequentialAgentTest.java index 2d57029..f02f32b 100644 --- a/src/test/java/com/superbiz/agent/service/ChatServiceSequentialAgentTest.java +++ b/src/test/java/com/superbiz/agent/service/ChatServiceSequentialAgentTest.java @@ -253,7 +253,8 @@ class ChatServiceSequentialAgentTest { "source_type": "tool_trace", "source_id": "trace-1", "tool_name": "query_metrics", - "source_invocation_ids": [101], + "source_invocation_id": 101, + "raw_path": "$.alerts[0]", "evidence_excerpt": "active=50 max=50" } ] @@ -300,7 +301,8 @@ class ChatServiceSequentialAgentTest { "source_type": "tool_trace", "source_id": "trace-1", "tool_name": "query_metrics", - "source_invocation_ids": [101], + "source_invocation_id": 101, + "raw_path": "$.alerts[0]", "evidence_excerpt": "active=50 max=50" } ] @@ -350,6 +352,7 @@ class ChatServiceSequentialAgentTest { .id(101L) .sessionId("sequential-gatekeeper-persist-session") .toolName("query_metrics") + .retrievalDetails(evidenceRefs("$.alerts[0]", "active=50 max=50")) .build())); ScriptedChatModel chatModel = new ScriptedChatModel(); chatModel.executorOutput = """ @@ -366,7 +369,8 @@ class ChatServiceSequentialAgentTest { "source_type": "tool_trace", "source_id": "trace-1", "tool_name": "query_metrics", - "source_invocation_ids": [101], + "source_invocation_id": 101, + "raw_path": "$.alerts[0]", "evidence_excerpt": "active=50 max=50" } ] @@ -393,6 +397,7 @@ class ChatServiceSequentialAgentTest { @SuppressWarnings("unchecked") Map gatekeeperResult = (Map) verifierEvaluation.get("gatekeeper_result"); assertEquals("pass", gatekeeperResult.get("status")); + assertEquals("none", gatekeeperResult.get("severity")); } @Test @@ -407,6 +412,7 @@ class ChatServiceSequentialAgentTest { .id(101L) .sessionId("sequential-claim-check-session") .toolName("query_metrics") + .retrievalDetails(evidenceRefs("$.alerts[0]", "active=50 max=50")) .build())); ScriptedChatModel chatModel = new ScriptedChatModel(""" { @@ -466,6 +472,7 @@ class ChatServiceSequentialAgentTest { .id(101L) .sessionId("sequential-gatekeeper-fail-session") .toolName("query_metrics") + .retrievalDetails(evidenceRefs("$.alerts[0]", "active=50 max=50")) .build())); ScriptedChatModel chatModel = new ScriptedChatModel(""" { @@ -493,7 +500,8 @@ class ChatServiceSequentialAgentTest { { "source_type": "tool_trace", "tool_name": "query_metrics", - "source_invocation_ids": [999], + "source_invocation_id": 999, + "raw_path": "$.alerts[0]", "evidence_excerpt": "active=50 max=50" } ] @@ -647,6 +655,7 @@ class ChatServiceSequentialAgentTest { when(toolInvocationRepository.findBySessionIdOrderByIdAsc(anyString())).thenReturn(List.of(ToolInvocation.builder() .id(101L) .toolName("query_metrics") + .retrievalDetails(evidenceRefs("$.alerts[0]", "active=50 max=50")) .build())); EvaluationService evaluationService = mock(EvaluationService.class); @@ -694,7 +703,8 @@ class ChatServiceSequentialAgentTest { "source_type": "tool_trace", "source_id": "trace-1", "tool_name": "query_metrics", - "source_invocation_ids": [101], + "source_invocation_id": 101, + "raw_path": "$.alerts[0]", "evidence_excerpt": "active=50 max=50" } ] @@ -707,6 +717,10 @@ class ChatServiceSequentialAgentTest { """; } + private String evidenceRefs(String rawPath, String text) { + return "{\"evidence_refs\":[{\"raw_path\":\"" + rawPath + "\",\"text\":\"" + text + "\"}]}"; + } + private static final class ScriptedChatModel implements ChatModel { private final java.util.ArrayList agentCalls = new java.util.ArrayList<>(); private String promptText = ""; @@ -728,7 +742,8 @@ class ChatServiceSequentialAgentTest { "source_type": "tool_trace", "source_id": "trace-1", "tool_name": "query_metrics", - "source_invocation_ids": [101], + "source_invocation_id": 101, + "raw_path": "$.alerts[0]", "evidence_excerpt": "active=50 max=50" } ] diff --git a/src/test/java/com/superbiz/agent/service/ExecutorGatekeeperServiceTest.java b/src/test/java/com/superbiz/agent/service/ExecutorGatekeeperServiceTest.java index 24c6708..5b14d59 100644 --- a/src/test/java/com/superbiz/agent/service/ExecutorGatekeeperServiceTest.java +++ b/src/test/java/com/superbiz/agent/service/ExecutorGatekeeperServiceTest.java @@ -18,14 +18,18 @@ class ExecutorGatekeeperServiceTest { void validatePassesForExecutorEvidenceV2WithMatchingInvocation() { ToolInvocationRepository repository = mock(ToolInvocationRepository.class); when(repository.findBySessionIdOrderByIdAsc("session-1")).thenReturn(List.of( - ToolInvocation.builder().id(101L).sessionId("session-1").toolName("query_metrics").build() + invocation(101L, "query_metrics", "$.alerts[0]", + "HighCPUUsage firing, service=payment-service, current=92%, duration=25m") )); ExecutorGatekeeperService service = new ExecutorGatekeeperService(repository); - Map result = service.validate("session-1", validOutput(101L, "query_metrics"), + Map result = service.validate("session-1", + validOutput(101L, "query_metrics", "$.alerts[0]", + "HighCPUUsage firing, service=payment-service, current=92%"), Map.of("status", "valid")); assertEquals("pass", result.get("status")); + assertEquals("none", result.get("severity")); assertTrue(((List) result.get("failed_rules")).isEmpty()); } @@ -34,12 +38,13 @@ class ExecutorGatekeeperServiceTest { ToolInvocationRepository repository = mock(ToolInvocationRepository.class); when(repository.findBySessionIdOrderByIdAsc("session-1")).thenReturn(List.of()); ExecutorGatekeeperService service = new ExecutorGatekeeperService(repository); - Map output = validOutput(101L, "query_metrics"); + Map output = validOutput(101L, "query_metrics", "$.alerts[0]", "cpu=92"); output.put("user_facing_answer", "旧版最终答案"); Map result = service.validate("session-1", output, Map.of("status", "valid")); assertEquals("fail", result.get("status")); + assertEquals("reject", result.get("severity")); assertTrue(((List) result.get("failed_rules")).contains("schema.executor_v2")); } @@ -47,14 +52,16 @@ class ExecutorGatekeeperServiceTest { void validateFailsForFabricatedInvocationId() { ToolInvocationRepository repository = mock(ToolInvocationRepository.class); when(repository.findBySessionIdOrderByIdAsc("session-1")).thenReturn(List.of( - ToolInvocation.builder().id(101L).sessionId("session-1").toolName("query_metrics").build() + invocation(101L, "query_metrics", "$.alerts[0]", "cpu=92") )); ExecutorGatekeeperService service = new ExecutorGatekeeperService(repository); - Map result = service.validate("session-1", validOutput(999L, "query_metrics"), + Map result = service.validate("session-1", + validOutput(999L, "query_metrics", "$.alerts[0]", "cpu=92"), Map.of("status", "valid")); assertEquals("fail", result.get("status")); + assertEquals("reject", result.get("severity")); assertTrue(((List) result.get("failed_rules")).contains("evidence.invocation_ref")); } @@ -62,19 +69,137 @@ class ExecutorGatekeeperServiceTest { void validateFailsForToolNameMismatch() { ToolInvocationRepository repository = mock(ToolInvocationRepository.class); when(repository.findBySessionIdOrderByIdAsc("session-1")).thenReturn(List.of( - ToolInvocation.builder().id(101L).sessionId("session-1").toolName("query_logs").build() + invocation(101L, "query_logs", "$.logs[0]", "cpu=92") )); ExecutorGatekeeperService service = new ExecutorGatekeeperService(repository); - Map result = service.validate("session-1", validOutput(101L, "query_metrics"), + Map result = service.validate("session-1", + validOutput(101L, "query_metrics", "$.alerts[0]", "cpu=92"), Map.of("status", "valid")); assertEquals("fail", result.get("status")); + assertEquals("reject", result.get("severity")); assertTrue(((List) result.get("failed_rules")).contains("evidence.invocation_ref")); } - @SuppressWarnings("unchecked") - private Map validOutput(Long invocationId, String toolName) { + @Test + void validateDowngradesMissingRawPathToLowConfidence() { + ToolInvocationRepository repository = mock(ToolInvocationRepository.class); + when(repository.findBySessionIdOrderByIdAsc("session-1")).thenReturn(List.of( + invocation(101L, "query_metrics", "$.alerts[0]", "cpu=92") + )); + ExecutorGatekeeperService service = new ExecutorGatekeeperService(repository); + Map output = validOutput(101L, "query_metrics", null, "cpu=92"); + + Map result = service.validate("session-1", output, Map.of("status", "valid")); + + assertEquals("fail", result.get("status")); + assertEquals("low_confid", result.get("severity")); + assertTrue(((List) result.get("failed_rules")).contains("evidence.raw_path")); + } + + @Test + void validateRejectsUnknownRawPath() { + ToolInvocationRepository repository = mock(ToolInvocationRepository.class); + when(repository.findBySessionIdOrderByIdAsc("session-1")).thenReturn(List.of( + invocation(101L, "query_metrics", "$.alerts[0]", "cpu=92") + )); + ExecutorGatekeeperService service = new ExecutorGatekeeperService(repository); + + Map result = service.validate("session-1", + validOutput(101L, "query_metrics", "$.alerts[99]", "cpu=92"), + Map.of("status", "valid")); + + assertEquals("fail", result.get("status")); + assertEquals("reject", result.get("severity")); + assertTrue(((List) result.get("failed_rules")).contains("evidence.raw_path")); + } + + @Test + void validateDowngradesOldInvocationWithoutEvidenceRefsToLowConfidence() { + ToolInvocationRepository repository = mock(ToolInvocationRepository.class); + when(repository.findBySessionIdOrderByIdAsc("session-1")).thenReturn(List.of( + ToolInvocation.builder().id(101L).sessionId("session-1").toolName("query_metrics") + .retrievalDetails("{\"evidence_status\":\"supported\"}") + .build() + )); + ExecutorGatekeeperService service = new ExecutorGatekeeperService(repository); + + Map result = service.validate("session-1", + validOutput(101L, "query_metrics", "$.alerts[0]", "cpu=92"), + Map.of("status", "valid")); + + assertEquals("fail", result.get("status")); + assertEquals("low_confid", result.get("severity")); + assertTrue(((List) result.get("failed_rules")).contains("evidence.missing")); + } + + + @Test + void validateRejectsMismatchedExcerpt() { + ToolInvocationRepository repository = mock(ToolInvocationRepository.class); + when(repository.findBySessionIdOrderByIdAsc("session-1")).thenReturn(List.of( + invocation(101L, "query_metrics", "$.alerts[0]", "HighCPUUsage firing service payment-service current 92") + )); + ExecutorGatekeeperService service = new ExecutorGatekeeperService(repository); + + Map result = service.validate("session-1", + validOutput(101L, "query_metrics", "$.alerts[0]", "HikariCP active=50/50 waiting=32"), + Map.of("status", "valid")); + + assertEquals("fail", result.get("status")); + assertEquals("reject", result.get("severity")); + assertTrue(((List) result.get("failed_rules")).contains("evidence.excerpt_mismatch")); + } + + @Test + void validateFailsForRecommendedActionFabricatedInvocationId() { + ToolInvocationRepository repository = mock(ToolInvocationRepository.class); + when(repository.findBySessionIdOrderByIdAsc("session-1")).thenReturn(List.of( + invocation(101L, "query_metrics", "$.alerts[0]", "cpu=92") + )); + ExecutorGatekeeperService service = new ExecutorGatekeeperService(repository); + Map output = validOutput(101L, "query_metrics", "$.alerts[0]", "cpu=92"); + output.put("recommended_actions", List.of(Map.of( + "action_text", "restart service", + "reason", "alert is firing", + "evidence_bindings", List.of(Map.of( + "source_type", "tool_trace", + "source_id", "trace-1", + "tool_name", "query_metrics", + "source_invocation_id", 999L, + "raw_path", "$.alerts[0]", + "evidence_excerpt", "cpu=92" + )) + ))); + + Map result = service.validate("session-1", output, Map.of("status", "valid")); + + assertEquals("fail", result.get("status")); + assertEquals("reject", result.get("severity")); + assertTrue(((List) result.get("failed_rules")).contains("evidence.invocation_ref")); + } + + private ToolInvocation invocation(Long id, String toolName, String rawPath, String text) { + return ToolInvocation.builder() + .id(id) + .sessionId("session-1") + .toolName(toolName) + .retrievalDetails("{\"evidence_refs\":[{\"raw_path\":\"" + rawPath + + "\",\"text\":\"" + text + "\"}]}") + .build(); + } + + private Map validOutput(Long invocationId, String toolName, String rawPath, String excerpt) { + Map binding = new java.util.LinkedHashMap<>(); + binding.put("source_type", "tool_trace"); + binding.put("source_id", "trace-1"); + binding.put("tool_name", toolName); + binding.put("source_invocation_id", invocationId); + if (rawPath != null) { + binding.put("raw_path", rawPath); + } + binding.put("evidence_excerpt", excerpt); return new java.util.LinkedHashMap<>(Map.of( "answer_version", "executor_evidence_v2", "claims", List.of(Map.of( @@ -82,13 +207,7 @@ class ExecutorGatekeeperServiceTest { "claim_type", "symptom", "claim_text", "连接池 active 达到上限", "support_level", "direct", - "evidence_bindings", List.of(Map.of( - "source_type", "tool_trace", - "source_id", "trace-1", - "tool_name", toolName, - "source_invocation_ids", List.of(invocationId), - "evidence_excerpt", "active=50 max=50" - )) + "evidence_bindings", List.of(binding) )), "hypotheses", List.of(), "recommended_actions", List.of(), diff --git a/src/test/java/com/superbiz/agent/service/ToolInvocationRecorderTest.java b/src/test/java/com/superbiz/agent/service/ToolInvocationRecorderTest.java index 6c9a6af..cdb1a51 100644 --- a/src/test/java/com/superbiz/agent/service/ToolInvocationRecorderTest.java +++ b/src/test/java/com/superbiz/agent/service/ToolInvocationRecorderTest.java @@ -1,5 +1,6 @@ package com.superbiz.agent.service; +import com.fasterxml.jackson.databind.JsonNode; import com.fasterxml.jackson.databind.ObjectMapper; import com.superbiz.agent.domain.entity.ToolInvocation; import com.superbiz.agent.dto.ContextPack; @@ -23,6 +24,8 @@ import static org.mockito.Mockito.when; class ToolInvocationRecorderTest { + private final ObjectMapper objectMapper = new ObjectMapper(); + @Test void recordEvidenceToolPreservesNoEvidenceSemantics() { ToolInvocationRepository repository = mock(ToolInvocationRepository.class); @@ -56,6 +59,72 @@ class ToolInvocationRecorderTest { assertTrue(saved.getRetrievalDetails().contains("\"retrieved_domains\":[\"application-logs\"]")); } + @Test + void recordEvidenceToolExtractsLogEvidenceRefs() throws Exception { + ToolInvocationRepository repository = mock(ToolInvocationRepository.class); + when(repository.save(any(ToolInvocation.class))).thenAnswer(invocation -> invocation.getArgument(0)); + ToolInvocationRecorder recorder = new ToolInvocationRecorder(repository, new ObjectMapper()); + SessionContextHolder.setSessionId("log-ref-session"); + + try { + recorder.recordEvidenceTool( + "query_logs", + Map.of("query", "HikariCP order-service"), + """ + {"success":true,"logs":[{"timestamp":"2026-07-08 10:00:00","level":"ERROR","service":"order-service","message":"HikariPool-1 - Connection is not available, request timed out after 30000ms","metrics":{"waiting":"32"}}]} + """, + true, + System.currentTimeMillis() - 10, + null, + "application-logs", + ToolInvocationRecorder.EVIDENCE_STATUS_SUPPORTED, + Map.of("log_topic", "application-logs") + ); + } finally { + SessionContextHolder.clear(); + } + + ArgumentCaptor captor = ArgumentCaptor.forClass(ToolInvocation.class); + verify(repository).save(captor.capture()); + JsonNode details = objectMapper.readTree(captor.getValue().getRetrievalDetails()); + + assertEquals("$.logs[0]", details.path("evidence_refs").get(0).path("raw_path").asText()); + assertTrue(details.path("evidence_refs").get(0).path("text").asText().contains("HikariPool-1")); + } + + @Test + void recordEvidenceToolExtractsMetricEvidenceRefs() throws Exception { + ToolInvocationRepository repository = mock(ToolInvocationRepository.class); + when(repository.save(any(ToolInvocation.class))).thenAnswer(invocation -> invocation.getArgument(0)); + ToolInvocationRecorder recorder = new ToolInvocationRecorder(repository, new ObjectMapper()); + SessionContextHolder.setSessionId("metric-ref-session"); + + try { + recorder.recordEvidenceTool( + "query_metrics", + Map.of("query", "active_prometheus_alerts"), + """ + {"success":true,"alerts":[{"alert_name":"HighMemoryUsage","state":"firing","description":"服务 order-service 当前值为 91%","active_at":"2026-07-08T10:00:00Z","duration":"15m"}]} + """, + true, + System.currentTimeMillis() - 10, + null, + "prometheus_alerts", + ToolInvocationRecorder.EVIDENCE_STATUS_SUPPORTED, + Map.of("metric_family", "prometheus_alerts") + ); + } finally { + SessionContextHolder.clear(); + } + + ArgumentCaptor captor = ArgumentCaptor.forClass(ToolInvocation.class); + verify(repository).save(captor.capture()); + JsonNode details = objectMapper.readTree(captor.getValue().getRetrievalDetails()); + + assertEquals("$.alerts[0]", details.path("evidence_refs").get(0).path("raw_path").asText()); + assertTrue(details.path("evidence_refs").get(0).path("text").asText().contains("HighMemoryUsage")); + } + @Test void recordLookupKnowledgePreservesRetrievalSpecificFields() { ToolInvocationRepository repository = mock(ToolInvocationRepository.class); @@ -113,6 +182,8 @@ class ToolInvocationRecorderTest { assertTrue(saved.getRetrievalDetails().contains("\"evidence_candidate_count\":2")); assertTrue(saved.getRetrievalDetails().contains("\"evidence_block_count\":1")); assertTrue(saved.getRetrievalDetails().contains("\"evidence_blocks\"")); + assertTrue(saved.getRetrievalDetails().contains("\"evidence_refs\"")); + assertTrue(saved.getRetrievalDetails().contains("\"raw_path\":\"$.evidence_blocks[0]\"")); assertTrue(saved.getRetrievalDetails().contains("\"query_transform\"")); assertTrue(saved.getRetrievalDetails().contains("\"retrieval_trace\"")); assertTrue(saved.getRetrievalDetails().contains("\"context_pack_summary\""));