From dc6cd32a6757790463eefd74502e00d8d3097c4b Mon Sep 17 00:00:00 2001 From: aruo <40362743+zyongxin@users.noreply.github.com> Date: Sat, 4 Jul 2026 22:36:30 +0800 Subject: [PATCH] Harden evidence trace semantics --- devflow/index.md | 1 + .../acceptance.md | 32 +++ .../brief.md | 32 +++ .../decisions.md | 24 +++ .../evidence.md | 11 + .../ISS-005-evidence-trace-hardening.md | 137 +++++++++++++ mvp/issues/README.md | 1 + .../.openspec.yaml | 2 + .../design.md | 64 ++++++ .../proposal.md | 26 +++ .../specs/chat-verifier-agent/spec.md | 14 ++ .../specs/evidence-trace-hardening/spec.md | 82 ++++++++ .../tasks.md | 22 ++ openspec/specs/chat-verifier-agent/spec.md | 14 ++ .../specs/evidence-trace-hardening/spec.md | 86 ++++++++ .../agent/agent/tool/QueryLogsTools.java | 34 ++-- .../agent/agent/tool/QueryMetricsTools.java | 15 +- .../agent/service/ToolInvocationRecorder.java | 188 ++++++++++++++++++ .../service/ToolTraceSummaryService.java | 55 ++++- .../agent/tool/LookupKnowledgeTool.java | 134 ++----------- .../ChatServiceSequentialAgentTest.java | 81 +++++++- .../service/ToolInvocationRecorderTest.java | 96 +++++++++ .../service/ToolTraceSummaryServiceTest.java | 75 +++++++ .../agent/tool/LookupKnowledgeToolTest.java | 11 + 24 files changed, 1096 insertions(+), 141 deletions(-) create mode 100644 devflow/projects/2026-07-04-evidence-trace-hardening/acceptance.md create mode 100644 devflow/projects/2026-07-04-evidence-trace-hardening/brief.md create mode 100644 devflow/projects/2026-07-04-evidence-trace-hardening/decisions.md create mode 100644 devflow/projects/2026-07-04-evidence-trace-hardening/evidence.md create mode 100644 mvp/issues/ISS-005-evidence-trace-hardening.md create mode 100644 openspec/changes/archive/2026-07-04-evidence-trace-hardening/.openspec.yaml create mode 100644 openspec/changes/archive/2026-07-04-evidence-trace-hardening/design.md create mode 100644 openspec/changes/archive/2026-07-04-evidence-trace-hardening/proposal.md create mode 100644 openspec/changes/archive/2026-07-04-evidence-trace-hardening/specs/chat-verifier-agent/spec.md create mode 100644 openspec/changes/archive/2026-07-04-evidence-trace-hardening/specs/evidence-trace-hardening/spec.md create mode 100644 openspec/changes/archive/2026-07-04-evidence-trace-hardening/tasks.md create mode 100644 openspec/specs/evidence-trace-hardening/spec.md create mode 100644 src/test/java/com/superbiz/agent/service/ToolInvocationRecorderTest.java create mode 100644 src/test/java/com/superbiz/agent/service/ToolTraceSummaryServiceTest.java diff --git a/devflow/index.md b/devflow/index.md index b06943b..a008b25 100644 --- a/devflow/index.md +++ b/devflow/index.md @@ -4,6 +4,7 @@ | 日期 | slug | 领域 | 关键词 | 状态 | |---|---|---|---|---| +| 2026-07-04 | evidence-trace-hardening | 证据链/降级契约/离线验证 | ToolInvocationRecorder, ToolTraceSummaryService, lookup_knowledge, query_logs, query_metrics, LOW_CONFID, REJECT | openspec/changes/evidence-trace-hardening | active | | 2026-07-03 | mvp-demo-trace-acceptance | MVP Demo/trace/acceptance | mvp-demo, trace API, diagnosis_session, agent_step, tool_invocation, feedback | openspec/changes/archive/2026-07-03-mvp-demo-trace-acceptance | archived | | 2026-05-29 | chatmodel-abstraction | 解耦/多模型路由 | ChatModel, EmbeddingModel, DeepSeek, BGE-M3, SiliconFlow, Spring AI | archived | | 2026-06-23 | phase1-infrastructure | 基础设施/文档管理 | MySQL, Redis, Milvus, Flyway, JPA, 向量检索, 类别过滤 | archived | diff --git a/devflow/projects/2026-07-04-evidence-trace-hardening/acceptance.md b/devflow/projects/2026-07-04-evidence-trace-hardening/acceptance.md new file mode 100644 index 0000000..57ef09e --- /dev/null +++ b/devflow/projects/2026-07-04-evidence-trace-hardening/acceptance.md @@ -0,0 +1,32 @@ +# Acceptance: evidence-trace-hardening + +## Classification + +standard-light + +## Task Status + +| Task | Status | Notes | +| --- | --- | --- | +| Issue and OpenSpec setup | Done | `ISS-005` and the initial OpenSpec artifacts were created. | +| Implementation | Done | Recorder contract, lookup persistence path, evidence summary semantics, and degraded-path tests were implemented. | +| Verification | Done | Targeted offline tests and compile verification passed. | + +## Verification + +### Script Verification + +- Command: `mvn -q "-Dtest=ToolInvocationRecorderTest,ToolTraceSummaryServiceTest,ChatServiceSequentialAgentTest,LookupKnowledgeToolTest" test` +- Result: passed +- Notes: Covers recorder contract, summary semantics for success/failure/no-evidence, and `ChatService` fallback / degraded paths. + +### Static Verification + +- Command: `mvn -q -DskipTests compile` +- Result: passed + +## Open Questions + +| Question | Current position | +| --- | --- | +| Should deduped retrievals be counted separately from generic no-hit events in future evaluation metrics? | Deferred to P1-B; this change preserves enough structure to decide later. | diff --git a/devflow/projects/2026-07-04-evidence-trace-hardening/brief.md b/devflow/projects/2026-07-04-evidence-trace-hardening/brief.md new file mode 100644 index 0000000..3a28e56 --- /dev/null +++ b/devflow/projects/2026-07-04-evidence-trace-hardening/brief.md @@ -0,0 +1,32 @@ +# Brief: evidence-trace-hardening + +## Background + +The MVP already has persisted tool traces and a verifier, but the evidence contract is still only partially standardized. For interview-focused hardening, the project now needs a tighter contract for evidence persistence, no-evidence / failure semantics, and degraded-output behavior. + +## Goals + +1. Standardize the persisted evidence-tool contract across `lookup_knowledge`, `query_logs`, and `query_metrics`. +2. Make verifier-facing summaries distinguish failed calls, no-hit calls, deduped retrievals, and actual supporting evidence. +3. Add offline tests for verifier fallback and degraded-output paths. + +## Scope + +- `ToolInvocationRecorder` +- `LookupKnowledgeTool` +- `QueryLogsTools` +- `QueryMetricsTools` +- `ToolTraceSummaryService` +- `ChatService` +- Focused offline tests + +## Non-Goals + +- No new API or schema +- No evaluation harness yet +- No trace UI +- No security/config cleanup + +## Related OpenSpec + +`openspec/changes/evidence-trace-hardening/` diff --git a/devflow/projects/2026-07-04-evidence-trace-hardening/decisions.md b/devflow/projects/2026-07-04-evidence-trace-hardening/decisions.md new file mode 100644 index 0000000..180e056 --- /dev/null +++ b/devflow/projects/2026-07-04-evidence-trace-hardening/decisions.md @@ -0,0 +1,24 @@ +# Evidence Trace Hardening Decisions + +## Clarify + +- Entry summary: harden the MVP evidence contract before building the P1-B evaluation harness. +- Slug: `evidence-trace-hardening` +- Devflow scale: standard-light + +## Context + +- `ISS-003` raised verifier traceability and failure-path concerns. +- Current code inspection shows `QueryLogsTools` and `QueryMetricsTools` already use `ToolInvocationRecorder`, while `LookupKnowledgeTool` still persists rows through a local helper. +- `ChatService` already contains fallback behavior for missing/invalid `verifier_output`, but coverage is narrow. + +## Key Decisions + +- Decision: Treat this as a contract-hardening change, not a new feature change. + - Reason: The project already has the necessary runtime pieces; the gap is semantic consistency and testability. + +- Decision: Keep the scope before P1-B. + - Reason: The evaluation harness will rely on stable evidence semantics, so this contract slice should land first. + +- Decision: Preserve schema and API stability. + - Reason: The interview value here is engineering rigor, not more surface area. diff --git a/devflow/projects/2026-07-04-evidence-trace-hardening/evidence.md b/devflow/projects/2026-07-04-evidence-trace-hardening/evidence.md new file mode 100644 index 0000000..2fb1883 --- /dev/null +++ b/devflow/projects/2026-07-04-evidence-trace-hardening/evidence.md @@ -0,0 +1,11 @@ +# Evidence Trace Hardening Evidence + +## Evidence + +| Source | Evidence | Conclusion | Reported | +|---|---|---|---| +| `ToolInvocationRecorder` | Provides a common persistence seam for evidence tools | Contract hardening should build on the existing recorder instead of introducing a new store path | Yes | +| `LookupKnowledgeTool` | Still constructs `ToolInvocation` rows through a local helper | Retrieval-aware evidence persistence is not yet unified with the recorder contract | Yes | +| `QueryLogsTools` / `QueryMetricsTools` | Already record evidence invocations through `recordEvidenceTool(...)` | Current gap is semantic alignment, not missing persistence | Yes | +| `ToolTraceSummaryService` | Merges rows by tool and topic domain and infers evidence level heuristically | Summary rules need explicit handling for failure, no-hit, and dedup cases | Yes | +| `ChatService` | Falls back to `LOW_CONFID` when verifier output is missing or invalid | These degraded paths exist and should now be covered by focused offline tests | Yes | diff --git a/mvp/issues/ISS-005-evidence-trace-hardening.md b/mvp/issues/ISS-005-evidence-trace-hardening.md new file mode 100644 index 0000000..dd872b9 --- /dev/null +++ b/mvp/issues/ISS-005-evidence-trace-hardening.md @@ -0,0 +1,137 @@ +# ISS-005 证据链补齐与降级契约收敛 + +**状态**:进行中(sm-flow) +**严重程度**:高 +**发现时间**:2026-07-04 +**来源**:P1-A 面试打磨项 / 基于 ISS-003 的当前实现复核 +**关联**:ISS-003(Verifier 证据链、失败路径可验证性)、`chat-verifier-agent`、`mvp-demo-trace-acceptance` + +--- + +## 背景 + +当前 MVP 已具备: + +- `lookup_knowledge`、`query_logs`、`query_metrics` 的工具调用落库 +- Verifier 基于 `tool_trace_summary` 做事实核查 +- `LOW_CONFID` / `REJECT` 的用户侧降级输出 +- trace API 可回放 session、agent_step、tool_invocation 和 self_evaluation + +但如果目标是拿这个项目去面试 Agent 工程师,当前实现仍有一个明显短板: + +**证据链已经“有了”,但还没有被收敛成清晰、稳定、可测试的工程契约。** + +这会直接影响三个面试问题的回答质量: + +1. 工具失败时系统会怎样降级? +2. Verifier 看到的 evidence 到底是否一致、可审计? +3. 这些失败路径和降级行为有没有稳定测试,而不是只靠 runtime 演示? + +--- + +## 当前现状复核 + +### 1. 工具落库入口已经存在,但契约不统一 + +- `QueryLogsTools` 和 `QueryMetricsTools` 通过 `ToolInvocationRecorder.recordEvidenceTool(...)` 记录 evidence tool 调用。 +- `LookupKnowledgeTool` 仍保留独立的 `saveToolInvocation(...)` 路径,自己构造 `ToolInvocation` 实体。 + +这意味着: + +- evidence tool 的公共字段有一套约定 +- knowledge retrieval 又有一套定制字段拼装 + +两者都能工作,但**没有形成统一的“证据调用记录契约”**。 + +### 2. 失败 / 无结果 / 去重命中的语义不够显式 + +当前实现里: + +- `query_logs` 未命中时会返回 `success=false` + `"未找到匹配的日志"` +- `query_metrics` 失败时会返回 `success=false` +- `lookup_knowledge` 去重命中时会返回 `found=false`,但 `tool_invocation.success=true` +- `ToolTraceSummaryService` 通过 `success`、`relevanceLevel`、`dedupReason` 等字段做启发式摘要 + +这些行为在代码里是分散成立的,但**没有被定义成统一契约**,导致: + +- Verifier 能看到的“失败”和“无证据”边界不够稳定 +- 评测时难以明确统计哪些是“调用失败”、哪些是“无命中”、哪些是“已检索过” + +### 3. ChatService 的降级路径有实现,但测试矩阵不完整 + +`ChatService` 已处理: + +- `verifier_output` 缺失或无法解析 → fallback `LOW_CONFID` +- `REJECT` → degraded output +- `LOW_CONFID` → disclaimer output + +但目前缺少成体系的专项验证,尤其是: + +- Verifier 输出非法 JSON +- evidence tool 查询失败 +- knowledge lookup 无有效证据 +- fallback 文案是否只基于 verifier 缺口拼装 + +--- + +## 影响 + +- **面试表达弱化**:你能讲“我有 trace”,但还不能很硬地讲“我的失败路径是有契约和测试保护的”。 +- **评测基础不稳**:后续 P1-B 做 case-based harness 时,统计口径会受 evidence 语义不一致影响。 +- **Verifier 可审计性打折**:当前实现可用,但 still relies on code convention,而不是一份明确收敛后的工程协议。 + +--- + +## 本 issue 目标 + +P1-A 只做三件事: + +1. 收敛 evidence tool 的落库契约,让 `lookup_knowledge`、`query_logs`、`query_metrics` 的公共语义一致。 +2. 明确失败 / 无证据 / 去重 / verifier 非法输出等降级契约,让 `ToolTraceSummaryService` 和 `ChatService` 面向统一状态工作。 +3. 增加专项离线测试,覆盖证据摘要与关键降级路径。 + +--- + +## 范围 + +### In scope + +- `ToolInvocationRecorder` 契约增强 +- `LookupKnowledgeTool` 入库路径收敛 +- `QueryLogsTools` / `QueryMetricsTools` evidence 语义对齐 +- `ToolTraceSummaryService` 对失败 / no-hit / mixed evidence 的摘要规则收敛 +- `ChatService` 对 verifier 非法输出与降级输出的专项测试 +- 与该 change 直接相关的文档、OpenSpec、devflow 记录 + +### Out of scope + +- 不引入新的数据库表或 schema 变更 +- 不扩展新的 evidence tool +- 不做 P1-B 评测集 / harness +- 不做前端 trace UI +- 不处理敏感配置和默认 `mvn test` 离线化 + +--- + +## 预期结果 + +完成后,项目在面试里应能更清楚地表述为: + +```text +我不仅把 Agent 的工具调用落到了库里, +还把 evidence trace、失败语义和 verifier 降级路径收敛成了稳定契约, +并用离线测试覆盖了这些关键失败场景。 +``` + +--- + +## 相关文件 + +- `src/main/java/com/superbiz/agent/service/ToolInvocationRecorder.java` +- `src/main/java/com/superbiz/agent/tool/LookupKnowledgeTool.java` +- `src/main/java/com/superbiz/agent/agent/tool/QueryLogsTools.java` +- `src/main/java/com/superbiz/agent/agent/tool/QueryMetricsTools.java` +- `src/main/java/com/superbiz/agent/service/ToolTraceSummaryService.java` +- `src/main/java/com/superbiz/agent/service/ChatService.java` +- `src/test/java/com/superbiz/agent/service/ChatServiceSequentialAgentTest.java` +- `src/test/java/com/superbiz/agent/tool/LookupKnowledgeToolTest.java` diff --git a/mvp/issues/README.md b/mvp/issues/README.md index a98a0b0..4836085 100644 --- a/mvp/issues/README.md +++ b/mvp/issues/README.md @@ -6,3 +6,4 @@ | ISS-002 | Executor 无约束重复调用 lookup_knowledge | 中 | 已修复 | [ISS-002-executor-unconstrained-lookup.md](ISS-002-executor-unconstrained-lookup.md) | | ISS-003 | MVP 设计与实现 Review 收敛 | 高 | 待规划 | [ISS-003-mvp-design-implementation-review.md](ISS-003-mvp-design-implementation-review.md) | | ISS-004 | Executor 域级检索水位控制(Phase 2) | 低 | 待规划 | [ISS-004-executor-domain-hard-limit.md](ISS-004-executor-domain-hard-limit.md) | +| ISS-005 | 证据链补齐与降级契约收敛 | 高 | 进行中(sm-flow) | [ISS-005-evidence-trace-hardening.md](ISS-005-evidence-trace-hardening.md) | diff --git a/openspec/changes/archive/2026-07-04-evidence-trace-hardening/.openspec.yaml b/openspec/changes/archive/2026-07-04-evidence-trace-hardening/.openspec.yaml new file mode 100644 index 0000000..d86f152 --- /dev/null +++ b/openspec/changes/archive/2026-07-04-evidence-trace-hardening/.openspec.yaml @@ -0,0 +1,2 @@ +schema: spec-driven +created: 2026-07-04 diff --git a/openspec/changes/archive/2026-07-04-evidence-trace-hardening/design.md b/openspec/changes/archive/2026-07-04-evidence-trace-hardening/design.md new file mode 100644 index 0000000..d0d2410 --- /dev/null +++ b/openspec/changes/archive/2026-07-04-evidence-trace-hardening/design.md @@ -0,0 +1,64 @@ +## Context + +The current MVP already has the core pieces required for traceable agent execution: + +- `LookupKnowledgeTool` writes rich retrieval metadata into `tool_invocation` +- `QueryLogsTools` and `QueryMetricsTools` use `ToolInvocationRecorder` +- `ToolTraceSummaryService` turns persisted rows into verifier-facing evidence summaries +- `ChatService` already contains fallback behavior for missing or invalid `verifier_output` + +The gap is no longer “there is no evidence trace”. The gap is that the evidence trace contract is split across two persistence paths and several implicit conventions: + +- `LookupKnowledgeTool` builds `ToolInvocation` rows itself +- the other evidence tools use `ToolInvocationRecorder.recordEvidenceTool(...)` +- “failed”, “no evidence”, “deduped”, and “successful but weak” are inferred differently across tools +- degraded output behavior exists in code but is only lightly covered by tests + +For interview-facing hardening, this slice should make those semantics explicit and testable without changing the database schema or the overall multi-agent workflow. + +## Goals / Non-Goals + +**Goals:** +- Centralize the common persistence contract for evidence-bearing tools. +- Preserve `lookup_knowledge`-specific retrieval fields while removing ad hoc duplication in how evidence rows are created. +- Define stable summarization semantics for: + - successful evidence + - no-hit / no-usable-evidence + - deduped retrievals + - failed evidence queries +- Make `ChatService` fallback and degraded-output paths testable as explicit product behavior. +- Keep the scope small enough to unblock the next P1-B evaluation harness. + +**Non-Goals:** +- No new table, column, or Flyway migration. +- No new public API. +- No new verifier verdict type beyond `PASS` / `LOW_CONFID` / `REJECT`. +- No attempt to redesign planner/executor routing. +- No full offline runtime or end-to-end benchmark harness in this slice. + +## Decisions + +| Decision | Choice | Alternative Considered | Rationale | +|---|---|---|---| +| Evidence persistence ownership | Keep `ToolInvocationRecorder` as the single common entry point | Let each tool continue building `ToolInvocation` rows ad hoc | The recorder already exists and is the right seam for contract hardening. | +| `lookup_knowledge` integration style | Add a richer recorder entry path for retrieval-aware calls | Force `lookup_knowledge` into the same minimal method used by logs/metrics | `lookup_knowledge` carries domain-specific fields such as L0/L1 counts, relevance, dedup reason, and retrieval details that should stay structured. | +| No-evidence semantics | Distinguish failed calls from successful calls that yield no usable evidence | Collapse all non-successful evidence into one bucket | Verifier and future evaluation harnesses need to separate “tool broke” from “tool succeeded but found nothing useful”. | +| Degraded-path hardening | Add focused unit tests around verifier fallback and output shaping | Rely on runtime demo only | Interview value comes from proving the system fails predictably, not just that the happy path ran once. | +| Scope boundary | Keep changes additive and contract-oriented | Expand into P1-B evaluation harness in the same change | This keeps the slice reviewable and avoids mixing infrastructure hardening with evaluation product work. | + +## Risks / Trade-offs + +- [Risk] Tightening persistence semantics could subtly change existing trace summaries. -> Mitigation: keep field names stable and add regression tests around summary output. +- [Risk] Over-generalizing the recorder could make retrieval-specific rows less informative. -> Mitigation: keep a retrieval-aware recording path rather than flattening all tools to the same minimal payload. +- [Risk] Tests may lock in the current fallback copy too aggressively. -> Mitigation: assert protocol-level behavior and key phrases, not brittle full-string snapshots. +- [Risk] `lookup_knowledge` dedup semantics are product-specific and may not fit generic “success/failure” labels cleanly. -> Mitigation: preserve `dedupReason` and treat dedup as a first-class no-new-evidence case in summary logic. + +## Migration Plan + +- No deployment migration is required beyond shipping the code changes. +- Existing `tool_invocation` rows remain valid because this change reuses the same schema. +- Rollback is code-only: revert the recorder/summary/fallback hardening and keep the persisted rows as-is. + +## Open Questions + +- Should P1-B metrics count deduped retrievals as “no-evidence”, or report them as a separate category? This change will preserve enough structure to decide later without another schema change. diff --git a/openspec/changes/archive/2026-07-04-evidence-trace-hardening/proposal.md b/openspec/changes/archive/2026-07-04-evidence-trace-hardening/proposal.md new file mode 100644 index 0000000..50b3cce --- /dev/null +++ b/openspec/changes/archive/2026-07-04-evidence-trace-hardening/proposal.md @@ -0,0 +1,26 @@ +## Why + +The MVP already persists evidence tool invocations and uses a Verifier to judge answer quality, but the current evidence trace semantics are still only partially standardized. For interview-grade agent engineering, the system needs a tighter contract for evidence persistence, no-evidence/failure states, and degraded output behavior, plus focused tests that prove those paths work offline. + +## What Changes + +- Standardize the persisted evidence-tool contract across `lookup_knowledge`, `query_logs`, and `query_metrics`. +- Align how evidence tools represent success, no-hit, deduped, and failed calls so `ToolTraceSummaryService` can summarize them consistently. +- Harden `ChatService` fallback behavior for invalid or missing verifier output and make the degraded-output paths explicitly testable. +- Add focused offline tests for evidence recording, trace summarization, and verifier fallback / degraded output behavior. +- Record this slice as a dedicated P1-A change tied to the interview-focused MVP hardening track. + +## Capabilities + +### New Capabilities +- `evidence-trace-hardening`: Covers standardized evidence invocation persistence, verifier-facing evidence summary semantics, and explicit degraded-output contracts for evidence gaps and verifier failures. + +### Modified Capabilities +- `chat-verifier-agent`: Tightens verifier input evidence semantics and fallback guarantees without changing the high-level planner/executor/verifier workflow. + +## Impact + +- Affected code: `ToolInvocationRecorder`, `LookupKnowledgeTool`, `QueryLogsTools`, `QueryMetricsTools`, `ToolTraceSummaryService`, `ChatService`, and focused test classes. +- Affected runtime behavior: evidence-bearing tools will persist more consistent invocation semantics; verifier fallback and degraded outputs remain additive hardening, not a product-flow rewrite. +- Affected APIs: none. No new endpoint or schema is introduced. +- Non-goals: no new evidence tools, no database migration, no evaluation harness, no trace UI, no security/config cleanup in this slice. diff --git a/openspec/changes/archive/2026-07-04-evidence-trace-hardening/specs/chat-verifier-agent/spec.md b/openspec/changes/archive/2026-07-04-evidence-trace-hardening/specs/chat-verifier-agent/spec.md new file mode 100644 index 0000000..8dc541b --- /dev/null +++ b/openspec/changes/archive/2026-07-04-evidence-trace-hardening/specs/chat-verifier-agent/spec.md @@ -0,0 +1,14 @@ +## ADDED Requirements + +### Requirement: Verifier inputs SHALL tolerate hardened no-evidence semantics +The verifier integration SHALL continue to work when evidence summaries distinguish failed calls, no-hit calls, and deduped retrievals more explicitly. + +#### Scenario: Failed evidence remains a verifier-visible gap +- **WHEN** an evidence-bearing tool invocation fails +- **THEN** the verifier-facing trace summary SHALL preserve that failure as a gap +- **AND** the verifier flow SHALL continue without crashing + +#### Scenario: Deduped retrievals do not count as fresh support +- **WHEN** the verifier-facing trace summary contains deduped `lookup_knowledge` entries +- **THEN** those entries SHALL be treated as no-new-evidence +- **AND** they SHALL NOT be interpreted as fresh direct support for the answer diff --git a/openspec/changes/archive/2026-07-04-evidence-trace-hardening/specs/evidence-trace-hardening/spec.md b/openspec/changes/archive/2026-07-04-evidence-trace-hardening/specs/evidence-trace-hardening/spec.md new file mode 100644 index 0000000..a10a114 --- /dev/null +++ b/openspec/changes/archive/2026-07-04-evidence-trace-hardening/specs/evidence-trace-hardening/spec.md @@ -0,0 +1,82 @@ +## ADDED Requirements + +### Requirement: Evidence-bearing tools SHALL persist a unified invocation contract +The system SHALL persist evidence-bearing tool calls through a unified contract that guarantees the same baseline fields across `lookup_knowledge`, `query_logs`, and `query_metrics`. + +#### Scenario: Common evidence fields are always persisted +- **WHEN** an evidence-bearing tool finishes a call +- **THEN** the persisted `tool_invocation` row SHALL include `session_id`, `tool_name`, `input_params`, `duration_ms`, `success`, and an output preview or explicit no-output state + +#### Scenario: Retrieval-aware tools preserve structured retrieval fields +- **WHEN** `lookup_knowledge` persists a tool invocation +- **THEN** the row SHALL preserve retrieval-specific fields such as retrieval layer, L0/L1 counts, relevance level, dedup reason, and retrieval details +- **AND** those fields SHALL be written through the same recorder contract rather than by ad hoc row construction in unrelated code paths + +### Requirement: Evidence trace semantics SHALL distinguish failure from no-evidence outcomes +The system SHALL keep failed calls separate from successful calls that return no usable evidence. + +#### Scenario: Tool failure is preserved as failure +- **WHEN** an evidence-bearing tool throws, times out, or returns an execution error +- **THEN** the persisted row SHALL set `success=false` +- **AND** it SHALL preserve an `error_message` explaining the failure + +#### Scenario: No usable evidence is preserved without pretending success +- **WHEN** an evidence-bearing tool completes normally but yields no usable evidence for the verifier +- **THEN** the persisted contract SHALL preserve that the call completed +- **AND** the verifier-facing summary SHALL describe it as a no-evidence outcome rather than direct support + +#### Scenario: Deduped retrieval remains auditable +- **WHEN** `lookup_knowledge` is blocked by session-level deduplication +- **THEN** the persisted row SHALL preserve the dedup reason +- **AND** the verifier-facing summary SHALL treat that row as no-new-evidence rather than as a fresh supporting hit + +### Requirement: Verifier-facing evidence summaries SHALL use stable no-evidence rules +The system SHALL summarize persisted tool rows into verifier-facing evidence entries using stable rules for success, failure, no-hit, and deduped outcomes. + +#### Scenario: Failed evidence calls remain visible in the summary +- **WHEN** `ToolTraceSummaryService` processes failed evidence-bearing tool rows +- **THEN** the summary SHALL retain them +- **AND** it SHALL mark them as unsuccessful evidence with an output summary that explains the gap + +#### Scenario: No-hit and deduped calls do not upgrade evidence level +- **WHEN** `ToolTraceSummaryService` processes rows that returned no usable evidence or were deduped +- **THEN** those rows SHALL NOT be promoted to direct or indirect evidence +- **AND** their counts SHALL still be reflected in the merged summary entry + +#### Scenario: Successful evidence keeps the strongest available support +- **WHEN** multiple rows for the same tool and topic domain are merged +- **THEN** the summary SHALL preserve the strongest successful evidence level among them +- **AND** it SHALL also retain repeated-call, failed-call, and no-hit counts for auditability + +### Requirement: ChatService SHALL degrade predictably on verifier output failures +The system SHALL treat missing or invalid verifier output as a bounded degraded path instead of an unstructured runtime error. + +#### Scenario: Missing verifier output falls back to LOW_CONFID +- **WHEN** the verifier step completes without a usable `verifier_output` +- **THEN** `ChatService` SHALL fall back to a `LOW_CONFID` decision +- **AND** the final user-facing output SHALL use the fixed low-confidence protocol + +#### Scenario: Invalid verifier JSON falls back to LOW_CONFID +- **WHEN** the verifier returns malformed or non-parseable JSON +- **THEN** `ChatService` SHALL fall back to a `LOW_CONFID` decision +- **AND** the fallback SHALL still persist a verifier-evaluation record + +#### Scenario: REJECT output hides unverified raw answer text +- **WHEN** the final verifier decision is `REJECT` +- **THEN** the user-facing output SHALL use the degraded template +- **AND** it SHALL NOT pass through the raw executor answer + +### Requirement: Evidence-trace hardening SHALL be covered by focused offline tests +The system SHALL provide offline tests for the hardened evidence contract and degraded-output behavior. + +#### Scenario: Evidence recorder contract is tested offline +- **WHEN** the test suite runs the focused recorder tests +- **THEN** it SHALL verify the persisted semantics for successful, failed, and no-evidence evidence-tool calls without requiring external infrastructure + +#### Scenario: Trace summary hardening is tested offline +- **WHEN** the test suite runs the focused trace-summary tests +- **THEN** it SHALL verify merged summary behavior for mixed success, failure, no-hit, and deduped tool rows + +#### Scenario: Verifier fallback behavior is tested offline +- **WHEN** the test suite runs the focused `ChatService` fallback tests +- **THEN** it SHALL verify the missing-output, invalid-JSON, `LOW_CONFID`, and `REJECT` degraded paths without requiring a real LLM or database diff --git a/openspec/changes/archive/2026-07-04-evidence-trace-hardening/tasks.md b/openspec/changes/archive/2026-07-04-evidence-trace-hardening/tasks.md new file mode 100644 index 0000000..d638c47 --- /dev/null +++ b/openspec/changes/archive/2026-07-04-evidence-trace-hardening/tasks.md @@ -0,0 +1,22 @@ +## 1. Evidence Persistence Contract + +- [x] 1.1 Extend `ToolInvocationRecorder` with a richer evidence-recording path that can preserve retrieval-aware fields as well as common evidence fields. +- [x] 1.2 Refactor `LookupKnowledgeTool` to persist `tool_invocation` rows through `ToolInvocationRecorder` instead of its own ad hoc row-construction path. +- [x] 1.3 Align `QueryLogsTools` and `QueryMetricsTools` no-hit / failure payloads with the hardened evidence contract. + +## 2. Verifier-Facing Summary Semantics + +- [x] 2.1 Harden `ToolTraceSummaryService` so failed, no-hit, and deduped evidence rows are summarized with stable no-evidence semantics. +- [x] 2.2 Preserve merged-call counts for repeated hits, failures, and no-new-evidence rows without overstating evidence strength. + +## 3. Chat Degraded Paths + +- [x] 3.1 Add focused `ChatService` tests for missing verifier output fallback to `LOW_CONFID`. +- [x] 3.2 Add focused `ChatService` tests for invalid verifier JSON fallback to `LOW_CONFID`. +- [x] 3.3 Add focused `ChatService` tests that `REJECT` output uses the degraded template and does not leak raw executor answer content. + +## 4. Verification + +- [x] 4.1 Add focused offline tests for the recorder contract and `ToolTraceSummaryService`. +- [x] 4.2 Run targeted test commands for the new/updated offline tests. +- [x] 4.3 Run compile verification. diff --git a/openspec/specs/chat-verifier-agent/spec.md b/openspec/specs/chat-verifier-agent/spec.md index f69d8ed..da5e23c 100644 --- a/openspec/specs/chat-verifier-agent/spec.md +++ b/openspec/specs/chat-verifier-agent/spec.md @@ -190,3 +190,17 @@ Verifier facts SHALL be linkable to the evidence summaries used during verificat - **WHEN** the ChatService persists `verifier_evaluation` - **THEN** it SHALL include `traceability_version` - **AND** it SHALL include the `tool_trace_summary` snapshot used by the Verifier + +### Requirement: Verifier inputs SHALL tolerate hardened no-evidence semantics +The verifier integration SHALL continue to work when evidence summaries distinguish failed calls, no-hit calls, and deduped retrievals more explicitly. + +#### Scenario: Failed evidence remains a verifier-visible gap +- **WHEN** an evidence-bearing tool invocation fails +- **THEN** the verifier-facing trace summary SHALL preserve that failure as a gap +- **AND** the verifier flow SHALL continue without crashing + +#### Scenario: Deduped retrievals do not count as fresh support +- **WHEN** the verifier-facing trace summary contains deduped `lookup_knowledge` entries +- **THEN** those entries SHALL be treated as no-new-evidence +- **AND** they SHALL NOT be interpreted as fresh direct support for the answer + diff --git a/openspec/specs/evidence-trace-hardening/spec.md b/openspec/specs/evidence-trace-hardening/spec.md new file mode 100644 index 0000000..bb34847 --- /dev/null +++ b/openspec/specs/evidence-trace-hardening/spec.md @@ -0,0 +1,86 @@ +# evidence-trace-hardening Specification + +## Purpose +TBD - created by archiving change evidence-trace-hardening. Update Purpose after archive. +## Requirements +### Requirement: Evidence-bearing tools SHALL persist a unified invocation contract +The system SHALL persist evidence-bearing tool calls through a unified contract that guarantees the same baseline fields across `lookup_knowledge`, `query_logs`, and `query_metrics`. + +#### Scenario: Common evidence fields are always persisted +- **WHEN** an evidence-bearing tool finishes a call +- **THEN** the persisted `tool_invocation` row SHALL include `session_id`, `tool_name`, `input_params`, `duration_ms`, `success`, and an output preview or explicit no-output state + +#### Scenario: Retrieval-aware tools preserve structured retrieval fields +- **WHEN** `lookup_knowledge` persists a tool invocation +- **THEN** the row SHALL preserve retrieval-specific fields such as retrieval layer, L0/L1 counts, relevance level, dedup reason, and retrieval details +- **AND** those fields SHALL be written through the same recorder contract rather than by ad hoc row construction in unrelated code paths + +### Requirement: Evidence trace semantics SHALL distinguish failure from no-evidence outcomes +The system SHALL keep failed calls separate from successful calls that return no usable evidence. + +#### Scenario: Tool failure is preserved as failure +- **WHEN** an evidence-bearing tool throws, times out, or returns an execution error +- **THEN** the persisted row SHALL set `success=false` +- **AND** it SHALL preserve an `error_message` explaining the failure + +#### Scenario: No usable evidence is preserved without pretending success +- **WHEN** an evidence-bearing tool completes normally but yields no usable evidence for the verifier +- **THEN** the persisted contract SHALL preserve that the call completed +- **AND** the verifier-facing summary SHALL describe it as a no-evidence outcome rather than direct support + +#### Scenario: Deduped retrieval remains auditable +- **WHEN** `lookup_knowledge` is blocked by session-level deduplication +- **THEN** the persisted row SHALL preserve the dedup reason +- **AND** the verifier-facing summary SHALL treat that row as no-new-evidence rather than as a fresh supporting hit + +### Requirement: Verifier-facing evidence summaries SHALL use stable no-evidence rules +The system SHALL summarize persisted tool rows into verifier-facing evidence entries using stable rules for success, failure, no-hit, and deduped outcomes. + +#### Scenario: Failed evidence calls remain visible in the summary +- **WHEN** `ToolTraceSummaryService` processes failed evidence-bearing tool rows +- **THEN** the summary SHALL retain them +- **AND** it SHALL mark them as unsuccessful evidence with an output summary that explains the gap + +#### Scenario: No-hit and deduped calls do not upgrade evidence level +- **WHEN** `ToolTraceSummaryService` processes rows that returned no usable evidence or were deduped +- **THEN** those rows SHALL NOT be promoted to direct or indirect evidence +- **AND** their counts SHALL still be reflected in the merged summary entry + +#### Scenario: Successful evidence keeps the strongest available support +- **WHEN** multiple rows for the same tool and topic domain are merged +- **THEN** the summary SHALL preserve the strongest successful evidence level among them +- **AND** it SHALL also retain repeated-call, failed-call, and no-hit counts for auditability + +### Requirement: ChatService SHALL degrade predictably on verifier output failures +The system SHALL treat missing or invalid verifier output as a bounded degraded path instead of an unstructured runtime error. + +#### Scenario: Missing verifier output falls back to LOW_CONFID +- **WHEN** the verifier step completes without a usable `verifier_output` +- **THEN** `ChatService` SHALL fall back to a `LOW_CONFID` decision +- **AND** the final user-facing output SHALL use the fixed low-confidence protocol + +#### Scenario: Invalid verifier JSON falls back to LOW_CONFID +- **WHEN** the verifier returns malformed or non-parseable JSON +- **THEN** `ChatService` SHALL fall back to a `LOW_CONFID` decision +- **AND** the fallback SHALL still persist a verifier-evaluation record + +#### Scenario: REJECT output hides unverified raw answer text +- **WHEN** the final verifier decision is `REJECT` +- **THEN** the user-facing output SHALL use the degraded template +- **AND** it SHALL NOT pass through the raw executor answer + +### Requirement: Evidence-trace hardening SHALL be covered by focused offline tests +The system SHALL provide offline tests for the hardened evidence contract and degraded-output behavior. + +#### Scenario: Evidence recorder contract is tested offline +- **WHEN** the test suite runs the focused recorder tests +- **THEN** it SHALL verify the persisted semantics for successful, failed, and no-evidence evidence-tool calls without requiring external infrastructure + +#### Scenario: Trace summary hardening is tested offline +- **WHEN** the test suite runs the focused trace-summary tests +- **THEN** it SHALL verify merged summary behavior for mixed success, failure, no-hit, and deduped tool rows + +#### Scenario: Verifier fallback behavior is tested offline +- **WHEN** the test suite runs the focused `ChatService` fallback tests +- **THEN** it SHALL verify the missing-output, invalid-JSON, `LOW_CONFID`, and `REJECT` degraded paths without requiring a real LLM or database + diff --git a/src/main/java/com/superbiz/agent/agent/tool/QueryLogsTools.java b/src/main/java/com/superbiz/agent/agent/tool/QueryLogsTools.java index af4ac39..d432981 100644 --- a/src/main/java/com/superbiz/agent/agent/tool/QueryLogsTools.java +++ b/src/main/java/com/superbiz/agent/agent/tool/QueryLogsTools.java @@ -131,13 +131,15 @@ public class QueryLogsTools { output.setMessage(String.format("共有 %d 个可用的日志主题。建议使用默认地域 'ap-guangzhou' 或省略 region 参数", topics.size())); String response = objectMapper.writerWithDefaultPrettyPrinter().writeValueAsString(output); - recordInvocation(startTime, "get_available_log_topics", null, null, null, response, true, null, "logs"); + recordInvocation("get_available_log_topics", startTime, "get_available_log_topics", null, null, null, + response, true, null, "logs", ToolInvocationRecorder.EVIDENCE_STATUS_SUPPORTED); return response; } catch (Exception e) { logger.error("获取日志主题列表失败", e); String response = "{\"success\":false,\"message\":\"获取日志主题列表失败: " + e.getMessage() + "\"}"; - recordInvocation(startTime, "get_available_log_topics", null, null, null, response, false, e.getMessage(), "logs"); + recordInvocation("get_available_log_topics", startTime, "get_available_log_topics", null, null, null, + response, false, e.getMessage(), "logs", ToolInvocationRecorder.EVIDENCE_STATUS_FAILED); return response; } } @@ -191,8 +193,9 @@ public class QueryLogsTools { } else { // 真实模式:调用 CLS API(这里预留接口,后续实现) String response = buildErrorResponse("CLS 真实查询尚未实现,请启用 mock 模式进行测试"); - recordInvocation(startTime, safeQuery, region, logTopic, actualLimit, response, false, - "CLS 真实查询尚未实现,请启用 mock 模式进行测试", normalizeTopicDomain(logTopic)); + recordInvocation("query_logs", startTime, safeQuery, region, logTopic, actualLimit, response, false, + "CLS 真实查询尚未实现,请启用 mock 模式进行测试", normalizeTopicDomain(logTopic), + ToolInvocationRecorder.EVIDENCE_STATUS_FAILED); return response; } @@ -208,23 +211,26 @@ public class QueryLogsTools { String jsonResult = objectMapper.writerWithDefaultPrettyPrinter().writeValueAsString(output); logger.info("日志查询完成: 找到 {} 条日志", logEntries.size()); - recordInvocation(startTime, safeQuery, region, logTopic, actualLimit, jsonResult, - !logEntries.isEmpty(), logEntries.isEmpty() ? "未找到匹配的日志" : null, - normalizeTopicDomain(logTopic)); + recordInvocation("query_logs", startTime, safeQuery, region, logTopic, actualLimit, jsonResult, + true, null, normalizeTopicDomain(logTopic), + logEntries.isEmpty() + ? ToolInvocationRecorder.EVIDENCE_STATUS_NO_EVIDENCE + : ToolInvocationRecorder.EVIDENCE_STATUS_SUPPORTED); return jsonResult; } catch (Exception e) { logger.error("查询日志失败", e); String response = buildErrorResponse("查询失败: " + e.getMessage()); - recordInvocation(startTime, safeQuery, region, logTopic, actualLimit, response, false, - e.getMessage(), normalizeTopicDomain(logTopic)); + recordInvocation("query_logs", startTime, safeQuery, region, logTopic, actualLimit, response, false, + e.getMessage(), normalizeTopicDomain(logTopic), ToolInvocationRecorder.EVIDENCE_STATUS_FAILED); return response; } } - private void recordInvocation(long startTime, String query, String region, String logTopic, Integer limit, - String output, boolean success, String errorMessage, String topicDomain) { + private void recordInvocation(String toolName, long startTime, String query, String region, String logTopic, Integer limit, + String output, boolean success, String errorMessage, String topicDomain, + String evidenceStatus) { Map input = new HashMap<>(); input.put("query", query == null || query.isBlank() ? "DEFAULT_QUERY" : query); if (region != null) { @@ -239,13 +245,15 @@ public class QueryLogsTools { input.put("mock_enabled", mockEnabled); toolInvocationRecorder.recordEvidenceTool( - "query_logs", + toolName, input, output, success, startTime, errorMessage, - topicDomain + topicDomain, + evidenceStatus, + Map.of("log_topic", logTopic == null ? "" : logTopic) ); } diff --git a/src/main/java/com/superbiz/agent/agent/tool/QueryMetricsTools.java b/src/main/java/com/superbiz/agent/agent/tool/QueryMetricsTools.java index e9f044d..7d2f9af 100644 --- a/src/main/java/com/superbiz/agent/agent/tool/QueryMetricsTools.java +++ b/src/main/java/com/superbiz/agent/agent/tool/QueryMetricsTools.java @@ -81,7 +81,7 @@ public class QueryMetricsTools { if (!"success".equals(result.getStatus())) { String response = buildErrorResponse("Prometheus API 返回非成功状态: " + result.getStatus(), result.getError()); - recordInvocation(startTime, response, false, result.getError()); + recordInvocation(startTime, response, false, result.getError(), ToolInvocationRecorder.EVIDENCE_STATUS_FAILED); return response; } @@ -119,19 +119,22 @@ public class QueryMetricsTools { String jsonResult = objectMapper.writerWithDefaultPrettyPrinter().writeValueAsString(output); logger.info("Prometheus 告警查询完成: 找到 {} 个告警", simplifiedAlerts.size()); - recordInvocation(startTime, jsonResult, true, null); + recordInvocation(startTime, jsonResult, true, null, + simplifiedAlerts.isEmpty() + ? ToolInvocationRecorder.EVIDENCE_STATUS_NO_EVIDENCE + : ToolInvocationRecorder.EVIDENCE_STATUS_SUPPORTED); return jsonResult; } catch (Exception e) { logger.error("查询 Prometheus 告警失败", e); String response = buildErrorResponse("查询失败", e.getMessage()); - recordInvocation(startTime, response, false, e.getMessage()); + recordInvocation(startTime, response, false, e.getMessage(), ToolInvocationRecorder.EVIDENCE_STATUS_FAILED); return response; } } - private void recordInvocation(long startTime, String output, boolean success, String errorMessage) { + private void recordInvocation(long startTime, String output, boolean success, String errorMessage, String evidenceStatus) { toolInvocationRecorder.recordEvidenceTool( "query_metrics", Map.of("query", "active_prometheus_alerts", "mock_enabled", mockEnabled), @@ -139,7 +142,9 @@ public class QueryMetricsTools { success, startTime, errorMessage, - "prometheus_alerts" + "prometheus_alerts", + evidenceStatus, + Map.of("metric_family", "prometheus_alerts") ); } diff --git a/src/main/java/com/superbiz/agent/service/ToolInvocationRecorder.java b/src/main/java/com/superbiz/agent/service/ToolInvocationRecorder.java index 893b160..b3cda51 100644 --- a/src/main/java/com/superbiz/agent/service/ToolInvocationRecorder.java +++ b/src/main/java/com/superbiz/agent/service/ToolInvocationRecorder.java @@ -3,11 +3,16 @@ package com.superbiz.agent.service; import com.fasterxml.jackson.core.JsonProcessingException; import com.fasterxml.jackson.databind.ObjectMapper; import com.superbiz.agent.domain.entity.ToolInvocation; +import com.superbiz.agent.dto.LookupResult; import com.superbiz.agent.repository.ToolInvocationRepository; +import com.superbiz.agent.dto.KnowledgeEntry; +import com.superbiz.agent.service.VectorSearchService; import com.superbiz.agent.util.SessionContextHolder; +import lombok.Builder; import lombok.extern.slf4j.Slf4j; import org.springframework.stereotype.Service; +import java.util.ArrayList; import java.util.LinkedHashMap; import java.util.List; import java.util.Map; @@ -21,6 +26,10 @@ import java.util.UUID; public class ToolInvocationRecorder { private static final int OUTPUT_PREVIEW_LIMIT = 500; + public static final String EVIDENCE_STATUS_SUPPORTED = "supported"; + public static final String EVIDENCE_STATUS_NO_EVIDENCE = "no_evidence"; + public static final String EVIDENCE_STATUS_DEDUPED = "deduped"; + public static final String EVIDENCE_STATUS_FAILED = "failed"; private final ToolInvocationRepository toolInvocationRepository; private final ObjectMapper objectMapper; @@ -52,12 +61,29 @@ public class ToolInvocationRecorder { long startTimeMillis, String errorMessage, String topicDomain) { + recordEvidenceTool(toolName, inputParams, output, success, startTimeMillis, errorMessage, topicDomain, + success ? EVIDENCE_STATUS_SUPPORTED : EVIDENCE_STATUS_FAILED, Map.of()); + } + + public void recordEvidenceTool(String toolName, + Map inputParams, + String output, + boolean success, + long startTimeMillis, + String errorMessage, + String topicDomain, + String evidenceStatus, + Map extraDetails) { String outputPreview = preview(output); Map details = new LinkedHashMap<>(); details.put("trace_id", UUID.randomUUID().toString()); if (topicDomain != null && !topicDomain.isBlank()) { details.put("retrieved_domains", List.of(topicDomain)); } + details.put("evidence_status", normalizeEvidenceStatus(success, evidenceStatus)); + if (extraDetails != null && !extraDetails.isEmpty()) { + details.putAll(extraDetails); + } ToolInvocation invocation = ToolInvocation.builder() .toolName(toolName) @@ -73,6 +99,67 @@ public class ToolInvocationRecorder { save(invocation); } + public void recordLookupKnowledge(LookupKnowledgeRecord record) { + Map details = new LinkedHashMap<>(); + details.put("trace_id", UUID.randomUUID().toString()); + if (record.l0MatchCount() != null) { + details.put("l0_match_count", record.l0MatchCount()); + } + if (record.l0Titles() != null && !record.l0Titles().isEmpty()) { + details.put("l0_titles", record.l0Titles()); + } + if (record.l1TopScore() != null) { + details.put("l1_top_score", record.l1TopScore()); + } + if (record.l1TopSimilarity() != null) { + details.put("l1_top_similarity", record.l1TopSimilarity()); + } + if (record.l1MatchCount() != null) { + details.put("l1_match_count", record.l1MatchCount()); + } + if (record.l1Scores() != null && !record.l1Scores().isEmpty()) { + details.put("l1_scores", record.l1Scores()); + } + if (record.relevanceLevel() != null) { + details.put("relevance_level", record.relevanceLevel()); + } + if (record.completenessHint() != null) { + details.put("completeness_hint", record.completenessHint()); + } + if (record.domain() != null && !record.domain().isBlank()) { + details.put("retrieved_domains", List.of(record.domain())); + } + if (record.dedupReason() != null) { + details.put("dedup_reason", record.dedupReason()); + } + details.put("evidence_status", normalizeEvidenceStatus(record.success(), record.evidenceStatus())); + + ToolInvocation invocation = ToolInvocation.builder() + .toolName("lookup_knowledge") + .inputParams(toJson(Map.of("query", record.query()))) + .outputPreview(preview(record.outputPreview())) + .outputLength(record.outputLength()) + .retrievalLayer(record.retrievalLayer()) + .l0MatchCount(record.l0MatchCount()) + .l1MatchCount(record.l1MatchCount()) + .isTruncated(Boolean.TRUE.equals(record.truncated())) + .retrievalDetails(toJson(details)) + .relevanceLevel(record.relevanceLevel()) + .dedupReason(record.dedupReason()) + .durationMs(record.durationMs()) + .success(record.success()) + .errorMessage(record.errorMessage()) + .build(); + save(invocation); + } + + private String normalizeEvidenceStatus(boolean success, String evidenceStatus) { + if (evidenceStatus != null && !evidenceStatus.isBlank()) { + return evidenceStatus; + } + return success ? EVIDENCE_STATUS_SUPPORTED : EVIDENCE_STATUS_FAILED; + } + private String preview(String output) { if (output == null) { return null; @@ -90,4 +177,105 @@ public class ToolInvocationRecorder { return "{}"; } } + + @Builder + public record LookupKnowledgeRecord( + String query, + String outputPreview, + Integer outputLength, + String retrievalLayer, + Integer l0MatchCount, + Integer l1MatchCount, + Boolean truncated, + String relevanceLevel, + String completenessHint, + String domain, + String dedupReason, + Integer durationMs, + boolean success, + String evidenceStatus, + String errorMessage, + List l0Titles, + Double l1TopScore, + Double l1TopSimilarity, + List l1Scores + ) { + public static LookupKnowledgeRecord from(String query, + List l0Matches, + List l1Results, + boolean highConfidence, + LookupResult result, + String domain, + String dedupReason, + int durationMs, + double l1TopSimilarity) { + boolean hasL0 = l0Matches != null && !l0Matches.isEmpty(); + boolean hasL1 = l1Results != null && !l1Results.isEmpty(); + String layer; + if (hasL0 && !highConfidence) { + layer = "L0+L1"; + } else if (hasL0) { + layer = "L0"; + } else if (hasL1) { + layer = "L1"; + } else { + layer = null; + } + + String outputPreview = null; + int outputLength = 0; + boolean truncated = false; + if (result != null && result.getPrimary() != null && result.getPrimary().getContent() != null) { + outputPreview = result.getPrimary().getContent(); + outputLength = outputPreview.length(); + truncated = outputLength > OUTPUT_PREVIEW_LIMIT; + } else if (hasL1 && l1Results.get(0).getContent() != null) { + outputPreview = l1Results.get(0).getContent(); + outputLength = outputPreview.length(); + truncated = outputLength > OUTPUT_PREVIEW_LIMIT; + } + + String evidenceStatus = EVIDENCE_STATUS_SUPPORTED; + if (dedupReason != null) { + evidenceStatus = EVIDENCE_STATUS_DEDUPED; + } else if (result == null || !result.isFound()) { + evidenceStatus = EVIDENCE_STATUS_NO_EVIDENCE; + } + + List l0Titles = new ArrayList<>(); + if (hasL0) { + for (int i = 0; i < Math.min(3, l0Matches.size()); i++) { + l0Titles.add(l0Matches.get(i).getTitle()); + } + } + + List l1Scores = new ArrayList<>(); + if (hasL1) { + for (int i = 0; i < Math.min(3, l1Results.size()); i++) { + l1Scores.add((double) l1Results.get(i).getScore()); + } + } + + return LookupKnowledgeRecord.builder() + .query(query) + .outputPreview(outputPreview) + .outputLength(outputLength) + .retrievalLayer(layer) + .l0MatchCount(hasL0 ? l0Matches.size() : null) + .l1MatchCount(hasL1 ? l1Results.size() : null) + .truncated(truncated) + .relevanceLevel(result != null ? result.getRelevanceLevel() : null) + .completenessHint(result != null ? result.getCompletenessHint() : null) + .domain(domain) + .dedupReason(dedupReason) + .durationMs(durationMs) + .success(true) + .evidenceStatus(evidenceStatus) + .l0Titles(l0Titles) + .l1TopScore(hasL1 ? (double) l1Results.get(0).getScore() : null) + .l1TopSimilarity(hasL1 ? l1TopSimilarity : null) + .l1Scores(l1Scores) + .build(); + } + } } diff --git a/src/main/java/com/superbiz/agent/service/ToolTraceSummaryService.java b/src/main/java/com/superbiz/agent/service/ToolTraceSummaryService.java index 1a8077f..ece980c 100644 --- a/src/main/java/com/superbiz/agent/service/ToolTraceSummaryService.java +++ b/src/main/java/com/superbiz/agent/service/ToolTraceSummaryService.java @@ -107,6 +107,7 @@ public class ToolTraceSummaryService { } private String extractOutputSummary(ToolInvocation invocation, String topicDomain) { + String evidenceStatus = extractEvidenceStatus(invocation); if (!Boolean.TRUE.equals(invocation.getSuccess())) { if (invocation.getErrorMessage() != null && !invocation.getErrorMessage().isBlank()) { return "call failed: " + truncate(invocation.getErrorMessage(), 120); @@ -114,6 +115,17 @@ public class ToolTraceSummaryService { return "no usable evidence returned"; } + if (ToolInvocationRecorder.EVIDENCE_STATUS_DEDUPED.equals(evidenceStatus)) { + return "retrieval skipped because the same document was already used in this session"; + } + + if (ToolInvocationRecorder.EVIDENCE_STATUS_NO_EVIDENCE.equals(evidenceStatus)) { + if (invocation.getOutputPreview() != null && !invocation.getOutputPreview().isBlank()) { + return "completed without usable evidence: " + truncate(invocation.getOutputPreview(), 120); + } + return "completed without usable evidence"; + } + if ("lookup_knowledge".equals(invocation.getToolName())) { String relevance = invocation.getRelevanceLevel() != null ? invocation.getRelevanceLevel() : "UNKNOWN"; String preview = invocation.getOutputPreview() != null && !invocation.getOutputPreview().isBlank() @@ -129,18 +141,46 @@ public class ToolTraceSummaryService { } private String determineEvidenceLevel(ToolInvocation invocation) { + String evidenceStatus = extractEvidenceStatus(invocation); if (!Boolean.TRUE.equals(invocation.getSuccess())) { return "none"; } + if (ToolInvocationRecorder.EVIDENCE_STATUS_NO_EVIDENCE.equals(evidenceStatus) + || ToolInvocationRecorder.EVIDENCE_STATUS_DEDUPED.equals(evidenceStatus)) { + return "none"; + } if ("PRECISE".equals(invocation.getRelevanceLevel()) || "HIGHLY_RELEVANT".equals(invocation.getRelevanceLevel())) { return "direct"; } if ("REFERENCE".equals(invocation.getRelevanceLevel())) { return "indirect"; } + if (EVIDENCE_TOOLS.contains(invocation.getToolName())) { + return "direct"; + } return "none"; } + private String extractEvidenceStatus(ToolInvocation invocation) { + if (invocation.getRetrievalDetails() == null || invocation.getRetrievalDetails().isBlank()) { + return Boolean.TRUE.equals(invocation.getSuccess()) + ? ToolInvocationRecorder.EVIDENCE_STATUS_SUPPORTED + : ToolInvocationRecorder.EVIDENCE_STATUS_FAILED; + } + try { + Map details = objectMapper.readValue(invocation.getRetrievalDetails(), MAP_TYPE); + Object evidenceStatus = details.get("evidence_status"); + if (evidenceStatus != null) { + return String.valueOf(evidenceStatus); + } + } catch (Exception e) { + log.debug("Failed to parse evidence_status", e); + } + return Boolean.TRUE.equals(invocation.getSuccess()) + ? ToolInvocationRecorder.EVIDENCE_STATUS_SUPPORTED + : ToolInvocationRecorder.EVIDENCE_STATUS_FAILED; + } + private List extractStringList(Object value) { if (!(value instanceof List list) || list.isEmpty()) { return List.of(); @@ -224,13 +264,22 @@ public class ToolTraceSummaryService { inputSummary = extractInputSummary(invocation); } - boolean invocationSuccess = Boolean.TRUE.equals(invocation.getSuccess()); - if (!invocationSuccess) { + String evidenceStatus = extractEvidenceStatus(invocation); + if (!Boolean.TRUE.equals(invocation.getSuccess())) { failedCount++; + if (outputSummary == null || outputSummary.isBlank()) { + outputSummary = extractOutputSummary(invocation, topicDomain); + } return; } - if (invocation.getDedupReason() != null) { + + if (ToolInvocationRecorder.EVIDENCE_STATUS_NO_EVIDENCE.equals(evidenceStatus) + || ToolInvocationRecorder.EVIDENCE_STATUS_DEDUPED.equals(evidenceStatus)) { noHitCount++; + if (outputSummary == null || outputSummary.isBlank()) { + outputSummary = extractOutputSummary(invocation, topicDomain); + } + return; } String invocationEvidenceLevel = determineEvidenceLevel(invocation); diff --git a/src/main/java/com/superbiz/agent/tool/LookupKnowledgeTool.java b/src/main/java/com/superbiz/agent/tool/LookupKnowledgeTool.java index 1bf36f7..4d1170f 100644 --- a/src/main/java/com/superbiz/agent/tool/LookupKnowledgeTool.java +++ b/src/main/java/com/superbiz/agent/tool/LookupKnowledgeTool.java @@ -14,6 +14,7 @@ import org.springframework.beans.factory.annotation.Value; import org.springframework.stereotype.Component; import java.util.List; +import java.util.Locale; import java.util.stream.Collectors; /** @@ -309,131 +310,30 @@ public class LookupKnowledgeTool { String sessionId = SessionContextHolder.getSessionId(); if (sessionId == null) return; - boolean hasL0 = l0Matches != null && !l0Matches.isEmpty(); - boolean hasL1 = l1Results != null && !l1Results.isEmpty(); long duration = System.currentTimeMillis() - startTime; + double l1TopSimilarity = (l1Results != null && !l1Results.isEmpty()) + ? normalizeL2(l1Results.get(0).getScore()) + : -1; - String layer; - String outputPreview = null; - int outputLength = 0; - int l0Count = 0; - int l1Count = 0; - boolean truncated = false; - - if (hasL0 && !highConfidence) { - layer = "L0+L1"; - l0Count = l0Matches.size(); - l1Count = l1Results.size(); - } else if (hasL0) { - layer = "L0"; - l0Count = l0Matches.size(); - } else if (hasL1) { - layer = "L1"; - l1Count = l1Results.size(); - } else { - layer = null; - } - - // output_preview - if (result != null && result.getPrimary() != null && result.getPrimary().getContent() != null) { - String content = result.getPrimary().getContent(); - outputLength = content.length(); - if (content.length() > 500) { - outputPreview = content.substring(0, 500) + "..."; - truncated = true; - } else { - outputPreview = content; - } - } else if (l1Results != null && !l1Results.isEmpty() && l1Results.get(0).getContent() != null) { - String content = l1Results.get(0).getContent(); - outputLength = content.length(); - if (content.length() > 500) { - outputPreview = content.substring(0, 500) + "..."; - truncated = true; - } else { - outputPreview = content; - } - } - - // L1 top score + similarity - float l1TopScore = (hasL1) ? l1Results.get(0).getScore() : -1; - double l1TopSimilarity = (hasL1) ? normalizeL2(l1TopScore) : -1; - - // 构建检索明细 JSON(扩展版) - StringBuilder details = new StringBuilder("{"); - if (hasL0) { - details.append("\"l0_match_count\":").append(l0Count).append(","); - details.append("\"l0_titles\":["); - for (int i = 0; i < Math.min(3, l0Matches.size()); i++) { - if (i > 0) details.append(","); - details.append("\"").append(escapeJson(l0Matches.get(i).getTitle())).append("\""); - } - details.append("],"); - } - if (hasL1) { - details.append("\"l1_top_score\":").append(String.format("%.4f", l1TopScore)).append(","); - details.append("\"l1_top_similarity\":").append(String.format("%.4f", l1TopSimilarity)).append(","); - details.append("\"l1_match_count\":").append(l1Count).append(","); - details.append("\"l1_scores\":["); - for (int i = 0; i < Math.min(3, l1Results.size()); i++) { - if (i > 0) details.append(","); - details.append(String.format("%.4f", l1Results.get(i).getScore())); - } - details.append("],"); - } - // 归一化信息 - if (result != null && result.getRelevanceLevel() != null) { - details.append("\"relevance_level\":\"").append(result.getRelevanceLevel()).append("\","); - details.append("\"completeness_hint\":\"").append(escapeJson(result.getCompletenessHint())).append("\","); - } - // 域信息 - if (domain != null) { - details.append("\"retrieved_domains\":[\"").append(escapeJson(domain)).append("\"],"); - } - // 去重原因 - if (dedupReason != null) { - details.append("\"dedup_reason\":\"").append(dedupReason).append("\","); - } - // 移除末尾逗号 - if (details.charAt(details.length() - 1) == ',') { - details.setLength(details.length() - 1); - } - details.append("}"); - - ToolInvocation inv = ToolInvocation.builder() - .sessionId(sessionId) - .toolName("lookup_knowledge") - .inputParams("{\"query\":\"" + escapeJson(query) + "\"}") - .outputPreview(outputPreview) - .outputLength(outputLength) - .retrievalLayer(layer) - .l0MatchCount(hasL0 ? l0Count : null) - .l1MatchCount(hasL1 ? l1Count : null) - .isTruncated(truncated) - .retrievalDetails(details.toString()) - .relevanceLevel(result != null ? result.getRelevanceLevel() : null) - .dedupReason(dedupReason) - .durationMs((int) duration) - .success(true) - .build(); - - toolInvocationRecorder.save(inv); + ToolInvocationRecorder.LookupKnowledgeRecord record = ToolInvocationRecorder.LookupKnowledgeRecord.from( + query, + l0Matches, + l1Results, + highConfidence, + result, + domain, + dedupReason, + (int) duration, + l1TopSimilarity + ); + toolInvocationRecorder.recordLookupKnowledge(record); log.debug("tool_invocation 已保存: sessionId={}, layer={}, relevanceLevel={}, duration={}ms", - sessionId, layer, result != null ? result.getRelevanceLevel() : null, duration); + sessionId, record.retrievalLayer(), record.relevanceLevel(), duration); } catch (Exception e) { log.error("保存 tool_invocation 失败", e); } } - private String escapeJson(String s) { - if (s == null) return ""; - return s.replace("\\", "\\\\") - .replace("\"", "\\\"") - .replace("\n", "\\n") - .replace("\r", "\\r") - .replace("\t", "\\t"); - } - // ==================== 结果组装 ==================== private LookupResult buildResult( diff --git a/src/test/java/com/superbiz/agent/service/ChatServiceSequentialAgentTest.java b/src/test/java/com/superbiz/agent/service/ChatServiceSequentialAgentTest.java index a3c5dad..c880a3d 100644 --- a/src/test/java/com/superbiz/agent/service/ChatServiceSequentialAgentTest.java +++ b/src/test/java/com/superbiz/agent/service/ChatServiceSequentialAgentTest.java @@ -24,6 +24,7 @@ import java.util.Optional; import java.util.concurrent.atomic.AtomicInteger; import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertFalse; import static org.junit.jupiter.api.Assertions.assertSame; import static org.junit.jupiter.api.Assertions.assertTrue; import static org.mockito.ArgumentMatchers.any; @@ -85,6 +86,73 @@ class ChatServiceSequentialAgentTest { assertEquals(List.of("chat_planner", "chat_executor", "chat_verifier"), chatModel.agentCalls); } + @Test + void executeChatComplexFallsBackToLowConfidenceWhenVerifierOutputMissing() throws Exception { + ChatService chatService = createChatService(); + ScriptedChatModel chatModel = new ScriptedChatModel("", ""); + + ChatService.ChatResult result = chatService.executeChatComplex( + chatModel, + new ToolCallback[0], + "请分析订单支付超时的原因,并给出修复建议", + List.of(), + "sequential-missing-verifier-session" + ); + + assertTrue(result.answer().startsWith("以下结论基于当前已获取证据")); + assertEquals(List.of("chat_planner", "chat_executor", "chat_verifier", "chat_verifier"), chatModel.agentCalls); + } + + @Test + void executeChatComplexFallsBackToLowConfidenceWhenVerifierJsonInvalid() throws Exception { + ChatService chatService = createChatService(); + ScriptedChatModel chatModel = new ScriptedChatModel("not-json"); + + ChatService.ChatResult result = chatService.executeChatComplex( + chatModel, + new ToolCallback[0], + "请分析订单支付超时的原因,并给出修复建议", + List.of(), + "sequential-invalid-verifier-session" + ); + + assertTrue(result.answer().startsWith("以下结论基于当前已获取证据")); + assertEquals(List.of("chat_planner", "chat_executor", "chat_verifier"), chatModel.agentCalls); + } + + @Test + void executeChatComplexRejectOutputDoesNotLeakExecutorAnswer() throws Exception { + ChatService chatService = createChatService(); + ScriptedChatModel chatModel = new ScriptedChatModel(""" + { + "verdict": "REJECT", + "groundedness_score": 0.0, + "critical_fact_count": 1, + "facts_checked": [ + { + "fact": "payment timeout root cause", + "is_critical": true, + "verification": "contradicted", + "detail": "scripted contradiction", + "evidence_refs": [] + } + ], + "rationale": "scripted reject" + } + """); + + ChatService.ChatResult result = chatService.executeChatComplex( + chatModel, + new ToolCallback[0], + "请分析订单支付超时的原因,并给出修复建议", + List.of(), + "sequential-reject-session" + ); + + assertTrue(result.answer().startsWith("当前无法基于已获取证据生成可靠结论")); + assertFalse(result.answer().contains("EXECUTOR_FINAL_ANSWER")); + } + @Test void executeChatComplexRunsPlannerExecutorVerifierInFixedOrder() throws Exception { ChatService chatService = createChatService(); @@ -174,7 +242,8 @@ class ChatServiceSequentialAgentTest { private final java.util.ArrayList agentCalls = new java.util.ArrayList<>(); private String promptText = ""; private boolean sawVerifierPrompt; - private final String verifierOutput; + private final java.util.List verifierOutputs; + private int verifierOutputIndex; private ScriptedChatModel() { this(""" @@ -197,7 +266,11 @@ class ChatServiceSequentialAgentTest { } private ScriptedChatModel(String verifierOutput) { - this.verifierOutput = verifierOutput; + this.verifierOutputs = java.util.List.of(verifierOutput); + } + + private ScriptedChatModel(String... verifierOutputs) { + this.verifierOutputs = java.util.List.of(verifierOutputs); } @Override @@ -213,7 +286,9 @@ class ChatServiceSequentialAgentTest { } else if (promptText.contains("VERIFIER_TEST_PROMPT")) { agentCalls.add("chat_verifier"); sawVerifierPrompt = true; - text = verifierOutput; + int index = Math.min(verifierOutputIndex, verifierOutputs.size() - 1); + text = verifierOutputs.get(index); + verifierOutputIndex++; } else { text = "UNEXPECTED_PROMPT"; } diff --git a/src/test/java/com/superbiz/agent/service/ToolInvocationRecorderTest.java b/src/test/java/com/superbiz/agent/service/ToolInvocationRecorderTest.java new file mode 100644 index 0000000..1956391 --- /dev/null +++ b/src/test/java/com/superbiz/agent/service/ToolInvocationRecorderTest.java @@ -0,0 +1,96 @@ +package com.superbiz.agent.service; + +import com.fasterxml.jackson.databind.ObjectMapper; +import com.superbiz.agent.domain.entity.ToolInvocation; +import com.superbiz.agent.repository.ToolInvocationRepository; +import com.superbiz.agent.util.SessionContextHolder; +import org.junit.jupiter.api.Test; +import org.mockito.ArgumentCaptor; + +import java.util.List; +import java.util.Map; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertTrue; +import static org.mockito.ArgumentMatchers.any; +import static org.mockito.Mockito.mock; +import static org.mockito.Mockito.verify; +import static org.mockito.Mockito.when; + +class ToolInvocationRecorderTest { + + @Test + void recordEvidenceToolPreservesNoEvidenceSemantics() { + ToolInvocationRepository repository = mock(ToolInvocationRepository.class); + when(repository.save(any(ToolInvocation.class))).thenAnswer(invocation -> invocation.getArgument(0)); + ToolInvocationRecorder recorder = new ToolInvocationRecorder(repository, new ObjectMapper()); + SessionContextHolder.setSessionId("recorder-test-session"); + + try { + recorder.recordEvidenceTool( + "query_logs", + Map.of("query", "timeout"), + "{\"success\":false,\"message\":\"未找到匹配的日志\"}", + true, + System.currentTimeMillis() - 10, + null, + "application-logs", + ToolInvocationRecorder.EVIDENCE_STATUS_NO_EVIDENCE, + Map.of("log_topic", "application-logs") + ); + } finally { + SessionContextHolder.clear(); + } + + ArgumentCaptor captor = ArgumentCaptor.forClass(ToolInvocation.class); + verify(repository).save(captor.capture()); + ToolInvocation saved = captor.getValue(); + + assertEquals("query_logs", saved.getToolName()); + assertEquals(Boolean.TRUE, saved.getSuccess()); + assertTrue(saved.getRetrievalDetails().contains("\"evidence_status\":\"no_evidence\"")); + assertTrue(saved.getRetrievalDetails().contains("\"retrieved_domains\":[\"application-logs\"]")); + } + + @Test + void recordLookupKnowledgePreservesRetrievalSpecificFields() { + ToolInvocationRepository repository = mock(ToolInvocationRepository.class); + when(repository.save(any(ToolInvocation.class))).thenAnswer(invocation -> invocation.getArgument(0)); + ToolInvocationRecorder recorder = new ToolInvocationRecorder(repository, new ObjectMapper()); + SessionContextHolder.setSessionId("lookup-recorder-session"); + + ToolInvocationRecorder.LookupKnowledgeRecord record = ToolInvocationRecorder.LookupKnowledgeRecord.builder() + .query("ERR_TIMEOUT") + .outputPreview("matched payment doc") + .outputLength(18) + .retrievalLayer("L0") + .l0MatchCount(1) + .l1MatchCount(null) + .truncated(false) + .relevanceLevel("PRECISE") + .completenessHint("already precise") + .domain("payment") + .dedupReason("doc_retrieved") + .durationMs(42) + .success(true) + .evidenceStatus(ToolInvocationRecorder.EVIDENCE_STATUS_DEDUPED) + .l0Titles(List.of("payment/errors.md")) + .build(); + + try { + recorder.recordLookupKnowledge(record); + } finally { + SessionContextHolder.clear(); + } + + ArgumentCaptor captor = ArgumentCaptor.forClass(ToolInvocation.class); + verify(repository).save(captor.capture()); + ToolInvocation saved = captor.getValue(); + + assertEquals("lookup_knowledge", saved.getToolName()); + assertEquals("PRECISE", saved.getRelevanceLevel()); + assertEquals("doc_retrieved", saved.getDedupReason()); + assertTrue(saved.getRetrievalDetails().contains("\"evidence_status\":\"deduped\"")); + assertTrue(saved.getRetrievalDetails().contains("\"retrieved_domains\":[\"payment\"]")); + } +} diff --git a/src/test/java/com/superbiz/agent/service/ToolTraceSummaryServiceTest.java b/src/test/java/com/superbiz/agent/service/ToolTraceSummaryServiceTest.java new file mode 100644 index 0000000..0f158f7 --- /dev/null +++ b/src/test/java/com/superbiz/agent/service/ToolTraceSummaryServiceTest.java @@ -0,0 +1,75 @@ +package com.superbiz.agent.service; + +import com.superbiz.agent.domain.entity.ToolInvocation; +import com.superbiz.agent.repository.ToolInvocationRepository; +import org.junit.jupiter.api.Test; + +import java.util.List; +import java.util.Map; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertFalse; +import static org.junit.jupiter.api.Assertions.assertTrue; +import static org.mockito.Mockito.mock; +import static org.mockito.Mockito.when; + +class ToolTraceSummaryServiceTest { + + @Test + void buildVerifierTraceSummaryTreatsNoEvidenceAsGapWithoutLosingSuccessfulEvidence() { + ToolInvocationRepository repository = mock(ToolInvocationRepository.class); + when(repository.findBySessionIdOrderByIdAsc("session-1")).thenReturn(List.of( + ToolInvocation.builder() + .id(1L) + .sessionId("session-1") + .toolName("query_logs") + .inputParams("{\"query\":\"timeout\"}") + .outputPreview("payment timeout stack trace") + .retrievalDetails("{\"retrieved_domains\":[\"application-logs\"],\"evidence_status\":\"supported\"}") + .success(true) + .build(), + ToolInvocation.builder() + .id(2L) + .sessionId("session-1") + .toolName("query_logs") + .inputParams("{\"query\":\"timeout\"}") + .outputPreview("{\"success\":false,\"message\":\"未找到匹配的日志\"}") + .retrievalDetails("{\"retrieved_domains\":[\"application-logs\"],\"evidence_status\":\"no_evidence\"}") + .success(true) + .build(), + ToolInvocation.builder() + .id(3L) + .sessionId("session-1") + .toolName("query_metrics") + .inputParams("{\"query\":\"active_prometheus_alerts\"}") + .errorMessage("prometheus timeout") + .retrievalDetails("{\"retrieved_domains\":[\"prometheus_alerts\"],\"evidence_status\":\"failed\"}") + .success(false) + .build() + )); + + ToolTraceSummaryService service = new ToolTraceSummaryService(repository); + + List> summaries = service.buildVerifierTraceSummary("session-1", "application-logs point to timeout"); + + assertEquals(2, summaries.size()); + + Map logsSummary = summaries.stream() + .filter(item -> "query_logs".equals(item.get("tool_name"))) + .findFirst() + .orElseThrow(); + assertEquals(Boolean.TRUE, logsSummary.get("success")); + assertEquals("direct", logsSummary.get("evidence_level")); + assertEquals(2, logsSummary.get("invocation_count")); + assertEquals(1, logsSummary.get("no_hit_invocation_count")); + + Map metricsSummary = summaries.stream() + .filter(item -> "query_metrics".equals(item.get("tool_name"))) + .findFirst() + .orElseThrow(); + assertEquals(Boolean.FALSE, metricsSummary.get("success")); + assertEquals("none", metricsSummary.get("evidence_level")); + assertEquals(1, metricsSummary.get("failed_invocation_count")); + assertTrue(String.valueOf(metricsSummary.get("output_summary")).contains("call failed")); + } +} diff --git a/src/test/java/com/superbiz/agent/tool/LookupKnowledgeToolTest.java b/src/test/java/com/superbiz/agent/tool/LookupKnowledgeToolTest.java index c2b3061..61ed412 100644 --- a/src/test/java/com/superbiz/agent/tool/LookupKnowledgeToolTest.java +++ b/src/test/java/com/superbiz/agent/tool/LookupKnowledgeToolTest.java @@ -3,7 +3,9 @@ package com.superbiz.agent.tool; import com.superbiz.agent.dto.KnowledgeEntry; import com.superbiz.agent.dto.LookupResult; import com.superbiz.agent.service.KnowledgeIndexService; +import com.superbiz.agent.service.ToolInvocationRecorder; import com.superbiz.agent.service.VectorSearchService; +import com.fasterxml.jackson.databind.ObjectMapper; import org.junit.jupiter.api.BeforeEach; import org.junit.jupiter.api.Test; import org.mockito.InjectMocks; @@ -28,6 +30,15 @@ class LookupKnowledgeToolTest { @Mock private VectorSearchService vectorSearchService; + @Mock + private ToolInvocationRecorder toolInvocationRecorder; + + @Mock + private RetrievedDocTracker retrievedDocTracker; + + @Mock + private ObjectMapper objectMapper; + @InjectMocks private LookupKnowledgeTool tool;