diff --git a/devflow/index.md b/devflow/index.md index 3ed2022..b75a82e 100644 --- a/devflow/index.md +++ b/devflow/index.md @@ -5,6 +5,7 @@ | 日期 | slug | 领域 | 关键词 | 状态 | |---|---|---|---|---| | 2026-07-05 | diagnosis-playbook-skills | Agent Skill/Playbook | read_skill, diagnosis playbook, progressive disclosure, payment timeout, MySQL pool, Redis timeout | openspec/changes/diagnosis-playbook-skills | implemented | +| 2026-07-07 | executor-evidence-output-contract | Chat质量门禁/证据归因 | Executor structured output, evidence bindings, Verifier structured claims, LOW_CONFID, hallucination | openspec/changes/executor-evidence-output-contract | proposed | | 2026-07-06 | rag-eval-pipeline-closure | RAG/评测/回归闭环 | lookupResult fixture, LookupKnowledgeTool snapshot, evidenceBlocks, contextPack, retrievalTrace, rerankTrace, baseline diff, fallback case | devflow/projects/2026-07-06-rag-eval-pipeline-closure | archived | | 2026-07-06 | modular-rag-pipeline | RAG/Agent工具/证据链 | modular RAG, lookup_knowledge, evidenceBlocks, contextPack, rerank, retrievalTrace, L0 hint, unfiltered retry | openspec/changes/archive/2026-07-06-modular-rag-pipeline | archived | | 2026-07-05 | mvp-demo-interview-runbook | MVP Demo/Interview | Plan C, payment timeout, runbook, trace checklist, demo script | openspec/changes/archive/2026-07-05-mvp-demo-interview-runbook | archived | diff --git a/devflow/projects/2026-07-07-executor-evidence-output-contract/acceptance.md b/devflow/projects/2026-07-07-executor-evidence-output-contract/acceptance.md new file mode 100644 index 0000000..84218e1 --- /dev/null +++ b/devflow/projects/2026-07-07-executor-evidence-output-contract/acceptance.md @@ -0,0 +1,16 @@ +# Acceptance: executor-evidence-output-contract + +## Draft Acceptance + +- [x] Issue exists: `mvp/issues/executor-evidence-attribution-hallucination.md`. +- [x] OpenSpec change artifacts exist. +- [x] devflow tracking files exist. +- [x] OpenSpec validation passes. +- [x] Implementation updates Chat Executor prompt. +- [x] Implementation passes structured Executor output to Verifier. +- [x] Verifier prefers structured claims and still falls back safely. +- [x] Focused tests cover parsing, payload assembly, and unsupported confirmed claims. + +## Notes + +This project is currently in proposal/design stage. Runtime code is intentionally not changed yet. diff --git a/devflow/projects/2026-07-07-executor-evidence-output-contract/brief.md b/devflow/projects/2026-07-07-executor-evidence-output-contract/brief.md new file mode 100644 index 0000000..f918408 --- /dev/null +++ b/devflow/projects/2026-07-07-executor-evidence-output-contract/brief.md @@ -0,0 +1,23 @@ +# Brief: executor-evidence-output-contract + +## Summary + +Executor currently returns natural-language diagnosis answers that may mix confirmed evidence, runbook guidance, historical patterns, and unsupported inference. Verifier catches many unsupported facts, but only after extracting claims from prose. + +This project defines a structured Executor evidence-attribution contract and updates the Verifier input/verification path to consume it. + +## Goal + +Make Chat Executor output machine-checkable so confirmed claims are explicitly bound to current-session evidence, while hypotheses and evidence gaps remain visibly separate. + +## Scope + +- Chat Executor prompt contract. +- Executor structured output parsing. +- Verifier payload extension. +- Chat Verifier prompt behavior. +- Focused tests/eval fixtures. + +## Related OpenSpec + +`openspec/changes/executor-evidence-output-contract/` diff --git a/devflow/projects/2026-07-07-executor-evidence-output-contract/decisions.md b/devflow/projects/2026-07-07-executor-evidence-output-contract/decisions.md new file mode 100644 index 0000000..e34498a --- /dev/null +++ b/devflow/projects/2026-07-07-executor-evidence-output-contract/decisions.md @@ -0,0 +1,71 @@ +# Decisions: executor-evidence-output-contract + +## sm-flow Progress + +### Clarify + +Entry summary: recent Chat diagnosis sessions are `LOW_CONFID` because Executor presents unsupported or weakly supported details as confirmed facts after successful tool calls. + +Slug: `executor-evidence-output-contract` + +Scale: standard. This affects prompts, verifier input assembly, parsing behavior, and tests, but does not require a database schema change. + +### Context + +Relevant history: + +- `executor-action-memory-relevance`: Executor already has retrieval quality constraints and should avoid repeated `lookup_knowledge`. +- `chat-verifier-agent`: Verifier should not see intermediate reasoning; it receives explicit `original_query`, `executor_final_answer`, `tool_trace_summary`, and `retry_context`. +- `evidence-trace-hardening`: evidence-bearing tools persist stable traces and no-evidence semantics. +- `modular-rag-pipeline`: `lookup_knowledge` exposes evidence blocks and context packs; L0 hints are not fact evidence. + +Current code shape: + +- `src/main/resources/prompts/chat-executor-prompt.md` is the Chat Executor prompt. +- `src/main/resources/prompts/executor-prompt.md` is for the AiOps flow and is not the target of this Chat change. +- `VerifierInputHook` currently builds a payload with `original_query`, `executor_final_answer`, `tool_trace_summary`, and `retry_context`. +- Verifier prompt currently extracts facts from `executor_final_answer`. + +### Grill + +Question: Should Executor output only JSON or JSON plus readable answer? + +Decision: use one JSON object containing both machine fields and `user_facing_answer`. This avoids losing a readable Chinese answer while giving Verifier structured claims. + +Question: Should evidence binding use `chunk_id`? + +Decision: no. Use generic binding fields because `query_logs` and `query_metrics` do not naturally expose RAG chunks. + +Question: Should Verifier trust Executor-provided claims completely? + +Decision: no. Verifier should verify structured claims first, then scan `user_facing_answer` for extra confirmed-sounding facts omitted from `claims`. + +Question: What happens when Executor JSON is malformed? + +Decision: preserve raw final answer, mark parse failure, and fall back to existing natural-language verification. + +### Specify + +OpenSpec artifacts: + +- `proposal.md`: why and scope +- `design.md`: contract, verifier behavior, risks +- `specs/chat-verifier-agent/spec.md`: modified and added requirements +- `tasks.md`: implementation checklist + +### Audit + +Cross-artifact alignment: + +- Issue describes evidence attribution hallucination. +- Proposal scopes the fix to Executor output and Verifier consumption. +- Design preserves existing verifier isolation. +- Spec adds observable behavior without changing database schema. +- Tasks remain implementation-oriented and unchecked. + +Interface impact: + +- Prompt/output contract: L2 internal Agent contract change. +- Verifier payload: L2 internal structured input extension. +- Database schema: no change. +- External HTTP API: no intended change. diff --git a/devflow/projects/2026-07-07-executor-evidence-output-contract/evidence.md b/devflow/projects/2026-07-07-executor-evidence-output-contract/evidence.md new file mode 100644 index 0000000..38002a0 --- /dev/null +++ b/devflow/projects/2026-07-07-executor-evidence-output-contract/evidence.md @@ -0,0 +1,17 @@ +# Evidence: executor-evidence-output-contract + +## Repository Evidence + +- `chat-executor-prompt.md` currently requires using real tool data but does not require a structured evidence-attribution output. +- `chat-verifier-prompt.md` currently extracts facts from `executor_final_answer` prose and compares them with `tool_trace_summary`. +- `VerifierInputHook` currently provides `original_query`, `executor_final_answer`, `tool_trace_summary`, and `retry_context`. +- `openspec/specs/chat-verifier-agent/spec.md` already requires explicit verifier inputs, auditable evidence refs, fixed verdicts, and low-confidence handling. +- `openspec/specs/evidence-trace-hardening/spec.md` already distinguishes failed, no-hit, deduped, and successful evidence-tool traces. + +## Runtime Evidence From Recent Sessions + +Recent MySQL inspection showed repeated `LOW_CONFID` verifier results with many `no_evidence` facts. Typical unsupported claims included OOM, Full GC frequency, specific slow SQL timings, lock waits, and service-specific timeout details that were not supported by current-session tool traces. + +## Design Evidence + +This change preserves the previous design that Verifier should not inspect intermediate reasoning. The new structured output is still final Executor output, not hidden chain-of-thought. diff --git a/mvp/eval/schema.md b/mvp/eval/schema.md index 21e5130..5d611d3 100644 --- a/mvp/eval/schema.md +++ b/mvp/eval/schema.md @@ -58,6 +58,7 @@ trace fixture 是一次 Agent 运行后的“留痕快照”。评测器不会 | `session.answer` | Agent 最终给用户的回答 | 用来检查根因关键词和禁用词 | | `session.totalDurationMs` | 这次运行耗时 | 进入报告,帮助观察性能是否明显变差 | | `session.selfEvaluation.verifier_evaluation.verdict` | Verifier 对最终回答的判断 | 必须存在,并且要落在 case 的 `allowedVerdicts` 里 | +| `session.selfEvaluation.verifier_evaluation.executor_structured_output.claims[*].evidence_bindings` | Executor 结构化输出里的确认事实证据绑定 | 如果 trace 中存在结构化 Executor 输出,每条 confirmed claim 必须有证据绑定 | | `session.selfEvaluation.verifier_evaluation.tool_trace_summary[*].tool_name` | Verifier 总结里看到的工具证据 | 用来补充判断证据工具是否出现 | | `toolInvocations[*].toolName` | Agent 实际调用过的工具名 | 用来检查 `requiredEvidenceTools` 是否满足 | | `toolInvocations[*].success` | 工具调用是否成功 | 当前主要保留在 trace 里,后续可以升级成更严格的成功率检查 | diff --git a/openspec/changes/executor-evidence-output-contract/design.md b/openspec/changes/executor-evidence-output-contract/design.md new file mode 100644 index 0000000..5a30202 --- /dev/null +++ b/openspec/changes/executor-evidence-output-contract/design.md @@ -0,0 +1,161 @@ +## Context + +Current Chat flow explicitly runs `Planner -> Executor -> Verifier`. `VerifierInputHook` builds a structured verifier payload containing: + +- `original_query` +- `executor_final_answer` +- `tool_trace_summary` +- `retry_context` + +This satisfies the earlier design goal that Verifier should not see Planner/Executor intermediate reasoning. However, Executor's answer is still natural language. Verifier must infer claims from prose, then compare those claims to tool traces. Recent low-confidence sessions show that Executor often converts weak or unrelated evidence into confirmed conclusions before Verifier sees it. + +The fix should move evidence attribution earlier: Executor must state which claims are confirmed, which evidence supports them, and which points remain hypotheses or gaps. + +## Goals / Non-Goals + +**Goals:** + +- Make Executor final output machine-checkable. +- Separate confirmed claims, hypotheses, recommended actions, and missing information. +- Make every confirmed claim bind to concrete tool evidence identifiers or excerpts. +- Let Verifier consume structured claims directly instead of reconstructing all facts from prose. +- Preserve Verifier input isolation from intermediate reasoning. +- Preserve existing behavior when Executor output is malformed by falling back to natural-language verification and low-confidence handling. + +**Non-Goals:** + +- Do not add database tables or columns. +- Do not change evidence tool method signatures. +- Do not let Verifier call tools. +- Do not require full raw tool outputs in verifier input. +- Do not make user-facing answers become raw JSON-only unless the product layer explicitly chooses that rendering. + +## Decisions + +### Decision: Executor emits an evidence-attribution contract + +Executor final output SHALL be a single JSON object. The recommended contract is: + +```json +{ + "answer_version": "executor_evidence_v1", + "diagnosis_summary": "1-2 sentence summary using only supported facts", + "claims": [ + { + "claim_id": "claim-1", + "claim_type": "root_cause", + "claim_text": "The payment-service connection pool is saturated", + "support_level": "direct", + "evidence_bindings": [ + { + "source_type": "tool_trace", + "source_id": "trace-1", + "tool_name": "query_metrics", + "source_invocation_ids": [101], + "evidence_excerpt": "active connections reached max pool size" + } + ] + } + ], + "hypotheses": [ + { + "hypothesis_text": "A connection leak may be contributing", + "basis": "metrics show saturation, but no leak evidence was returned", + "needed_evidence": ["connection lifetime metrics", "leak detection logs"] + } + ], + "recommended_actions": [ + { + "action_text": "Check HikariCP active/idle/pending connection metrics", + "reason": "Needed to confirm pool saturation scope", + "evidence_bindings": [] + } + ], + "missing_info": [ + "No log evidence confirming a connection leak was returned" + ], + "user_facing_answer": "Chinese answer rendered from the same confirmed claims, hypotheses, actions, and gaps" +} +``` + +Rationale: this keeps the machine contract explicit while still allowing the product to return a readable Chinese answer. + +### Decision: Evidence binding uses generic source fields + +`chunk_id` alone is too specific to `lookup_knowledge`. The binding shape SHALL support all evidence-bearing tools: + +- `source_type`: `tool_trace`, `lookup_evidence_block`, or another stable source family +- `source_id`: `tool_trace_summary.trace_ref`, evidence block id, or equivalent stable id +- `tool_name`: evidence tool name when available +- `source_invocation_ids`: persisted `tool_invocation` ids when available +- `evidence_excerpt`: bounded excerpt copied or summarized from tool evidence + +Rationale: `query_logs`, `query_metrics`, and `lookup_knowledge` expose evidence differently. A generic binding prevents the prompt contract from overfitting to RAG chunks. + +### Decision: Confirmed claims are stricter than hypotheses and actions + +Confirmed `claims` SHALL contain only facts supported by current-session evidence. Runbook instructions, skill guidance, and historical case patterns SHALL NOT appear as confirmed claims unless the current tool trace supports that exact fact. + +Unsupported but useful diagnostic ideas SHALL be placed under `hypotheses` or `recommended_actions`. + +Rationale: this directly addresses the observed hallucination: turning plausible patterns into current incident facts. + +### Decision: Verifier prefers structured claims but keeps fallback + +Verifier prompt SHALL use this order: + +1. If `executor_structured_output.claims` is valid, verify each claim directly. +2. Also scan `user_facing_answer` for extra confirmed-sounding facts not present in `claims`; mark them as facts to check. +3. If the structured output is missing or invalid, fall back to the existing natural-language extraction from `executor_final_answer`. + +Invalid structured output SHALL NOT crash the flow. It SHOULD produce a fallback verifier decision and make malformed structure visible in `verifier_evaluation`. + +Rationale: the new contract should improve precision without making runtime brittle. + +### Decision: Verifier still does not see intermediate reasoning + +The verifier payload MAY add: + +- `executor_structured_output` +- `executor_output_parse_status` + +It SHALL continue to exclude Planner reasoning, Executor intermediate reasoning, and raw conversation noise. + +Rationale: this preserves the original verifier design: verify the final answer and tool traces, not hidden reasoning. + +### Decision: User-facing output remains Chinese + +For Chat user-facing pages and answers, the rendered answer SHALL be Chinese. If the Executor emits JSON, either: + +- `user_facing_answer` is returned to the user after verifier routing, or +- ChatService renders a Chinese answer from the structured contract. + +The raw machine contract may remain visible in trace/debug views, but the normal user answer should not become an English/JSON-only artifact. + +Rationale: this aligns with current UI/product language requirements while keeping machine-checkable evidence. + +## Risks / Trade-offs + +- [Risk] LLM may emit malformed JSON. Mitigation: parse-status fallback and verifier natural-language fallback. +- [Risk] Executor may put unsupported facts in `user_facing_answer` but omit them from `claims`. Mitigation: Verifier scans `user_facing_answer` for extra confirmed-sounding facts. +- [Risk] Evidence excerpts may be fabricated. Mitigation: Verifier checks binding ids against `tool_trace_summary` and treats missing refs as no evidence. +- [Risk] More verbose output increases tokens. Mitigation: keep excerpts bounded and put detailed raw evidence only in tool traces. +- [Risk] Prompt-only enforcement is soft. Mitigation: add focused tests/eval fixtures and later consider code-level schema validation. + +## Migration Plan + +1. Update `chat-executor-prompt.md` with the evidence-attribution JSON contract. +2. Add parsing in Chat runtime for Executor output: + - valid JSON -> preserve parsed `executor_structured_output` + - invalid JSON -> mark parse status and keep raw `executor_final_answer` +3. Update verifier payload assembly to include `executor_structured_output` and parse status. +4. Update `chat-verifier-prompt.md` to prefer structured claims and verify extra facts in `user_facing_answer`. +5. Persist parsed or raw structured output in existing trace/evaluation snapshots without schema changes. +6. Add focused tests and regression fixtures for unsupported confirmed claims. +7. Validate OpenSpec and run targeted tests. + +## Open Questions + +- Should `user_facing_answer` be mandatory in Executor JSON, or should ChatService render it from structured fields? +- Should malformed Executor JSON force `LOW_CONFID`, or should Verifier decide based on natural-language fallback alone? +- Should `support_level` allow only `direct`, `indirect`, and `none`, or should it also include `contradicted` for Executor self-reporting? diff --git a/openspec/changes/executor-evidence-output-contract/proposal.md b/openspec/changes/executor-evidence-output-contract/proposal.md new file mode 100644 index 0000000..c68cff8 --- /dev/null +++ b/openspec/changes/executor-evidence-output-contract/proposal.md @@ -0,0 +1,42 @@ +# Proposal: executor-evidence-output-contract + +## Why + +Recent Chat diagnosis sessions frequently end as `LOW_CONFID` even though evidence tools were called successfully. The main failure mode is not missing tool execution; it is that Executor blends tool evidence, runbook/reference patterns, and model inference into one natural-language answer, then presents unsupported details as confirmed incident facts. + +Verifier currently receives `original_query`, `executor_final_answer`, `tool_trace_summary`, and optional `retry_context`. It can catch unsupported claims, but it must first infer facts from unstructured prose. That makes the quality gate reactive and noisy: unsupported claims are detected after the answer has already been shaped as a confident story. + +## What Changes + +- Update the Chat Executor prompt so its final output follows a strict evidence-attribution JSON contract. +- Require confirmed claims to carry explicit evidence bindings and require unsupported items to be placed in `hypotheses`, `missing_info`, or `recommended_actions` instead of confirmed conclusions. +- Extend the verifier input contract with a parsed or raw `executor_structured_output` field while preserving the existing `executor_final_answer` field for fallback compatibility. +- Update the Chat Verifier prompt so it prefers Executor-provided structured claims over natural-language fact extraction. +- Keep Verifier isolated from Planner/Executor intermediate reasoning; it still receives only the original query, Executor final output, structured claim contract, tool trace summary, and retry context. +- Do not change evidence tool signatures, database schema, or the `Planner -> Executor -> Verifier` topology. + +## Capabilities + +### New Capabilities + +None. + +### Modified Capabilities + +- `chat-verifier-agent`: Chat verification now consumes an Executor evidence-attribution contract when available. + +## Impact + +- Affected prompts: `chat-executor-prompt.md`, `chat-verifier-prompt.md`. +- Affected integration: `VerifierInputHook` or equivalent verifier payload assembly must include `executor_structured_output` when Executor returns valid structured JSON. +- Affected parsing: `ChatService` may parse Executor output to separate machine contract from user-facing text, with fallback when parsing fails. +- Affected tests/eval: prompt contract tests, verifier input assembly tests, and unsupported-claim regression fixtures. +- No database schema change is required. Structured Executor output can be persisted in existing step output fields and verifier snapshots. + +## Out Of Scope + +- Loosening Verifier scoring or `PASS` criteria. +- Raising `verifier.low-confidence-threshold` to hide unsupported claims. +- Treating runbook, skill, or historical case text as current incident evidence unless it was returned as evidence for the current query and bound explicitly. +- Adding a new verifier tool or allowing Verifier to perform retrieval. +- Replacing the existing tool trace summary contract. diff --git a/openspec/changes/executor-evidence-output-contract/specs/chat-verifier-agent/spec.md b/openspec/changes/executor-evidence-output-contract/specs/chat-verifier-agent/spec.md new file mode 100644 index 0000000..94bc0dd --- /dev/null +++ b/openspec/changes/executor-evidence-output-contract/specs/chat-verifier-agent/spec.md @@ -0,0 +1,95 @@ +## MODIFIED Requirements + +### Requirement: Verifier SHALL consume explicit verification inputs +The Verifier SHALL receive explicit verification inputs rather than inferring them only from raw conversation history. + +#### Scenario: explicit input blocks available to Verifier +- **WHEN** the Verifier starts +- **THEN** the system SHALL provide `original_query`, `executor_final_answer`, and `tool_trace_summary` as explicit inputs +- **AND** `retry_context` SHALL be provided on the second round only +- **AND** when Executor returns a valid evidence-attribution contract, the system SHALL provide `executor_structured_output` +- **AND** when Executor output parsing fails, the system SHALL provide an `executor_output_parse_status` that indicates the failure +- **AND** message filtering MAY be used only to remove intermediate reasoning or unrelated noise + +#### Scenario: Verifier remains isolated from intermediate reasoning +- **WHEN** `executor_structured_output` is added to the verifier input +- **THEN** the input SHALL still exclude Planner reasoning and Executor intermediate reasoning +- **AND** the input SHALL be limited to the original query, final Executor output, parsed Executor evidence contract, tool trace summary, and retry context + +### Requirement: Verifier SHALL fact-check Executor answers +The system SHALL have a Verifier Agent that reads the Executor's answer and the tool call history, then produces a structured verdict. + +#### Scenario: Structured Executor claims are verified first +- **WHEN** `executor_structured_output.claims` is present and valid +- **THEN** Verifier SHALL verify each structured claim against `tool_trace_summary` +- **AND** each claim's evidence bindings SHALL reference existing trace or invocation identifiers when those identifiers are available +- **AND** a claim with fabricated or missing evidence references SHALL NOT be classified as `direct_evidence` + +#### Scenario: Extra confirmed-sounding answer facts are still checked +- **WHEN** `executor_structured_output.user_facing_answer` contains confirmed-sounding facts that are absent from `executor_structured_output.claims` +- **THEN** Verifier SHALL add those facts to `facts_checked` +- **AND** unsupported extra facts SHALL lower the verdict according to the existing verdict matrix + +#### Scenario: Natural-language fallback remains available +- **WHEN** Executor does not return parseable structured output +- **THEN** Verifier SHALL fall back to extracting facts from `executor_final_answer` +- **AND** the final verdict SHALL still follow the existing groundedness and evidence classification rules + +## ADDED Requirements + +### Requirement: Executor SHALL output an evidence-attribution contract +The Chat Executor SHALL produce a machine-checkable final output that separates confirmed claims from hypotheses, recommendations, and missing information. + +#### Scenario: Executor final output contains required top-level fields +- **WHEN** Executor completes a Chat diagnosis step +- **THEN** its final output SHALL contain `answer_version`, `diagnosis_summary`, `claims`, `hypotheses`, `recommended_actions`, `missing_info`, and `user_facing_answer` +- **AND** the output SHOULD be parseable as one JSON object without Markdown fences + +#### Scenario: Confirmed claims carry evidence bindings +- **WHEN** Executor emits an item under `claims` +- **THEN** the item SHALL include `claim_id`, `claim_type`, `claim_text`, `support_level`, and `evidence_bindings` +- **AND** `support_level` SHALL be one of `direct`, `indirect`, or `none` +- **AND** claims with `support_level=direct` or `support_level=indirect` SHALL include at least one evidence binding + +#### Scenario: Evidence bindings support multiple tool types +- **WHEN** Executor binds evidence to a claim +- **THEN** each binding SHALL include `source_type`, `source_id`, `tool_name`, `source_invocation_ids`, and `evidence_excerpt` +- **AND** the binding SHALL be able to reference `lookup_knowledge`, `query_logs`, `query_metrics`, or other evidence-bearing tool traces +- **AND** the binding SHALL NOT rely only on a RAG-specific `chunk_id` + +#### Scenario: Unsupported conclusions are not confirmed claims +- **WHEN** a possible root cause, detail, or remediation lacks current-session tool evidence +- **THEN** Executor SHALL place it under `hypotheses`, `recommended_actions`, or `missing_info` +- **AND** Executor SHALL NOT present it as a confirmed claim + +#### Scenario: Runbook and skill guidance do not become incident facts +- **WHEN** Executor uses runbook, skill, or historical-case guidance +- **THEN** the guidance MAY influence `recommended_actions` +- **AND** the guidance SHALL NOT be emitted as a current incident fact unless current-session tool evidence supports it + +### Requirement: User-facing Chat answers SHALL remain readable Chinese +The system SHALL preserve a readable Chinese answer for normal Chat users even when Executor emits a machine-checkable contract. + +#### Scenario: User-facing answer is available +- **WHEN** Executor emits structured output +- **THEN** `user_facing_answer` SHALL be written in Chinese +- **AND** it SHALL be consistent with the confirmed claims, hypotheses, recommended actions, and missing information in the same JSON object + +#### Scenario: Machine contract remains available for trace inspection +- **WHEN** the Chat trace or verifier evaluation is inspected +- **THEN** the structured Executor contract MAY be shown for debugging or audit +- **AND** normal user output SHALL use the existing verifier-routed display path rather than exposing raw JSON by default + +### Requirement: Structured Executor output SHALL degrade safely +The system SHALL tolerate malformed or absent structured Executor output without crashing the Chat flow. + +#### Scenario: Malformed Executor JSON is captured +- **WHEN** Executor returns malformed JSON or text outside the expected contract +- **THEN** Chat runtime SHALL preserve the raw `executor_final_answer` +- **AND** it SHALL mark `executor_output_parse_status` as failed +- **AND** Verifier SHALL use natural-language fallback behavior + +#### Scenario: Structured parse failure remains observable +- **WHEN** Executor output parsing fails +- **THEN** the verifier evaluation or trace snapshot SHALL make the parse failure visible +- **AND** the failure SHALL NOT be silently treated as a successful evidence-attribution contract diff --git a/openspec/changes/executor-evidence-output-contract/tasks.md b/openspec/changes/executor-evidence-output-contract/tasks.md new file mode 100644 index 0000000..392490d --- /dev/null +++ b/openspec/changes/executor-evidence-output-contract/tasks.md @@ -0,0 +1,31 @@ +## 1. Executor Contract + +- [x] 1.1 Update `chat-executor-prompt.md` with the evidence-attribution JSON contract. +- [x] 1.2 Require confirmed claims, hypotheses, recommended actions, missing information, and Chinese `user_facing_answer`. +- [x] 1.3 Add prompt-level constraints that runbook/skill/history guidance cannot become current incident facts without current-session evidence. + +## 2. Runtime Parsing And Verifier Input + +- [x] 2.1 Parse Executor final output into `executor_structured_output` when it is valid JSON. +- [x] 2.2 Add `executor_output_parse_status` for valid, missing, and malformed outputs. +- [x] 2.3 Pass `executor_structured_output` and parse status into the verifier payload while preserving existing `executor_final_answer`. +- [x] 2.4 Persist or expose the structured output through existing trace/evaluation snapshots without a schema change. + +## 3. Verifier Prompt + +- [x] 3.1 Update `chat-verifier-prompt.md` to verify structured claims before natural-language extraction. +- [x] 3.2 Add a verifier rule to scan `user_facing_answer` for extra confirmed-sounding facts not present in structured `claims`. +- [x] 3.3 Keep the existing natural-language fallback for malformed or absent structured output. + +## 4. Tests And Evaluation + +- [x] 4.1 Add focused tests for Executor output parsing and parse-status fallback. +- [x] 4.2 Add focused tests for verifier payload assembly with structured Executor output. +- [x] 4.3 Add prompt/fixture regression coverage for unsupported claims being placed outside confirmed `claims`. +- [x] 4.4 Add or update eval fixture checks that confirmed claims must have evidence bindings. + +## 5. Verification + +- [x] 5.1 Run targeted unit tests for the parsing and verifier-input changes. +- [x] 5.2 Run relevant Chat verifier/eval regression tests. +- [x] 5.3 Run OpenSpec validation for `executor-evidence-output-contract`. diff --git a/src/main/java/com/superbiz/agent/eval/DiagnosisTraceEvaluator.java b/src/main/java/com/superbiz/agent/eval/DiagnosisTraceEvaluator.java index 77588f5..0abd070 100644 --- a/src/main/java/com/superbiz/agent/eval/DiagnosisTraceEvaluator.java +++ b/src/main/java/com/superbiz/agent/eval/DiagnosisTraceEvaluator.java @@ -101,6 +101,8 @@ public class DiagnosisTraceEvaluator { failedChecks.add("reject output does not use degraded template"); } + failedChecks.addAll(validateExecutorStructuredOutput(trace)); + Integer toolCallCount = trace.getToolInvocations() == null ? 0 : trace.getToolInvocations().size(); Integer durationMs = trace.getSession() == null ? null : trace.getSession().getTotalDurationMs(); @@ -174,6 +176,32 @@ public class DiagnosisTraceEvaluator { return value == null ? null : String.valueOf(value); } + private List validateExecutorStructuredOutput(DiagnosisTraceResponse trace) { + Object structuredOutput = nestedValue(trace, "verifier_evaluation", "executor_structured_output"); + if (!(structuredOutput instanceof Map output)) { + return List.of(); + } + Object claims = output.get("claims"); + if (!(claims instanceof List claimList)) { + return List.of("executor structured output missing claims array"); + } + + List failedChecks = new ArrayList<>(); + for (Object item : claimList) { + if (!(item instanceof Map claim)) { + failedChecks.add("executor structured claim is not an object"); + continue; + } + Object claimIdValue = claim.get("claim_id"); + String claimId = claimIdValue == null ? "unknown" : String.valueOf(claimIdValue); + Object bindings = claim.get("evidence_bindings"); + if (!(bindings instanceof List bindingList) || bindingList.isEmpty()) { + failedChecks.add("executor confirmed claim missing evidence bindings: " + claimId); + } + } + return failedChecks; + } + private Object nestedValue(DiagnosisTraceResponse trace, String firstKey, String secondKey) { if (trace.getSession() == null || trace.getSession().getSelfEvaluation() == null) { return null; diff --git a/src/main/java/com/superbiz/agent/hook/VerifierInputHook.java b/src/main/java/com/superbiz/agent/hook/VerifierInputHook.java index 6e599be..4a9fe83 100644 --- a/src/main/java/com/superbiz/agent/hook/VerifierInputHook.java +++ b/src/main/java/com/superbiz/agent/hook/VerifierInputHook.java @@ -5,6 +5,8 @@ import com.alibaba.cloud.ai.graph.agent.hook.HookPosition; import com.alibaba.cloud.ai.graph.agent.hook.HookPositions; import com.alibaba.cloud.ai.graph.agent.hook.messages.AgentCommand; import com.alibaba.cloud.ai.graph.agent.hook.messages.MessagesModelHook; +import com.fasterxml.jackson.core.type.TypeReference; +import com.fasterxml.jackson.databind.JsonNode; import com.fasterxml.jackson.databind.ObjectMapper; import com.superbiz.agent.service.ToolTraceSummaryService; import com.superbiz.agent.util.SessionContextHolder; @@ -27,6 +29,8 @@ public class VerifierInputHook extends MessagesModelHook { private final ToolTraceSummaryService toolTraceSummaryService; private final ObjectMapper objectMapper = new ObjectMapper(); + private static final TypeReference> MAP_TYPE = new TypeReference<>() { + }; public VerifierInputHook(ToolTraceSummaryService toolTraceSummaryService) { this.toolTraceSummaryService = toolTraceSummaryService; @@ -48,6 +52,10 @@ public class VerifierInputHook extends MessagesModelHook { executorFinalAnswer = extractLastAssistantText(previousMessages); } + ExecutorOutputParseResult parseResult = parseExecutorOutput(executorFinalAnswer); + VerifierContextHolder.setExecutorStructuredOutput(parseResult.structuredOutput()); + VerifierContextHolder.setExecutorOutputParseStatus(parseResult.status()); + List> toolTraceSummary = toolTraceSummaryService.buildVerifierTraceSummary(sessionId, executorFinalAnswer); VerifierContextHolder.setToolTraceSummary(toolTraceSummary); @@ -55,6 +63,8 @@ public class VerifierInputHook extends MessagesModelHook { Map verifierInput = new LinkedHashMap<>(); verifierInput.put("original_query", VerifierContextHolder.getOriginalQuery()); verifierInput.put("executor_final_answer", executorFinalAnswer); + verifierInput.put("executor_structured_output", parseResult.structuredOutput()); + verifierInput.put("executor_output_parse_status", parseResult.status()); verifierInput.put("tool_trace_summary", toolTraceSummary); verifierInput.put("retry_context", VerifierContextHolder.getRetryContext()); @@ -66,6 +76,59 @@ public class VerifierInputHook extends MessagesModelHook { } } + private ExecutorOutputParseResult parseExecutorOutput(String executorFinalAnswer) { + if (executorFinalAnswer == null || executorFinalAnswer.isBlank()) { + return new ExecutorOutputParseResult(null, status("missing", "executor_final_answer is blank")); + } + + String sanitized = sanitizeJsonPayload(executorFinalAnswer); + if (!looksJsonLike(sanitized)) { + return new ExecutorOutputParseResult(null, status("missing", "executor output is not JSON")); + } + + try { + JsonNode root = objectMapper.readTree(sanitized); + if (!root.isObject() || !root.path("claims").isArray()) { + return new ExecutorOutputParseResult(null, status("malformed", + "executor output JSON does not match evidence-attribution contract")); + } + Map structuredOutput = objectMapper.convertValue(root, MAP_TYPE); + return new ExecutorOutputParseResult(structuredOutput, status("valid", "parsed executor evidence contract")); + } catch (Exception e) { + log.debug("Failed to parse executor structured output", e); + return new ExecutorOutputParseResult(null, status("malformed", e.getMessage())); + } + } + + private String sanitizeJsonPayload(String raw) { + String trimmed = raw.trim(); + int fenceStart = trimmed.indexOf("```"); + if (fenceStart >= 0) { + int firstNewline = trimmed.indexOf('\n', fenceStart); + int lastFence = trimmed.indexOf("```", firstNewline + 1); + if (firstNewline >= 0 && lastFence > firstNewline) { + return trimmed.substring(firstNewline + 1, lastFence).trim(); + } + } + int objectStart = trimmed.indexOf('{'); + int objectEnd = trimmed.lastIndexOf('}'); + if (objectStart >= 0 && objectEnd > objectStart) { + return trimmed.substring(objectStart, objectEnd + 1).trim(); + } + return trimmed; + } + + private boolean looksJsonLike(String text) { + return text.startsWith("{") && text.endsWith("}"); + } + + private Map status(String status, String detail) { + Map result = new LinkedHashMap<>(); + result.put("status", status); + result.put("detail", detail == null ? "" : detail); + return result; + } + private String extractLastAssistantText(List previousMessages) { for (int i = previousMessages.size() - 1; i >= 0; i--) { if (previousMessages.get(i) instanceof AssistantMessage assistantMessage) { @@ -102,4 +165,10 @@ public class VerifierInputHook extends MessagesModelHook { } return message.toString(); } + + private record ExecutorOutputParseResult( + Map structuredOutput, + Map status + ) { + } } diff --git a/src/main/java/com/superbiz/agent/service/ChatService.java b/src/main/java/com/superbiz/agent/service/ChatService.java index 83f7055..d0e2b9f 100644 --- a/src/main/java/com/superbiz/agent/service/ChatService.java +++ b/src/main/java/com/superbiz/agent/service/ChatService.java @@ -444,7 +444,8 @@ public class ChatService { } if ("PASS".equals(finalDecision.verdict())) { - answer = answer == null || answer.isBlank() ? "抱歉,多 Agent 分析未能生成有效结论。" : answer; + answer = extractUserFacingAnswer(answer) + .orElse(answer == null || answer.isBlank() ? "抱歉,多 Agent 分析未能生成有效结论。" : answer); persistVerifierEvaluation(session, finalDecision, round); break; } @@ -744,6 +745,10 @@ public class ChatService { verifierEvaluation.put("rationale", decision.rationale()); verifierEvaluation.put("round", round); verifierEvaluation.put("traceability_version", "v1"); + verifierEvaluation.put("executor_output_parse_status", + Optional.ofNullable(VerifierContextHolder.getExecutorOutputParseStatus()) + .orElse(Map.of("status", "missing", "detail", "executor parse status unavailable"))); + verifierEvaluation.put("executor_structured_output", VerifierContextHolder.getExecutorStructuredOutput()); verifierEvaluation.put("tool_trace_summary", Optional.ofNullable(VerifierContextHolder.getToolTraceSummary()).orElse(List.of())); @@ -795,6 +800,22 @@ public class ChatService { return output.toString(); } + private Optional extractUserFacingAnswer(String executorAnswer) { + if (executorAnswer == null || executorAnswer.isBlank()) { + return Optional.empty(); + } + try { + JsonNode root = objectMapper.readTree(sanitizeJsonPayload(executorAnswer)); + String userFacingAnswer = root.path("user_facing_answer").asText(""); + if (!userFacingAnswer.isBlank()) { + return Optional.of(userFacingAnswer); + } + } catch (Exception e) { + logger.debug("Executor answer is not structured JSON, keep raw answer"); + } + return Optional.empty(); + } + private String buildDegradedOutput(VerifierDecision decision) { StringBuilder output = new StringBuilder(DEGRADED_PREFIX); diff --git a/src/main/java/com/superbiz/agent/util/VerifierContextHolder.java b/src/main/java/com/superbiz/agent/util/VerifierContextHolder.java index 3bad0a4..d6fab9b 100644 --- a/src/main/java/com/superbiz/agent/util/VerifierContextHolder.java +++ b/src/main/java/com/superbiz/agent/util/VerifierContextHolder.java @@ -11,6 +11,8 @@ public final class VerifierContextHolder { private static final ThreadLocal ORIGINAL_QUERY = new ThreadLocal<>(); private static final ThreadLocal RETRY_CONTEXT = new ThreadLocal<>(); private static final ThreadLocal EXECUTOR_FINAL_ANSWER = new ThreadLocal<>(); + private static final ThreadLocal> EXECUTOR_STRUCTURED_OUTPUT = new ThreadLocal<>(); + private static final ThreadLocal> EXECUTOR_OUTPUT_PARSE_STATUS = new ThreadLocal<>(); private static final ThreadLocal>> TOOL_TRACE_SUMMARY = new ThreadLocal<>(); private VerifierContextHolder() { @@ -40,6 +42,22 @@ public final class VerifierContextHolder { return EXECUTOR_FINAL_ANSWER.get(); } + public static void setExecutorStructuredOutput(Map executorStructuredOutput) { + EXECUTOR_STRUCTURED_OUTPUT.set(executorStructuredOutput); + } + + public static Map getExecutorStructuredOutput() { + return EXECUTOR_STRUCTURED_OUTPUT.get(); + } + + public static void setExecutorOutputParseStatus(Map executorOutputParseStatus) { + EXECUTOR_OUTPUT_PARSE_STATUS.set(executorOutputParseStatus); + } + + public static Map getExecutorOutputParseStatus() { + return EXECUTOR_OUTPUT_PARSE_STATUS.get(); + } + public static void setToolTraceSummary(List> toolTraceSummary) { TOOL_TRACE_SUMMARY.set(toolTraceSummary); } @@ -52,6 +70,8 @@ public final class VerifierContextHolder { ORIGINAL_QUERY.remove(); RETRY_CONTEXT.remove(); EXECUTOR_FINAL_ANSWER.remove(); + EXECUTOR_STRUCTURED_OUTPUT.remove(); + EXECUTOR_OUTPUT_PARSE_STATUS.remove(); TOOL_TRACE_SUMMARY.remove(); } } diff --git a/src/main/resources/prompts/chat-executor-prompt.md b/src/main/resources/prompts/chat-executor-prompt.md index e30f608..703bda0 100644 --- a/src/main/resources/prompts/chat-executor-prompt.md +++ b/src/main/resources/prompts/chat-executor-prompt.md @@ -1,15 +1,17 @@ 你是任务执行器。执行 Planner 分配给你的具体步骤,并及时反馈结果。 ## 职责 -- 按步骤执行具体的查询任务 -- 需要外部信息时调用工具,但须遵守下方的检索约束 -- 不要凭记忆回答,必须基于工具返回的真实数据 -- 执行完成后,综合所有结果给出完整的答案 +- 按步骤执行具体的查询任务。 +- 需要外部信息时调用工具,但必须遵守下方的检索约束。 +- 严禁凭记忆回答,必须基于本轮工具返回的真实数据。 +- 执行完成后,输出严格的证据归因 JSON,供 Verifier 校验。 ## 规则 -- 按顺序执行,不可跳过步骤 -- 不要凭记忆回答,必须基于工具返回的真实数据 -- 执行完成后,综合所有结果给出完整的答案 +- 按顺序执行,不可跳过步骤。 +- 所有事实性结论必须来自本轮 evidence tools 的返回。 +- runbook、skill、历史案例、知识库中的通用模式只能作为排查指导或建议动作,不能直接写成本次事故的已确认事实。 +- 如果检索内容不足以支撑结论,必须显式声明证据不足,严禁补全事故故事。 +- 不要使用“通常情况下”“根据经验”“很可能已经发生”等无证据推断词来伪装事实。 ## 检索约束 @@ -20,22 +22,99 @@ ### 2. 重复了该怎么办 如果当前想检索的内容与【已检索上下文】语义相似: -- 禁止换关键词重新检索 -- 直接基于已有事实回答 +- 禁止换关键词重新检索。 +- 直接基于已有事实回答。 - 如果信息不足,先明确指出缺少什么具体维度 - (如:"缺少 HikariCP 具体配置参数"、"缺少连接池耗尽的日志样例"), - 再针对该维度进行一次定向补充检索——而非盲目换词重查 + (如:“缺少 HikariCP 具体配置参数”、“缺少连接池耗尽的日志样例”), + 再针对该维度进行一次定向补充检索,而非盲目换词重查。 -### 3. 合法出口:允许信息不全时给出结论 -如果你认为已有信息足以回答核心问题,即使细节不全, -也请直接给出结论并说明局限性(如:"基于已有信息,连接池配置建议如下, -但具体参数值需结合实际负载调整")。 -**不查全不会被追责,重复检索才会被惩罚。** +### 3. 合法出口:允许信息不全时给出有限结论 +如果已有信息足以回答核心问题,即使细节不全,也可以给出有限结论。 +但你只能把有证据支撑的内容放入 `claims`。 +缺失的细节必须写入 `missing_info`,可疑但未证实的方向必须写入 `hypotheses`。 +**不查全不会被追责,重复检索或编造细节才会被惩罚。** ### 4. 利用质量信号判断 -- relevanceLevel=PRECISE → 信息精准,直接使用,不再检索 -- relevanceLevel=HIGHLY_RELEVANT + 域已在 retrievedDomainsThisSession → 禁止再次调用 -- relevanceLevel=REFERENCE → 先指出缺什么维度,再定向补充一次 -- completenessHint 是知识库给你的天花板信号,信任它 -- lookup_knowledge 的事实证据以 evidenceBlocks 和 contextPack.packedText 为准,不要假设 L0 hint 本身就是事实证据 -- retrievalTrace 只用于理解检索路径和降级原因,不能单独作为诊断事实 +- relevanceLevel=PRECISE → 信息精准,直接使用,不再检索。 +- relevanceLevel=HIGHLY_RELEVANT + 域已在 retrievedDomainsThisSession → 禁止再次调用。 +- relevanceLevel=REFERENCE → 先指出缺什么维度,再定向补充一次。 +- completenessHint 是知识库给你的天花板信号,信任它。 +- lookup_knowledge 的事实证据以 evidenceBlocks 和 contextPack.packedText 为准,不要假设 L0 hint 本身就是事实证据。 +- retrievalTrace 只用于理解检索路径和降级原因,不能单独作为诊断事实。 + +## 证据归因要求 + +### confirmed claims +`claims` 只允许放已证实或有明确间接支撑的事实断言。 +每条 claim 必须带证据绑定。 + +支持等级: +- `direct`:工具返回中有直接事实。 +- `indirect`:工具返回可支撑方向,但没有直接陈述完整结论。 +- `none`:不能放入 `claims`,应放入 `hypotheses`、`recommended_actions` 或 `missing_info`。 + +### hypotheses +`hypotheses` 用来放合理怀疑但未被工具证实的方向。 +例如:工具只显示连接池耗尽,但没有泄漏日志,则“可能存在连接泄漏”只能是 hypothesis。 + +### recommended_actions +`recommended_actions` 用来放下一步排查或修复动作。 +建议可以来自 runbook/skill,但必须说明 reason,不能写成“已确认根因”。 + +### missing_info +`missing_info` 用来列出无法确认结论所缺少的具体证据。 + +## 最终输出格式(严格契约) + +你必须输出且只能输出一个 JSON 对象,不要输出 Markdown,不要输出代码块,不要输出 JSON 之外的解释文字。 +所有用户可读内容必须使用中文。 + +```json +{ + "answer_version": "executor_evidence_v1", + "diagnosis_summary": "1-2句话总结,仅包含有证据支撑的事实和证据边界", + "claims": [ + { + "claim_id": "claim-1", + "claim_type": "root_cause", + "claim_text": "事实断言或有限结论", + "support_level": "direct", + "evidence_bindings": [ + { + "source_type": "tool_trace", + "source_id": "工具返回中的 evidence block id、trace_ref 或可定位标识", + "tool_name": "lookup_knowledge/query_logs/query_metrics/read_skill 等", + "source_invocation_ids": [], + "evidence_excerpt": "从工具返回中摘取的原话、指标值、日志片段或关键数据" + } + ] + } + ], + "hypotheses": [ + { + "hypothesis_text": "未被证实但值得排查的方向", + "basis": "它基于哪些已知证据或为什么只是推测", + "needed_evidence": ["需要补充的证据"] + } + ], + "recommended_actions": [ + { + "action_text": "建议动作", + "reason": "为什么建议做这个动作", + "evidence_bindings": [] + } + ], + "missing_info": [ + "导致无法确认完整根因的证据缺口" + ], + "user_facing_answer": "面向用户的中文回答。必须与 claims/hypotheses/recommended_actions/missing_info 一致,不得额外加入未绑定证据的确认式事实。" +} +``` + +## 输出校验 +- `claims[*].support_level` 只能是 `direct` 或 `indirect`。 +- `claims[*].evidence_bindings` 不能为空。 +- `evidence_excerpt` 必须来自工具返回,不允许编造。 +- 如果没有任何可确认事实,`claims` 返回空数组,并在 `missing_info` 说明缺少什么。 +- `user_facing_answer` 不得出现 `claims` 中没有、且又被写成确认结论的事实。 +- 不要把其它服务、其它历史案例、其它会话的事实迁移为当前会话事实。 diff --git a/src/main/resources/prompts/chat-verifier-prompt.md b/src/main/resources/prompts/chat-verifier-prompt.md index 8a5c9cd..11891eb 100644 --- a/src/main/resources/prompts/chat-verifier-prompt.md +++ b/src/main/resources/prompts/chat-verifier-prompt.md @@ -10,6 +10,8 @@ - `original_query`:用户原始问题 - `executor_final_answer`:本轮 Executor 最终答案 +- `executor_structured_output`:如果 Executor 输出了合法证据归因 JSON,这里会提供解析后的对象。结构包含 `claims`、`hypotheses`、`recommended_actions`、`missing_info`、`user_facing_answer` +- `executor_output_parse_status`:Executor 输出解析状态,包含 `status` 和 `detail`。`status` 可能是 `valid` / `missing` / `malformed` - `tool_trace_summary`:基于真实工具调用整理出的证据索引。每一项都带有: - `trace_ref` - `tool_name` @@ -23,7 +25,18 @@ ## 任务步骤 ### 步骤一:提取关键事实 -优先提取并校验 `executor_final_answer` 里的全部实质性结论。关键事实至少包括: +如果 `executor_output_parse_status.status="valid"` 且 `executor_structured_output.claims` 存在: +- 优先逐条校验 `executor_structured_output.claims` +- 每个 claim 至少形成一条 `facts_checked` +- 必须检查 claim 的 `evidence_bindings` 是否能对应到 `tool_trace_summary` 中真实存在的 trace、tool 或 source_invocation_ids +- 如果 claim 声称 direct/indirect 支撑,但 evidence binding 不存在、无法定位、或 excerpt 与工具摘要不匹配,不得判为 `direct_evidence` + +然后必须扫描 `executor_structured_output.user_facing_answer`: +- 如果其中出现 confirmed-sounding facts(确认式事实、根因、指标值、错误码、服务名、修复结论) +- 且这些事实没有出现在 `executor_structured_output.claims` +- 必须额外加入 `facts_checked` 并按工具证据校验 + +如果 structured output 缺失或 malformed,则回退到旧逻辑:提取并校验 `executor_final_answer` 里的全部实质性结论。关键事实至少包括: - 每一个根因结论 - 每一个错误码、接口、组件归属或语义判断 - 每一个明确的修复建议、参数建议、排查步骤 @@ -49,6 +62,15 @@ - `no_evidence` - `contradicted` +结构化 claim 的校验规则: +- claim 有真实 evidence binding,且工具摘要直接包含该事实 → `direct_evidence` +- claim 有真实 evidence binding,但工具摘要只能支持方向或背景 → `indirect_support` +- claim 无法绑定真实 trace、invocation 或 excerpt → `no_evidence` +- claim 与工具摘要冲突,或编造了不存在的关键实体、服务、错误码、指标值 → `contradicted` + +`hypotheses` 和 `missing_info` 默认不是 confirmed facts,不应因为它们承认缺证据而惩罚。 +但如果 `user_facing_answer` 把 hypothesis 写成确认结论,必须按 confirmed fact 校验。 + ### 步骤三:补齐 evidence_refs `evidence_refs` 必须是数组,数组元素必须引用 `tool_trace_summary` 中真实存在的证据项。每个元素包含: - `trace_ref` diff --git a/src/test/java/com/superbiz/agent/eval/DiagnosisTraceEvaluatorTest.java b/src/test/java/com/superbiz/agent/eval/DiagnosisTraceEvaluatorTest.java index 8a0c9a8..6d71520 100644 --- a/src/test/java/com/superbiz/agent/eval/DiagnosisTraceEvaluatorTest.java +++ b/src/test/java/com/superbiz/agent/eval/DiagnosisTraceEvaluatorTest.java @@ -73,6 +73,41 @@ class DiagnosisTraceEvaluatorTest { assertTrue(result.getFailedChecks().contains("reject output does not use degraded template")); } + @Test + void evaluateFailsWhenStructuredConfirmedClaimHasNoEvidenceBindings() { + DiagnosisEvalCase evalCase = DiagnosisEvalCase.builder() + .id("structured-claim-case") + .title("Structured claim case") + .expectedRootCauseKeywords(List.of()) + .requiredEvidenceTools(List.of()) + .allowedVerdicts(List.of("LOW_CONFID", "PASS")) + .build(); + DiagnosisTraceResponse trace = DiagnosisTraceResponse.builder() + .session(DiagnosisTraceResponse.SessionTrace.builder() + .answer("以下结论基于当前已获取证据,仍存在部分证据缺口,请谨慎参考。") + .selfEvaluation(java.util.Map.of( + "verifier_evaluation", java.util.Map.of( + "verdict", "LOW_CONFID", + "executor_structured_output", java.util.Map.of( + "claims", java.util.List.of(java.util.Map.of( + "claim_id", "claim-unsupported", + "claim_text", "OOM 导致连接泄漏", + "support_level", "direct", + "evidence_bindings", java.util.List.of() + )) + ) + ))) + .build()) + .toolInvocations(List.of()) + .build(); + + DiagnosisEvalResult result = evaluator.evaluate(evalCase, trace); + + assertFalse(result.isPassed()); + assertTrue(result.getFailedChecks().contains( + "executor confirmed claim missing evidence bindings: claim-unsupported")); + } + @Test void reportWriterOutputsJsonAndMarkdown(@TempDir Path tempDir) throws Exception { DiagnosisEvalReport report = evaluator.evaluate(readCases(), Path.of("mvp/eval/fixtures")); diff --git a/src/test/java/com/superbiz/agent/hook/VerifierInputHookTest.java b/src/test/java/com/superbiz/agent/hook/VerifierInputHookTest.java new file mode 100644 index 0000000..c4b20e3 --- /dev/null +++ b/src/test/java/com/superbiz/agent/hook/VerifierInputHookTest.java @@ -0,0 +1,181 @@ +package com.superbiz.agent.hook; + +import com.alibaba.cloud.ai.graph.RunnableConfig; +import com.alibaba.cloud.ai.graph.agent.hook.messages.AgentCommand; +import com.fasterxml.jackson.databind.JsonNode; +import com.fasterxml.jackson.databind.ObjectMapper; +import com.superbiz.agent.service.ToolTraceSummaryService; +import com.superbiz.agent.util.VerifierContextHolder; +import org.junit.jupiter.api.AfterEach; +import org.junit.jupiter.api.Test; +import org.springframework.ai.chat.messages.AssistantMessage; +import org.springframework.ai.chat.messages.Message; +import org.springframework.ai.chat.messages.UserMessage; + +import java.util.List; +import java.util.Map; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertFalse; +import static org.junit.jupiter.api.Assertions.assertNotNull; +import static org.junit.jupiter.api.Assertions.assertTrue; +import static org.mockito.ArgumentMatchers.anyString; +import static org.mockito.Mockito.mock; +import static org.mockito.Mockito.when; + +class VerifierInputHookTest { + + private final ObjectMapper objectMapper = new ObjectMapper(); + + @AfterEach + void tearDown() { + VerifierContextHolder.clear(); + } + + @Test + void beforeModelAddsStructuredExecutorOutputWhenJsonContractIsValid() throws Exception { + ToolTraceSummaryService traceSummaryService = mock(ToolTraceSummaryService.class); + when(traceSummaryService.buildVerifierTraceSummary(anyString(), anyString())).thenReturn(List.of( + Map.of("trace_ref", "trace-1", "tool_name", "query_metrics") + )); + VerifierInputHook hook = new VerifierInputHook(traceSummaryService); + VerifierContextHolder.setOriginalQuery("分析 MySQL 连接池耗尽"); + + String executorOutput = """ + { + "answer_version": "executor_evidence_v1", + "diagnosis_summary": "连接池已满,但缺少泄漏证据。", + "claims": [ + { + "claim_id": "claim-1", + "claim_type": "symptom", + "claim_text": "连接池 active 达到上限", + "support_level": "direct", + "evidence_bindings": [ + { + "source_type": "tool_trace", + "source_id": "trace-1", + "tool_name": "query_metrics", + "source_invocation_ids": [101], + "evidence_excerpt": "active=50 max=50" + } + ] + } + ], + "hypotheses": [], + "recommended_actions": [], + "missing_info": ["缺少泄漏检测日志"], + "user_facing_answer": "已确认连接池 active 达到上限。" + } + """; + + AgentCommand command = hook.beforeModel( + List.of(new AssistantMessage(executorOutput)), + RunnableConfig.builder().addMetadata("sessionId", "structured-session").build() + ); + + JsonNode payload = readPayload(command); + assertEquals("valid", payload.path("executor_output_parse_status").path("status").asText()); + assertEquals("executor_evidence_v1", + payload.path("executor_structured_output").path("answer_version").asText()); + assertEquals("连接池 active 达到上限", + payload.path("executor_structured_output").path("claims").get(0).path("claim_text").asText()); + assertNotNull(VerifierContextHolder.getExecutorStructuredOutput()); + assertEquals("valid", VerifierContextHolder.getExecutorOutputParseStatus().get("status")); + } + + @Test + void beforeModelExtractsStructuredOutputFromPrefixedJsonFence() throws Exception { + ToolTraceSummaryService traceSummaryService = mock(ToolTraceSummaryService.class); + when(traceSummaryService.buildVerifierTraceSummary(anyString(), anyString())).thenReturn(List.of()); + VerifierInputHook hook = new VerifierInputHook(traceSummaryService); + + String executorOutput = """ + 现在我已经收集了足够的数据,最终输出如下。 + + ```json + { + "answer_version": "executor_evidence_v1", + "diagnosis_summary": "已确认连接池 active 达到上限。", + "claims": [ + { + "claim_id": "claim-1", + "claim_type": "symptom", + "claim_text": "连接池 active 达到上限", + "support_level": "direct", + "evidence_bindings": [ + { + "source_type": "tool_trace", + "source_id": "trace-1", + "tool_name": "query_metrics", + "source_invocation_ids": [101], + "evidence_excerpt": "active=50 max=50" + } + ] + } + ], + "hypotheses": [], + "recommended_actions": [], + "missing_info": [], + "user_facing_answer": "已确认连接池 active 达到上限。" + } + ``` + """; + + AgentCommand command = hook.beforeModel( + List.of(new AssistantMessage(executorOutput)), + RunnableConfig.builder().addMetadata("sessionId", "fenced-session").build() + ); + + JsonNode payload = readPayload(command); + assertEquals("valid", payload.path("executor_output_parse_status").path("status").asText()); + assertEquals("claim-1", + payload.path("executor_structured_output").path("claims").get(0).path("claim_id").asText()); + } + + @Test + void beforeModelMarksMalformedJsonAndKeepsRawAnswerFallback() throws Exception { + ToolTraceSummaryService traceSummaryService = mock(ToolTraceSummaryService.class); + when(traceSummaryService.buildVerifierTraceSummary(anyString(), anyString())).thenReturn(List.of()); + VerifierInputHook hook = new VerifierInputHook(traceSummaryService); + + AgentCommand command = hook.beforeModel( + List.of(new AssistantMessage("{\"diagnosis_summary\":\"缺少 claims\"}")), + RunnableConfig.builder().addMetadata("sessionId", "malformed-session").build() + ); + + JsonNode payload = readPayload(command); + assertEquals("malformed", payload.path("executor_output_parse_status").path("status").asText()); + assertTrue(payload.path("executor_structured_output").isNull()); + assertEquals("{\"diagnosis_summary\":\"缺少 claims\"}", payload.path("executor_final_answer").asText()); + assertEquals("malformed", VerifierContextHolder.getExecutorOutputParseStatus().get("status")); + } + + @Test + void beforeModelMarksPlainTextAsMissingStructuredOutput() throws Exception { + ToolTraceSummaryService traceSummaryService = mock(ToolTraceSummaryService.class); + when(traceSummaryService.buildVerifierTraceSummary(anyString(), anyString())).thenReturn(List.of()); + VerifierInputHook hook = new VerifierInputHook(traceSummaryService); + + AgentCommand command = hook.beforeModel( + List.of(new AssistantMessage("普通自然语言答案")), + RunnableConfig.builder().addMetadata("sessionId", "plain-session").build() + ); + + JsonNode payload = readPayload(command); + assertEquals("missing", payload.path("executor_output_parse_status").path("status").asText()); + assertTrue(payload.path("executor_structured_output").isNull()); + assertFalse(payload.path("executor_final_answer").asText().isBlank()); + } + + private JsonNode readPayload(AgentCommand command) throws Exception { + var field = AgentCommand.class.getDeclaredField("messages"); + field.setAccessible(true); + @SuppressWarnings("unchecked") + List messages = (List) field.get(command); + assertEquals(1, messages.size()); + Message message = messages.get(0); + assertTrue(message instanceof UserMessage); + return objectMapper.readTree(((UserMessage) message).getText()); + } +} diff --git a/src/test/java/com/superbiz/agent/service/ChatServiceSequentialAgentTest.java b/src/test/java/com/superbiz/agent/service/ChatServiceSequentialAgentTest.java index c5d25a3..2628363 100644 --- a/src/test/java/com/superbiz/agent/service/ChatServiceSequentialAgentTest.java +++ b/src/test/java/com/superbiz/agent/service/ChatServiceSequentialAgentTest.java @@ -226,6 +226,53 @@ class ChatServiceSequentialAgentTest { assertTrue(chatModel.sawVerifierPrompt); } + @Test + void verifierReceivesStructuredExecutorPayloadFields() throws Exception { + ChatService chatService = createChatService(); + ScriptedChatModel chatModel = new ScriptedChatModel(); + chatModel.executorOutput = """ + { + "answer_version": "executor_evidence_v1", + "diagnosis_summary": "已确认连接池 active 达到上限。", + "claims": [ + { + "claim_id": "claim-1", + "claim_type": "symptom", + "claim_text": "连接池 active 达到上限", + "support_level": "direct", + "evidence_bindings": [ + { + "source_type": "tool_trace", + "source_id": "trace-1", + "tool_name": "query_metrics", + "source_invocation_ids": [101], + "evidence_excerpt": "active=50 max=50" + } + ] + } + ], + "hypotheses": [], + "recommended_actions": [], + "missing_info": [], + "user_facing_answer": "已确认连接池 active 达到上限。" + } + """; + + ChatService.ChatResult result = chatService.executeChatComplex( + chatModel, + new ToolCallback[0], + "请分析 MySQL 连接池耗尽", + List.of(), + "sequential-structured-executor-session" + ); + + assertEquals("已确认连接池 active 达到上限。", result.answer()); + assertTrue(chatModel.verifierPromptText.contains("\"executor_structured_output\"")); + assertTrue(chatModel.verifierPromptText.contains("\"executor_output_parse_status\"")); + assertTrue(chatModel.verifierPromptText.contains("\"status\" : \"valid\"")); + assertTrue(chatModel.verifierPromptText.contains("连接池 active 达到上限")); + } + @Test void buildMethodToolsArrayIncludesLogsAndMetricsWhenAvailable() { ChatService chatService = new ChatService(); @@ -353,6 +400,7 @@ class ChatServiceSequentialAgentTest { private String plannerPromptText = ""; private String executorPromptText = ""; private String verifierPromptText = ""; + private String executorOutput = "EXECUTOR_FINAL_ANSWER"; private boolean sawVerifierPrompt; private final java.util.List verifierOutputs; private int verifierOutputIndex; @@ -396,7 +444,7 @@ class ChatServiceSequentialAgentTest { } else if (promptText.contains("EXECUTOR_TEST_PROMPT")) { agentCalls.add("chat_executor"); executorPromptText = promptText; - text = "EXECUTOR_FINAL_ANSWER"; + text = executorOutput; } else if (promptText.contains("VERIFIER_TEST_PROMPT")) { agentCalls.add("chat_verifier"); verifierPromptText = promptText;