diff --git a/devflow/index.md b/devflow/index.md index 9dfd555..f5bbb83 100644 --- a/devflow/index.md +++ b/devflow/index.md @@ -6,6 +6,7 @@ |---|---|---|---|---| | 2026-07-05 | diagnosis-playbook-skills | Agent Skill/Playbook | read_skill, diagnosis playbook, progressive disclosure, payment timeout, MySQL pool, Redis timeout | openspec/changes/diagnosis-playbook-skills | implemented | | 2026-07-07 | executor-evidence-output-contract | Chat质量门禁/证据归因 | Executor structured output, evidence bindings, Verifier structured claims, LOW_CONFID, hallucination | openspec/changes/archive/2026-07-07-executor-evidence-output-contract | archived | +| 2026-07-07 | executor-v2-output-contract | Chat质量门禁/证据归因 | executor_evidence_v2, user_facing_answer removal, diagnosis_summary removal, structured renderer | openspec/changes/archive/2026-07-07-executor-v2-output-contract | archived | | 2026-07-06 | rag-eval-pipeline-closure | RAG/评测/回归闭环 | lookupResult fixture, LookupKnowledgeTool snapshot, evidenceBlocks, contextPack, retrievalTrace, rerankTrace, baseline diff, fallback case | devflow/projects/2026-07-06-rag-eval-pipeline-closure | archived | | 2026-07-06 | modular-rag-pipeline | RAG/Agent工具/证据链 | modular RAG, lookup_knowledge, evidenceBlocks, contextPack, rerank, retrievalTrace, L0 hint, unfiltered retry | openspec/changes/archive/2026-07-06-modular-rag-pipeline | archived | | 2026-07-05 | mvp-demo-interview-runbook | MVP Demo/Interview | Plan C, payment timeout, runbook, trace checklist, demo script | openspec/changes/archive/2026-07-05-mvp-demo-interview-runbook | archived | diff --git a/devflow/projects/2026-07-07-executor-v2-output-contract/acceptance.md b/devflow/projects/2026-07-07-executor-v2-output-contract/acceptance.md new file mode 100644 index 0000000..0f030ad --- /dev/null +++ b/devflow/projects/2026-07-07-executor-v2-output-contract/acceptance.md @@ -0,0 +1,41 @@ +# Acceptance: executor-v2-output-contract + +## Implementation Result + +Completed stage one of Executor Structured Output V2. + +- Executor prompt now emits `executor_evidence_v2`. +- Executor output no longer includes `diagnosis_summary` or `user_facing_answer`. +- ChatService PASS path renders V2 structured output into readable Chinese. +- VerifierInputHook remains parse-only and accepts V2 output without final-expression fields. + +## Static Verification + +- `cmd /c openspec validate executor-v2-output-contract` + - Result: passed. + - Coverage: OpenSpec syntax and change validity. + +## Script Verification + +- `mvn "-Dtest=VerifierInputHookTest,ChatServiceSequentialAgentTest" test` + - Result: passed. + - Coverage: VerifierInputHook V2 parsing; ChatService PASS rendering for V2; existing sequential workflow tests. + +## Browser / Manual Verification + +Not run. This stage changes backend prompt/runtime contract and unit-level behavior only. + +## Unverified + +- Full live application run with a real LLM. +- MySQL trace inspection after a real chat session. + +Reason: Stage one is covered by focused unit tests; live verification is more useful after Gatekeeper and Composer phases. + +## Remaining Work + +- Phase two: Gatekeeper in `VerifierInputHook`. +- Phase three: Verifier V2 `claim_checks`. +- Phase four: Composer. +- Phase five: eval fixtures and full audit closure. + diff --git a/devflow/projects/2026-07-07-executor-v2-output-contract/brief.md b/devflow/projects/2026-07-07-executor-v2-output-contract/brief.md new file mode 100644 index 0000000..48aa751 --- /dev/null +++ b/devflow/projects/2026-07-07-executor-v2-output-contract/brief.md @@ -0,0 +1,32 @@ +# Brief: executor-v2-output-contract + +## Background + +`executor_evidence_v1` still made Chat Executor produce both evidence attribution and final user-facing prose through `diagnosis_summary` and `user_facing_answer`. + +This kept Executor in a "diagnose and narrate" role and left room for unsupported conclusions to appear before later verification and composition stages. + +## Goal + +Narrow Chat Executor output to `executor_evidence_v2`: structured diagnostic material only, with final expression removed from Executor. + +## Scope + +- Update Chat Executor prompt to emit `executor_evidence_v2`. +- Remove `diagnosis_summary` and `user_facing_answer` from Executor output. +- Keep `claims`, `hypotheses`, `recommended_actions`, `missing_info`, and evidence bindings. +- Add temporary ChatService rendering for PASS + V2 output so normal users do not see raw JSON. +- Preserve V1 `user_facing_answer` extraction for compatibility. + +## Non-Goals + +- No Gatekeeper implementation. +- No Verifier V2 `claim_checks`. +- No Composer. +- No Planner changes. +- No database schema changes. +- No evidence tool signature changes. + +## Related OpenSpec + +`openspec/changes/archive/2026-07-07-executor-v2-output-contract/` diff --git a/devflow/projects/2026-07-07-executor-v2-output-contract/decisions.md b/devflow/projects/2026-07-07-executor-v2-output-contract/decisions.md new file mode 100644 index 0000000..d105eba --- /dev/null +++ b/devflow/projects/2026-07-07-executor-v2-output-contract/decisions.md @@ -0,0 +1,39 @@ +# Decisions: executor-v2-output-contract + +## Key Decisions + +### Executor V2 removes final-expression fields + +Decision: Chat Executor final output now uses `executor_evidence_v2` and must not include `diagnosis_summary` or `user_facing_answer`. + +Reason: Executor should collect evidence and produce structured diagnostic material, not write final user-facing conclusions. + +### Temporary renderer bridges the gap before Composer + +Decision: `ChatService` renders V2 structured fields into readable Chinese only when Verifier returns `PASS`. + +Reason: Composer is a later phase, but external users must not receive raw JSON during this intermediate stage. + +### V1 compatibility remains + +Decision: Existing V1 `user_facing_answer` extraction remains. + +Reason: It keeps old tests and any lingering V1 output compatible while the staged migration continues. + +### Gatekeeper and Verifier V2 are deferred + +Decision: This phase does not add Gatekeeper or `claim_checks`. + +Reason: The user requested phase-by-phase implementation with archive and commit after each phase. Gatekeeper is phase two. + +## Interface Impact + +- Internal Agent output contract: L4, because fields are removed. +- Verifier payload: L2, because raw `executor_final_answer` and parsed `executor_structured_output` remain. +- External Chat answer: compatible intent; users still get readable Chinese. + +## Risks + +- The temporary renderer is not a full Composer and should be replaced in the Composer phase. +- Verifier prompt still uses V1 `facts_checked`; Verifier V2 is a later phase. + diff --git a/devflow/projects/2026-07-07-executor-v2-output-contract/evidence.md b/devflow/projects/2026-07-07-executor-v2-output-contract/evidence.md new file mode 100644 index 0000000..8b14339 --- /dev/null +++ b/devflow/projects/2026-07-07-executor-v2-output-contract/evidence.md @@ -0,0 +1,21 @@ +# Evidence: executor-v2-output-contract + +## Context Used + +- `devflow/projects/2026-07-07-executor-evidence-output-contract`: V1 evidence-attribution contract kept `user_facing_answer`. +- `devflow/projects/2026-07-02-chat-verifier-agent`: Verifier consumes explicit inputs and should not see intermediate reasoning. +- `devflow/projects/2026-07-04-evidence-trace-hardening`: evidence summaries and tool invocation references are the evidence foundation. +- `mvp/issues/executor-structured-output-v2.md`: staged implementation design; stage one is Executor V2 output contract. + +## Code Evidence + +- `src/main/resources/prompts/chat-executor-prompt.md`: V2 contract now uses `answer_version="executor_evidence_v2"` and removes final-expression fields. +- `src/main/java/com/superbiz/agent/service/ChatService.java`: PASS path now tries V1 `user_facing_answer`, then renders V2 structured output to readable Chinese. +- `src/main/resources/prompts/chat-verifier-prompt.md`: `user_facing_answer` is now described as compatibility-only. +- `src/test/java/com/superbiz/agent/hook/VerifierInputHookTest.java`: V2 structured output without `user_facing_answer` parses successfully. +- `src/test/java/com/superbiz/agent/service/ChatServiceSequentialAgentTest.java`: PASS + V2 output renders Chinese and does not expose raw JSON. + +## Key Finding + +The previous V1 contract intentionally kept `user_facing_answer`, but the V2 staged design intentionally removes it. This is an internal Agent contract break, mitigated by a temporary renderer until Composer is implemented. + diff --git a/mvp/architecture/executor-evidence-pipeline-refactor.md b/mvp/architecture/executor-evidence-pipeline-refactor.md new file mode 100644 index 0000000..d34f5ac --- /dev/null +++ b/mvp/architecture/executor-evidence-pipeline-refactor.md @@ -0,0 +1,514 @@ +# Current Chat Agent Data Contracts + +**状态**:当前实现 +**日期**:2026-07-07 +**范围**:当前 Chat 复杂诊断链路的数据结构定义 + +当前代码实现是三 Agent 顺序链路: + +```text +chat_planner -> chat_executor -> chat_verifier +``` + +对应 `ChatService.executeChatComplex(...)` 中的 `SequentialAgent`。 + +--- + +## 1. Workflow Input + +由 `ChatService.buildWorkflowInput(...)` 构造,传给 `chat_workflow`。 + +```text +请按固定工作流完成本轮 Planner -> Executor -> Verifier。 + +--- 用户问题 --- +{question} + +--- retry_context --- +{retry_context} + +Verifier 完成后由外层代码读取 verifier_output 并决定最终用户输出。 +``` + +| 字段 | 来源 | 定义 | +|---|---|---| +| `question` | 用户输入 | 用户本轮原始问题 | +| `retry_context` | ChatService | 第二轮补证据约束;首轮为空 | + +--- + +## 2. chat_planner + +### 2.1 Input + +`chat_planner` 的输入来自 workflow input 和 system prompt 追加上下文。 + +```json +{ + "question": "用户原始问题", + "history": [], + "available_knowledge_domains": "...", + "skill_catalog": {}, + "retry_context": null +} +``` + +| 字段 | 来源 | 定义 | +|---|---|---| +| `question` | workflow input | 用户原始问题 | +| `history` | `ChatService.buildChatPlannerAgent(...)` | 对话历史,拼接到 planner system prompt | +| `available_knowledge_domains` | `KnowledgeDomainService.buildKnowledgeMap()` | 可用知识域地图,拼接到 planner system prompt | +| `skill_catalog` | `PlannerSkillMetadataHook` | Planner 可见的 skill name/description 元数据 | +| `retry_context` | `ChatService` | Verifier 低置信后构造的补证据上下文 | + +### 2.2 Output:`planner_plan` + +当前 prompt 要求输出 JSON: + +```json +{ + "selected_skill": "匹配的 skill 名称;如果没有匹配则为 null", + "selection_reason": "选择该 skill 的原因;如果没有匹配则说明不使用 skill", + "plan": ["步骤1描述", "步骤2描述", "步骤3描述"], + "reasoning": "规划思路说明" +} +``` + +| 字段 | 类型 | 定义 | +|---|---|---| +| `selected_skill` | string/null | Planner 选择的诊断 skill 名称 | +| `selection_reason` | string | skill 选择理由 | +| `plan` | array | 给 Executor 的执行步骤 | +| `reasoning` | string | 规划思路说明 | + +运行态输出 key: + +```text +planner_plan +``` + +--- + +## 3. chat_executor + +### 3.1 Input + +`chat_executor` 接收前序 `planner_plan`,并通过 system prompt 获得历史、skill 读取约束、retry 约束和工具权限。 + +```json +{ + "planner_plan": {}, + "history": [], + "retry_context": null, + "tool_permissions": { + "method_tools": ["dateTimeTools", "lookupKnowledgeTool", "queryMetricsTools", "queryLogsTools"], + "tool_callbacks": [] + } +} +``` + +| 字段 | 来源 | 定义 | +|---|---|---| +| `planner_plan` | `chat_planner` | Planner 输出的计划 | +| `history` | `ChatService.buildChatExecutorAgent(...)` | 对话历史,拼接到 executor system prompt | +| `retry_context` | `ChatService` | 本轮补证据约束 | +| `method_tools` | `ChatService.buildMethodToolsArray()` | Executor 可直接调用的本地工具 | +| `tool_callbacks` | `ToolCallback[]` | 框架发现或外部注入工具 | +| `read_skill` | `SkillsAgentHook` | 当存在 skillRegistry 时,Executor 可读取 Planner 选中的 skill | + +### 3.2 Output:`executor_feedback` + +当前 `chat-executor-prompt.md` 要求输出一个 JSON 对象,即 `executor_evidence_v1`。 + +```json +{ + "answer_version": "executor_evidence_v1", + "diagnosis_summary": "1-2句话总结,仅包含有证据支撑的事实和证据边界", + "claims": [ + { + "claim_id": "claim-1", + "claim_type": "root_cause", + "claim_text": "事实断言或有限结论", + "support_level": "direct", + "evidence_bindings": [ + { + "source_type": "tool_trace", + "source_id": "工具返回中的 evidence block id、trace_ref 或可定位标识", + "tool_name": "lookup_knowledge/query_logs/query_metrics/read_skill 等", + "source_invocation_ids": [], + "evidence_excerpt": "从工具返回中摘取的原话、指标值、日志片段或关键数据" + } + ] + } + ], + "hypotheses": [ + { + "hypothesis_text": "未被证实但值得排查的方向", + "basis": "它基于哪些已知证据或为什么只是推测", + "needed_evidence": ["需要补充的证据"] + } + ], + "recommended_actions": [ + { + "action_text": "建议动作", + "reason": "为什么建议做这个动作", + "evidence_bindings": [] + } + ], + "missing_info": [ + "导致无法确认完整根因的证据缺口" + ], + "user_facing_answer": "面向用户的中文回答。必须与 claims/hypotheses/recommended_actions/missing_info 一致。" +} +``` + +| 字段 | 类型 | 定义 | +|---|---|---| +| `answer_version` | string | 当前固定为 `executor_evidence_v1` | +| `diagnosis_summary` | string | 有证据边界的简短诊断摘要 | +| `claims` | array | 已证实或有明确间接支撑的事实断言 | +| `claims[].claim_id` | string | claim 标识 | +| `claims[].claim_type` | string | claim 类型,例如 `root_cause`、`symptom`、`impact` | +| `claims[].claim_text` | string | 事实断言文本 | +| `claims[].support_level` | string | `direct` 或 `indirect` | +| `claims[].evidence_bindings` | array | 支撑 claim 的证据绑定,不能为空 | +| `evidence_bindings[].source_type` | string | 证据来源类型,例如 `tool_trace` | +| `evidence_bindings[].source_id` | string | evidence block id、trace_ref 或其它定位标识 | +| `evidence_bindings[].tool_name` | string | 来源工具名 | +| `evidence_bindings[].source_invocation_ids` | array | 来源 `tool_invocation.id` | +| `evidence_bindings[].evidence_excerpt` | string | 工具返回中的原话、指标值、日志片段或关键数据 | +| `hypotheses` | array | 未证实但值得排查的方向 | +| `hypotheses[].hypothesis_text` | string | 假设文本 | +| `hypotheses[].basis` | string | 假设依据和未证实原因 | +| `hypotheses[].needed_evidence` | array | 确认该假设还需要的证据 | +| `recommended_actions` | array | 建议动作 | +| `recommended_actions[].action_text` | string | 建议动作文本 | +| `recommended_actions[].reason` | string | 建议原因 | +| `recommended_actions[].evidence_bindings` | array | 建议动作关联证据,可为空 | +| `missing_info` | array | 证据缺口 | +| `user_facing_answer` | string | 候选用户答案,PASS 时由 ChatService 提取输出 | + +运行态输出 key: + +```text +executor_feedback +``` + +--- + +## 4. chat_verifier + +### 4.1 Input + +`VerifierInputHook` 会在 Verifier 调用前替换消息历史,构造显式 JSON payload。 + +```json +{ + "original_query": "用户原始问题", + "executor_final_answer": "{...executor_feedback raw text...}", + "executor_structured_output": {}, + "executor_output_parse_status": { + "status": "valid", + "detail": "parsed executor evidence contract" + }, + "tool_trace_summary": [], + "retry_context": null +} +``` + +| 字段 | 来源 | 定义 | +|---|---|---| +| `original_query` | `VerifierContextHolder` | 用户原始问题 | +| `executor_final_answer` | `VerifierContextHolder` 或上一条 AssistantMessage | Executor 原始输出文本 | +| `executor_structured_output` | `VerifierInputHook.parseExecutorOutput(...)` | Executor 输出可解析且包含 `claims` 时的 JSON 对象;否则为 null | +| `executor_output_parse_status.status` | `VerifierInputHook` | `valid` / `missing` / `malformed` | +| `executor_output_parse_status.detail` | `VerifierInputHook` | 解析状态说明 | +| `tool_trace_summary` | `ToolTraceSummaryService.buildVerifierTraceSummary(...)` | 基于真实 `tool_invocation` 构建的证据索引 | +| `retry_context` | `VerifierContextHolder` | 当前补证据上下文 | + +### 4.2 `tool_trace_summary` + +`ToolTraceSummaryService` 聚合 evidence tools: + +```text +lookup_knowledge, query_logs, query_metrics, query_order +``` + +输出项结构: + +```json +{ + "trace_ref": "trace-1", + "tool_name": "query_logs", + "success": true, + "input_summary": "query=payment-service timeout", + "output_summary": "log_evidence: ...", + "evidence_level": "direct", + "topic_domain": "general", + "source_invocation_ids": [394], + "invocation_count": 1, + "failed_invocation_count": 0, + "no_hit_invocation_count": 0, + "query_samples": ["payment-service timeout"], + "retrieval_layers": [], + "relevance_levels": [], + "source_documents": [] +} +``` + +| 字段 | 类型 | 定义 | +|---|---|---| +| `trace_ref` | string | Verifier 可引用的证据摘要编号 | +| `tool_name` | string | 聚合后的工具名 | +| `success` | boolean | 是否存在可用证据 | +| `input_summary` | string | 工具输入摘要 | +| `output_summary` | string | 工具输出摘要 | +| `evidence_level` | string | `direct` / `indirect` / `none` | +| `topic_domain` | string | 主题域,优先来自 `retrieval_details.retrieved_domains` | +| `source_invocation_ids` | array | 聚合的 `tool_invocation.id` | +| `invocation_count` | number | 聚合调用次数 | +| `failed_invocation_count` | number | 失败调用次数 | +| `no_hit_invocation_count` | number | 无证据或去重调用次数 | +| `query_samples` | array | 查询样例 | +| `retrieval_layers` | array | 检索层级 | +| `relevance_levels` | array | 相关性等级 | +| `source_documents` | array | 来源文档标签 | + +### 4.3 Output:`verifier_output` + +当前 `chat-verifier-prompt.md` 要求输出: + +```json +{ + "verdict": "PASS", + "groundedness_score": 0.8, + "critical_fact_count": 2, + "facts_checked": [ + { + "fact": "ERR_TIMEOUT 表示请求超时", + "is_critical": true, + "verification": "direct_evidence", + "detail": "知识库文档明确给出该错误码定义", + "evidence_refs": [ + { + "trace_ref": "trace-1", + "tool_name": "lookup_knowledge", + "topic_domain": "api", + "source_invocation_ids": [101, 104], + "note": "trace-1 的文档摘要直接给出错误码定义" + } + ] + } + ], + "rationale": "所有关键事实均有支撑,且至少一条具有直接证据" +} +``` + +| 字段 | 类型 | 定义 | +|---|---|---| +| `verdict` | string | `PASS` / `LOW_CONFID` / `REJECT` | +| `groundedness_score` | number | 关键事实证据支撑评分 | +| `critical_fact_count` | number | `facts_checked` 中 `is_critical=true` 的数量 | +| `facts_checked` | array | Verifier 校验过的事实列表 | +| `facts_checked[].fact` | string | 被校验事实 | +| `facts_checked[].is_critical` | boolean | 是否关键事实 | +| `facts_checked[].verification` | string | `direct_evidence` / `indirect_support` / `no_evidence` / `contradicted` | +| `facts_checked[].detail` | string | 校验说明 | +| `facts_checked[].evidence_refs` | array | 证据引用 | +| `evidence_refs[].trace_ref` | string | 引用的 `tool_trace_summary.trace_ref` | +| `evidence_refs[].tool_name` | string | 引用工具 | +| `evidence_refs[].topic_domain` | string | 引用主题域 | +| `evidence_refs[].source_invocation_ids` | array | 引用的 `tool_invocation.id` | +| `evidence_refs[].note` | string | 引用说明 | +| `rationale` | string | verdict 判定理由 | + +运行态输出 key: + +```text +verifier_output +``` + +--- + +## 5. VerifierDecision + +`ChatService.parseVerifierDecision(...)` 将 `verifier_output` 解析为内部 record: + +```json +{ + "verdict": "LOW_CONFID", + "groundednessScore": 0.5, + "criticalFactCount": 2, + "factsChecked": [], + "rationale": "证据不足", + "round": 1 +} +``` + +| 字段 | 类型 | 定义 | +|---|---|---| +| `verdict` | string | Verifier verdict | +| `groundednessScore` | number | groundedness score | +| `criticalFactCount` | number | 关键事实数量 | +| `factsChecked` | array | 解析后的 facts_checked | +| `rationale` | string | 判定理由 | +| `round` | number | 当前验证轮次 | + +--- + +## 6. retry_context + +当 `LOW_CONFID` 且满足重试条件时,`ChatService.buildRetryContext(...)` 构造: + +```json +{ + "round": 1, + "missing_evidence_facts": [ + "某关键事实:缺少直接证据" + ], + "instruction": "仅补充以上断言相关证据,不要重复已完成检索" +} +``` + +| 字段 | 类型 | 定义 | +|---|---|---| +| `round` | number | 触发 retry 的轮次 | +| `missing_evidence_facts` | array | 来自 Verifier 的证据缺口 | +| `instruction` | string | 补证据约束 | + +--- + +## 7. diagnosis_session.self_evaluation.verifier_evaluation + +`ChatService.persistVerifierEvaluation(...)` 将 Verifier 结果合并进 `diagnosis_session.self_evaluation`。 + +```json +{ + "verifier_evaluation": { + "verdict": "LOW_CONFID", + "groundedness_score": 0.5, + "critical_fact_count": 2, + "facts_checked": [], + "rationale": "证据不足", + "round": 1, + "traceability_version": "v1", + "executor_output_parse_status": { + "status": "valid", + "detail": "parsed executor evidence contract" + }, + "executor_structured_output": {}, + "tool_trace_summary": [] + } +} +``` + +| 字段 | 类型 | 定义 | +|---|---|---| +| `verifier_evaluation.verdict` | string | Verifier verdict | +| `verifier_evaluation.groundedness_score` | number | groundedness score | +| `verifier_evaluation.critical_fact_count` | number | 关键事实数量 | +| `verifier_evaluation.facts_checked` | array | 校验事实列表 | +| `verifier_evaluation.rationale` | string | 判定理由 | +| `verifier_evaluation.round` | number | 验证轮次 | +| `verifier_evaluation.traceability_version` | string | 当前固定为 `v1` | +| `verifier_evaluation.executor_output_parse_status` | object | Executor 输出解析状态 | +| `verifier_evaluation.executor_structured_output` | object/null | 解析后的 Executor 结构化输出 | +| `verifier_evaluation.tool_trace_summary` | array | Verifier 使用的工具证据索引 | + +--- + +## 8. Final Answer Rendering + +ChatService 根据 Verifier verdict 决定最终 `diagnosis_session.answer`。 + +| Verdict | 当前行为 | +|---|---| +| `PASS` | 优先提取 `executor_feedback.user_facing_answer`;提取失败则使用 executor 原文 | +| `LOW_CONFID` | 输出低置信模板:已确认信息、当前缺口、建议下一步 | +| `REJECT` | 输出降级模板:已确认信息、证据缺口、建议下一步 | + +低置信模板使用: + +```text +以下结论基于当前已获取证据,仍存在部分证据缺口,请谨慎参考。 + +已确认信息: +- ... + +当前缺口: +- ... + +建议下一步: +- ... +``` + +拒绝模板使用: + +```text +当前无法基于已获取证据生成可靠结论。 + +已确认信息: +- ... + +证据缺口: +- ... + +建议下一步: +- ... +``` + +--- + +## 9. Trace Persistence Data + +### 9.1 diagnosis_session + +| 字段 | 类型 | 定义 | +|---|---|---| +| `session_id` | string | 会话 id | +| `query` | text | 用户问题 | +| `status` | string | 会话状态 | +| `agent_flow` | string | 当前 Chat 链路为 `CHAT` | +| `total_duration_ms` | number | 总耗时 | +| `total_token_count` | number | 总 token | +| `step_count` | number | agent step 数 | +| `tool_call_count` | number | tool invocation 数 | +| `answer` | longtext | 最终用户答案 | +| `self_evaluation` | json | 包含 verifier_evaluation | +| `feedback` | string | 用户反馈 | + +### 9.2 agent_step + +| 字段 | 类型 | 定义 | +|---|---|---| +| `session_id` | string | 会话 id | +| `step_index` | number | 步骤序号 | +| `agent_name` | string | `planner` / `executor` / `verifier` | +| `model_input` | text | 模型输入摘要 | +| `model_output` | text | 模型输出摘要 | +| `thought` | text | hook 记录的摘要信息 | +| `has_tool_call` | boolean | 是否包含工具调用 | +| `duration_ms` | number | 模型调用耗时 | +| `token_count` | number | token 数 | + +### 9.3 tool_invocation + +| 字段 | 类型 | 定义 | +|---|---|---| +| `id` | number | 工具调用 id | +| `session_id` | string | 会话 id | +| `step_id` | number | 对应 agent_step id | +| `tool_name` | string | 工具名 | +| `input_params` | json | 工具输入参数 | +| `output_preview` | text | 工具输出预览 | +| `output_length` | number | 原始输出长度 | +| `retrieval_layer` | string | 检索层 | +| `l0_match_count` | number | L0 命中数 | +| `l1_match_count` | number | L1 命中数 | +| `is_truncated` | boolean | 输出是否截断 | +| `relevance_level` | string | 相关性等级 | +| `dedup_reason` | string | 去重原因 | +| `retrieval_details` | json | 检索细节 | +| `duration_ms` | number | 工具耗时 | +| `success` | boolean | 是否成功 | +| `error_message` | text | 错误信息 | diff --git a/mvp/issues/executor-structured-output-v2.md b/mvp/issues/executor-structured-output-v2.md new file mode 100644 index 0000000..746fb28 --- /dev/null +++ b/mvp/issues/executor-structured-output-v2.md @@ -0,0 +1,1497 @@ +# Executor Structured Output V2 实施设计 + +**状态**:待规划 +**严重程度**:高 +**日期**:2026-07-07 +**文档类型**:实施设计文档 +**范围**:Chat Executor 输出结构、Gatekeeper、Verifier、Composer 数据契约调整 +**关联问题**: + +- `executor-evidence-attribution-hallucination` +- `executor-self-evidence-loop-design-note` +- `chat-verifier-agent` +- `executor-evidence-output-contract` + +--- + +## 0. 给实现 Agent 的阅读入口 + +这份文档用于指导 `Chat` 复杂诊断链路的一次增量改造。实现时不要先从字段表开始,而应按以下顺序阅读: + +1. 先读 `0.1 - 0.7`,理解为什么改、当前链路是什么、目标链路是什么、哪些地方不能改。 +2. 再读 `20 - 25`,理解代码影响面、实施阶段、回滚策略和完成标准。 +3. 最后按需查阅 `2 - 19` 的详细数据契约、Gatekeeper 规则、Verifier/Composer 输入输出。 + +一句话目标: + +```text +把 Executor 从“诊断 + 表达”收敛为“证据收集 + 结构化事实输出”, +在 Verifier 前加 Gatekeeper 做确定性拦截, +让 Verifier 只判断 claim 是否能由证据合理推出, +最终用户答案交给 Composer 生成。 +``` + +--- + +## 0.1 背景 + +当前 Chat 复杂诊断链路中,Executor 的输出既包含结构化证据归因,也包含最终面向用户的自然语言答案: + +```text +diagnosis_summary +user_facing_answer +``` + +这导致一个核心问题:Executor 在证据还不充分时,容易提前把“可能方向”写成“确认结论”。后续 Verifier 虽然会校验,但它面对的是自然语言答案和结构化字段混在一起的输出,容易出现以下风险: + +- Executor 编造不存在的 `source_invocation_ids`。 +- Executor 使用真实 invocation id,但 `evidence_excerpt` 与真实工具输出不一致。 +- Executor 把 hypothesis 写成 confirmed claim。 +- Executor 在 `user_facing_answer` 里夹带 claims 中没有的根因、错误码、指标值或修复建议。 +- Verifier 被迫从自然语言里逐字抽事实,校验边界不稳定。 + +因此,本次改造不是为了增加 Agent 数量,而是为了收紧职责边界和证据链路。 + +--- + +## 0.2 当前实现 + +当前代码中的复杂 Chat 链路是三 Agent 顺序执行: + +```text +chat_planner -> chat_executor -> chat_verifier +``` + +对应入口在: + +```text +ChatService.executeChatComplex(...) +``` + +当前数据流: + +```text +用户问题 + -> Planner 生成 planner_plan + -> Executor 调用工具并输出 executor_evidence_v1 + -> VerifierInputHook 构造 verifier payload + -> Verifier 输出 facts_checked/verdict + -> ChatService 根据 verdict 生成最终 answer +``` + +当前关键问题在 Executor 与 Verifier 之间: + +```text +Executor 同时输出结构化 claims 和 user_facing_answer +Verifier 既要校验 structured output,又要扫描自然语言答案 +``` + +这会让最终答案和证据归因之间出现缝隙。 + +--- + +## 0.3 目标设计 + +目标链路仍保持单条顺序链路,不新增 Controller,不改 Planner: + +```text +chat_planner + -> chat_executor + -> VerifierInputHook 内 Gatekeeper + -> chat_verifier + -> chat_composer + -> final answer +``` + +职责边界: + +| 组件 | 职责 | +|---|---| +| Planner | 选择 skill、拆解计划;本次不改 | +| Executor | 调用工具收集证据,输出结构化 claims/hypotheses/actions/missing_info | +| Gatekeeper | 在 Verifier 前做确定性校验,拦截伪造 ID、工具名不匹配、明显张冠李戴、非法 schema | +| Verifier | 判断 claims 是否能由证据合理推出,不再逐字扫描最终答案 | +| Composer | 只根据 Verifier 允许的材料生成最终用户答案 | + +目标数据流: + +```text +Executor raw JSON + -> parse JSON + -> Gatekeeper.validate(...) + -> Verifier claim-level validation + -> ChatService 过滤 allowed_claims/allowed_hypotheses + -> Composer 生成 user_facing_answer +``` + +--- + +## 0.4 分阶段实施总览 + +| 阶段 | 目标 | 主要改动 | 验收重点 | +|---|---|---|---| +| 阶段一 | Executor V2 输出契约 | 改 `chat-executor-prompt.md`,移除 `diagnosis_summary` / `user_facing_answer` | Executor 只输出结构化诊断材料 | +| 阶段二 | Gatekeeper 接入 | 在 `VerifierInputHook` 中加入 Gatekeeper,写入 payload 和审计 | 伪造 id、工具名不匹配、schema 退化可被拦截 | +| 阶段三 | Verifier V2 | 改 `chat-verifier-prompt.md`,从 `facts_checked` 转向 `claim_checks` | Verifier 判断可推导性,不再逐字抽自然语言事实 | +| 阶段四 | Composer | 新增 Composer prompt/调用,由 ChatService 过滤输入 | 最终答案只使用 Verifier 允许的材料 | +| 阶段五 | 回归与审计 | 扩测试和 eval fixtures | 证明幻觉拦截、生效路径、审计回溯都可验证 | + +更详细的实施拆解见 `21. 实施阶段`。 + +--- + +## 0.5 非目标 + +本 issue 第一版明确不做以下事情: + +- 不改 Planner 的输入输出。 +- 不引入 Controller 或新的编排层。 +- 不改变现有 retry 机制。 +- 不要求 Gatekeeper 失败后自动回退 Executor 重试。 +- 不新增数据库表。 +- 不让 Gatekeeper 做根因判断。 +- 不让 Composer 调用工具或重新诊断。 + +--- + +## 0.6 关键设计决策 + +| 决策 | 结论 | 原因 | +|---|---|---| +| 是否改 Planner | 不改 | 当前问题主要在 Executor 输出和 Verifier 校验边界 | +| Gatekeeper 放在哪里 | 放在 `VerifierInputHook` | 保持当前 workflow,不改 Executor hook 和编排 | +| Gatekeeper 失败是否重试 | 第一版不重试 | 先建立拦截和审计闭环,避免扩大改动面 | +| Executor 是否继续输出最终答案 | 不输出 | 防止未经 Verifier 的自然语言结论泄露 | +| Verifier 是否逐字扫描答案 | 不扫描 | 主校验对象改为 `claims` | +| Composer 是否可以新增事实 | 不可以 | Composer 只是表达层,不是诊断层 | +| `facts_checked` 是否保留 | 兼容期保留 | 当前低置信模板、retry_context、评测可能依赖 | + +--- + +## 0.7 实现约束 + +实现时必须遵守: + +- 优先保持现有三 Agent 复杂链路可运行。 +- 每个阶段都应能单独测试和回滚。 +- 第一版以“降低幻觉风险”为目标,不追求一次性重构所有历史字段。 +- 所有最终用户可见答案都必须来自 Verifier 允许输出的材料。 +- 所有 Gatekeeper 校验结果都必须进入审计,至少能回溯到 `self_evaluation.verifier_evaluation.gatekeeper_result`。 +- 如果结构化输出 malformed,不允许回退到自然语言逐字抽取后给 PASS。 + +--- + +## 0.8 推荐实现切片 + +建议不要把所有改动塞进一个大 PR。推荐拆成以下可独立验收的切片: + +| 切片 | 内容 | 可单独合入条件 | +|---|---|---| +| PR-1 | Executor V2 prompt + parse-only 边界调整 | Executor 能输出 V2 JSON;不会再通过 Executor 直接生成最终用户答案 | +| PR-2 | Gatekeeper 基础框架 + schema/invocation 规则 | Verifier payload 和审计中出现 `gatekeeper_result`;伪造 invocation id 可被拦截 | +| PR-3 | Verifier V2 claim_checks + facts_checked 兼容 | Verifier 主校验 `claims`;现有低置信模板和 retry_context 不坏 | +| PR-4 | Composer 接入 + 最终答案渲染 | PASS/LOW_CONFID/REJECT 最终答案都不再读取 Executor `user_facing_answer` | +| PR-5 | eval fixtures + 回归测试补齐 | 覆盖伪造 ID、工具名不匹配、excerpt 张冠李戴、hypothesis 写成事实 | + +每个切片都应保留当前复杂 Chat 链路可运行。若某个切片失败,优先回滚该切片,不要连带回滚已经稳定的前置切片。 + +--- + +## 1. 设计目标 + +当前 `executor_evidence_v1` 同时包含结构化诊断材料和自然语言表达字段: + +- `diagnosis_summary` +- `user_facing_answer` + +这两个字段容易让 Executor 提前进入“总结报告 / 用户表达”模式,并可能把未证实内容写成确认结论。 + +V2 的目标是让 Executor 只输出结构化诊断材料,不负责最终用户表达。 + +--- + +## 2. 输出结构 + +```json +{ + "answer_version": "executor_evidence_v2", + "claims": [ + { + "claim_id": "claim-1", + "claim_type": "symptom", + "claim_text": "payment-service 出现请求超时日志。", + "support_level": "direct", + "evidence_bindings": [ + { + "source_type": "tool_trace", + "source_id": "trace-1", + "tool_name": "query_logs", + "source_invocation_ids": [394], + "evidence_excerpt": "request timeout" + } + ] + } + ], + "hypotheses": [ + { + "hypothesis_text": "连接池压力可能参与了超时问题。", + "basis": "当前已有超时现象,但缺少连接池 active/idle/pending 指标。", + "needed_evidence": ["HikariCP active/idle/pending 指标"] + } + ], + "recommended_actions": [ + { + "action_text": "补充查询连接池 active/idle/pending 指标。", + "reason": "用于确认连接池是否达到上限。", + "evidence_bindings": [] + } + ], + "missing_info": [ + "缺少连接池指标,无法确认连接池耗尽是根因。" + ] +} +``` + +--- + +## 3. 字段定义 + +| 字段 | 类型 | 必填 | 定义 | +|---|---|---:|---| +| `answer_version` | string | 是 | 固定为 `executor_evidence_v2` | +| `claims` | array | 是 | 已确认或有明确支撑的事实断言 | +| `hypotheses` | array | 否 | 合理怀疑但未被当前证据确认的方向 | +| `recommended_actions` | array | 否 | 建议补证据、排查或处理动作 | +| `missing_info` | array | 否 | 当前无法确认结论所缺少的证据 | + +## 4. claims + +```json +{ + "claim_id": "claim-1", + "claim_type": "symptom", + "claim_text": "payment-service 出现请求超时日志。", + "support_level": "direct", + "evidence_bindings": [] +} +``` + +| 字段 | 类型 | 必填 | 定义 | +|---|---|---:|---| +| `claim_id` | string | 是 | claim 唯一标识 | +| `claim_type` | string | 是 | claim 类型,例如 `symptom` / `root_cause` / `impact` / `risk` | +| `claim_text` | string | 是 | 已确认或有限确认的事实断言 | +| `support_level` | string | 是 | `direct` / `indirect` | +| `evidence_bindings` | array | 是 | 支撑该 claim 的证据绑定,不能为空 | + +## 5. evidence_bindings + +```json +{ + "source_type": "tool_trace", + "source_id": "trace-1", + "tool_name": "query_logs", + "source_invocation_ids": [394], + "evidence_excerpt": "request timeout" +} +``` + +| 字段 | 类型 | 必填 | 定义 | +|---|---|---:|---| +| `source_type` | string | 是 | 证据来源类型,例如 `tool_trace` | +| `source_id` | string | 否 | evidence block id、trace_ref 或可定位标识;第一版不作为硬校验依据 | +| `tool_name` | string | 是 | 来源工具名 | +| `source_invocation_ids` | array | 是 | 来源 `tool_invocation.id` | +| `evidence_excerpt` | string | 是 | 从工具返回中摘取的原话、指标值、日志片段或关键数据 | + +## 6. hypotheses + +```json +{ + "hypothesis_text": "连接池压力可能参与了超时问题。", + "basis": "当前已有超时现象,但缺少连接池 active/idle/pending 指标。", + "needed_evidence": ["HikariCP active/idle/pending 指标"] +} +``` + +| 字段 | 类型 | 必填 | 定义 | +|---|---|---:|---| +| `hypothesis_text` | string | 是 | 未证实但值得排查的方向 | +| `basis` | string | 是 | 它基于哪些已知证据,以及为什么仍只是推测 | +| `needed_evidence` | array | 是 | 要确认该假设还缺少的证据 | + +## 7. recommended_actions + +```json +{ + "action_text": "补充查询连接池 active/idle/pending 指标。", + "reason": "用于确认连接池是否达到上限。", + "evidence_bindings": [] +} +``` + +| 字段 | 类型 | 必填 | 定义 | +|---|---|---:|---| +| `action_text` | string | 是 | 建议动作 | +| `reason` | string | 是 | 建议原因 | +| `evidence_bindings` | array | 否 | 建议动作关联的证据绑定,可为空 | + +## 8. 移除字段 + +V2 移除以下字段: + +| 字段 | 移除原因 | +|---|---| +| `diagnosis_summary` | 容易让 Executor 提前总结诊断故事 | +| `user_facing_answer` | 容易夹带 claims 之外的确认式事实 | + +最终用户答案应由 ChatService 或后续专门渲染阶段基于结构化字段生成。 + +--- + +## 9. Gatekeeper 设计 + +Executor 输出后、Verifier 接收前,增加一层确定性校验: + +```text +Executor raw output + -> parse JSON + -> Gatekeeper + -> Verifier payload +``` + +Gatekeeper 的目标是拦截物理级幻觉,而不是替代 Verifier 做语义判断。 + +### 9.1 职责边界 + +| 组件 | 职责 | +|---|---| +| Gatekeeper | schema、引用 ID、工具名、excerpt、黑名单措辞等确定性校验 | +| Verifier | 判断 claim 是否被证据语义支撑、是否说重、是否缺关键证据 | + +Gatekeeper 不判断根因是否正确,也不做新的检索。 + +### 9.2 接入位置 + +V1 接入位置保持在 `VerifierInputHook` 内,不改 Executor hook,不改 Chat workflow 编排: + +```text +parseExecutorOutput(...) + -> buildVerifierTraceSummary(...) + -> Gatekeeper.validate(...) + -> verifierInput.put("gatekeeper_result", ...) +``` + +`VerifierInputHook` 的职责收敛为: + +- 解析 Executor raw output。 +- 构造 `executor_output_parse_status`。 +- 构造 `tool_trace_summary`。 +- 调用 Gatekeeper。 +- 组装 Verifier payload。 + +除 JSON parse 与最小可解析性判断外,`VerifierInputHook` 不再维护独立校验规则。 + +后续持久化时,将 `gatekeeper_result` 写入: + +```text +diagnosis_session.self_evaluation.verifier_evaluation.gatekeeper_result +``` + +--- + +## 10. Gatekeeper 规则加载 + +Gatekeeper 采用“规则代码稳定、规则配置可调”的设计。 + +```text +Rule Index + -> Rule Metadata + -> Rule Implementation +``` + +### 10.1 索引层 + +索引层列出所有可用规则名和一句话描述,用于加载、展示和审计。 + +```yaml +rules: + - id: schema.executor_v2 + name: Executor V2 Schema + description: 校验 Executor 输出是否符合 V2 schema + enabled: true + + - id: evidence.invocation_ref + name: Invocation Reference Check + description: 校验 source_invocation_ids 是否属于当前 session 且工具名匹配 + enabled: true + + - id: evidence.excerpt_similarity + name: Excerpt Similarity Check + description: 校验 evidence_excerpt 与真实工具输出是否相似 + enabled: true + + - id: claim.hallucination_phrases + name: Hallucination Phrase Check + description: 禁止 confirmed claims 使用经验化、推测化措辞 + enabled: true + + - id: claim.evidence_utilization + name: Evidence Utilization Check + description: 检查 claim_text 与 evidence_excerpt 的关键词覆盖率 + enabled: false +``` + +### 10.2 元数据层 + +元数据层定义规则参数、阈值、严重级别和适用字段。调整阈值或黑名单时只改元数据,不改代码。 + +```yaml +ruleMetadata: + schema.executor_v2: + severity: error + verdictImpact: low_confid_on_fail + params: + requiredFields: + - answer_version + - claims + deprecatedFields: + - diagnosis_summary + - user_facing_answer + + evidence.excerpt_similarity: + severity: error + verdictImpact: low_confid_on_fail + params: + passThreshold: 0.5 + warnThreshold: 0.3 + compareAgainst: + - output_preview + - retrieval_details + - tool_trace_summary + + claim.hallucination_phrases: + severity: error + verdictImpact: low_confid_on_fail + params: + targetFields: + - claims[].claim_text + blacklist: + - 通常情况下 + - 根据经验 + - 一般来说 + - 我认为 + - 理论上 + - 可能是 + - 推测 + allowInFields: + - hypotheses[].hypothesis_text + - hypotheses[].basis + + claim.evidence_utilization: + severity: warning + verdictImpact: warn_only + params: + minKeywordCoverage: 0.5 + + evidence.invocation_ref: + severity: error + verdictImpact: reject_on_fail +``` + +### 10.3 规则实现层 + +规则实现使用代码中的固定接口。新增全新类型规则需要代码;调整阈值、词表、启用状态不需要改代码。 + +```text +GatekeeperRule + id() + validate(context, metadata) + -> RuleResult +``` + +### 10.4 与 VerifierInputHook 原有校验的关系 + +V2 中,`VerifierInputHook` 原有的结构判断只保留 parse-only 能力: + +```text +raw text + -> sanitized JSON + -> JsonNode / Map + -> parse status +``` + +以下校验统一迁移到 Gatekeeper: + +- schema 字段是否完整。 +- `answer_version` 是否正确。 +- `claims` 是否为空或类型错误。 +- `claims[].evidence_bindings` 是否为空。 +- `source_invocation_ids` 是否存在。 +- `tool_name` 是否匹配。 +- `evidence_excerpt` 是否可信。 +- 是否出现已移除字段 `diagnosis_summary` / `user_facing_answer`。 + +这样避免 `VerifierInputHook` 和 Gatekeeper 出现两套规则、两套错误口径。 + +--- + +## 11. 推荐规则集 + +### 11.1 Schema 校验 + +规则 id: + +```text +schema.executor_v2 +``` + +校验内容: + +- 输出必须是 JSON object。 +- `answer_version` 必须为 `executor_evidence_v2`。 +- `claims` 必须存在且为 array。 +- `hypotheses`、`recommended_actions`、`missing_info` 缺省时按空数组处理。 +- 不允许出现 `diagnosis_summary`。 +- 不允许出现 `user_facing_answer`。 + +Gatekeeper 输出给 Verifier 前应 normalize 结构: + +```text +missing hypotheses -> [] +missing recommended_actions -> [] +missing missing_info -> [] +``` + +也就是说,Verifier 和 Composer 可以假定这三个字段存在且为数组。 + +### 11.2 Invocation 引用校验 + +规则 id: + +```text +evidence.invocation_ref +``` + +校验内容: + +- `claims[].evidence_bindings` 不能为空。 +- `source_invocation_ids` 必须为非空数组。 +- 每个 invocation id 必须属于当前 session。 +- `tool_name` 必须和对应 `tool_invocation.tool_name` 匹配。 +- 不允许引用其它 session 的工具调用。 + +该规则只检查 `claims[].evidence_bindings`。`recommended_actions[].evidence_bindings` 可以为空,不参与硬失败判定。 + +### 11.3 Excerpt 相似度校验 + +规则 id: + +```text +evidence.excerpt_similarity +``` + +校验内容: + +- 将 `evidence_excerpt` 与真实工具输出进行相似度比对。 +- 可比较来源包括: + - `tool_invocation.output_preview` + - `tool_invocation.retrieval_details` + - `tool_trace_summary.output_summary` + +建议阈值: + +| 相似度 | 结果 | +|---:|---| +| `>= 0.5` | pass | +| `0.3 - 0.5` | warn | +| `< 0.3` | fail | + +如果只是 excerpt 相似度低,第一版判为 `LOW_CONFID` 级风险,不直接 `REJECT`,因为 `output_preview` 可能被截断,`tool_trace_summary` 也可能是摘要。 + +只有以下情况才进入 `REJECT` 级别: + +- invocation id 不存在。 +- invocation id 不属于当前 session。 +- `tool_name` 与 invocation 的真实工具名不匹配。 +- invocation 存在但 excerpt 明显指向另一个服务、错误码或指标值。 + +### 11.4 幻觉措辞校验 + +规则 id: + +```text +claim.hallucination_phrases +``` + +校验内容: + +- `claims[].claim_text` 不允许出现经验化、推测化措辞。 +- `hypotheses` 中允许出现推测措辞。 + +默认黑名单: + +```json +[ + "通常情况下", + "根据经验", + "一般来说", + "可能是", + "推测", + "我认为", + "理论上" +] +``` + +### 11.5 证据利用率校验 + +规则 id: + +```text +claim.evidence_utilization +``` + +校验内容: + +- 提取 `claim_text` 中的实词。 +- 提取 `evidence_excerpt` 中的实词。 +- 计算关键词覆盖率。 + +伪代码: + +```python +def check_evidence_utilization(claim_text, excerpt): + claim_keywords = extract_keywords(claim_text) + excerpt_keywords = extract_keywords(excerpt) + overlap = claim_keywords & excerpt_keywords + utilization = len(overlap) / len(claim_keywords) + return utilization >= 0.5 +``` + +第一版建议作为 warning,不作为硬拒绝条件。 + +--- + +## 12. Gatekeeper 输出 + +Gatekeeper 第一版只保留必要审计字段。 + +```json +{ + "gatekeeper_result": { + "status": "fail", + "failed_rules": ["evidence.invocation_ref"], + "warnings": ["evidence.excerpt_similarity"], + "errors": [ + { + "rule_id": "evidence.invocation_ref", + "target": "claims[0].evidence_bindings[0]", + "message": "source_invocation_ids not found in current session" + } + ] + } +} +``` + +### 12.1 字段定义 + +| 字段 | 类型 | 定义 | +|---|---|---| +| `status` | string | `pass` / `warn` / `fail` | +| `failed_rules` | array | 失败规则 id 列表 | +| `warnings` | array | warning 规则 id 列表 | +| `errors` | array | 最小错误明细 | +| `errors[].rule_id` | string | 失败规则 id | +| `errors[].target` | string | 出问题的字段路径 | +| `errors[].message` | string | 人类可读原因 | + +### 12.2 总状态计算 + +```text +任一规则 fail -> gatekeeper status = fail +否则任一规则 warn -> gatekeeper status = warn +否则 gatekeeper status = pass +``` + +第一版不记录 `version`、`rule_set`、`summary`、`severity`、`details`、`checked_at` 等字段,避免审计结构过重。 + +Verifier 如需判断 Gatekeeper fail 是否应导致 `REJECT`,使用规则元数据中的 `verdictImpact`,不依赖落库字段。 + +--- + +## 13. 审计落点 + +Gatekeeper 结果需要进入两处: + +### 13.1 Verifier payload + +```json +{ + "executor_structured_output": {}, + "executor_output_parse_status": {}, + "tool_trace_summary": [], + "gatekeeper_result": {} +} +``` + +Verifier 可根据 `gatekeeper_result` 调整判定,尤其是: + +- `gatekeeper_result.status=fail` 时,不应输出 `PASS`。 +- schema 或 invocation 引用失败时,应倾向 `LOW_CONFID` 或 `REJECT`。 + +### 13.1.1 parse status 与 schema status 边界 + +`executor_output_parse_status` 只表示 JSON 解析结果: + +| parse status | 含义 | +|---|---| +| `valid` | raw output 可解析为 JSON object | +| `missing` | raw output 为空或不是 JSON | +| `malformed` | raw output 像 JSON 但解析失败 | + +schema 合法性不再由 parse status 表达,统一交给 Gatekeeper: + +```text +JSON parse valid but schema invalid + -> executor_output_parse_status.status = valid + -> gatekeeper_result.status = fail + -> failed_rules contains schema.executor_v2 +``` + +### 13.2 self_evaluation + +持久化位置: + +```text +diagnosis_session.self_evaluation.verifier_evaluation.gatekeeper_result +``` + +审计链路: + +```text +executor_raw_output + -> executor_output_parse_status + -> gatekeeper_result + -> verifier_evaluation + -> final_answer +``` + +--- + +## 14. Verifier 输入结构 + +V2 中 Verifier 以结构化输出为主输入,不再逐字抽取自然语言事实。 + +```json +{ + "original_query": "用户原始问题", + "executor_raw_output": "{raw executor output, debug/fallback only}", + "executor_structured_output": { + "answer_version": "executor_evidence_v2", + "claims": [], + "hypotheses": [], + "recommended_actions": [], + "missing_info": [] + }, + "executor_output_parse_status": { + "status": "valid", + "detail": "parsed executor evidence contract" + }, + "tool_trace_summary": [], + "gatekeeper_result": { + "status": "pass", + "failed_rules": [], + "warnings": [], + "errors": [] + }, + "retry_context": null +} +``` + +| 字段 | 用途 | +|---|---| +| `original_query` | 判断结构化诊断是否围绕用户问题 | +| `executor_raw_output` | 原始输出,仅用于 debug/fallback,不作为主校验来源 | +| `executor_structured_output` | Verifier 主校验对象 | +| `executor_output_parse_status` | 判断结构化输出是否可用 | +| `tool_trace_summary` | 证据索引 | +| `gatekeeper_result` | Gatekeeper 物理级校验结果 | +| `retry_context` | 第二轮时判断缺口是否被补足 | + +兼容期可以保留旧字段名: + +```text +executor_final_answer +``` + +第一期 payload 可以同时包含 `executor_raw_output` 和 `executor_final_answer`,二者内容相同。`executor_final_answer` 语义上视为 deprecated;Verifier 不应在结构化输出合法时从该字段抽取额外事实。 + +--- + +## 15. Verifier 字段使用规则 + +### 15.1 executor_structured_output + +Verifier 主要校验: + +- `claims` +- `hypotheses` +- `recommended_actions` +- `missing_info` + +重点是 `claims`: + +```text +claim_text 是否能由 evidence_bindings 指向的证据合理推出? +support_level 是否说重? +是否引入证据外的新实体、新指标、新错误码或新根因? +``` + +### 15.2 tool_trace_summary + +`tool_trace_summary` 用于判断 evidence binding 是否有语义支撑。 + +Verifier 不要求 claim 与证据逐字一致,而是判断可推导性: + +```text +证据:payment-service CPU 92%,线程数 245 +claim:payment-service 当前存在 CPU 使用率过高现象 +=> direct_observation + +claim:CPU 过高可能导致支付接口超时 +=> reasonable_inference + +claim:CPU 过高是支付超时的唯一根因 +=> overstated +``` + +### 15.3 gatekeeper_result + +Gatekeeper 负责物理合法性,Verifier 消费结果: + +```text +status = pass + -> 正常校验 claims + +status = warn + -> 正常校验 claims,但 rationale 可提及 warning + +status = fail + -> 不允许 PASS + -> 相关 claim 至少应判 unsupported / external_unknown / contradicted +``` + +### 15.4 executor_output_parse_status + +```text +status = valid + -> 使用结构化可推导性校验 + +status = missing / malformed + -> 不再回退到自然语言逐字抽取 + -> verdict = LOW_CONFID + -> groundedness_score = 0 +``` + +### 15.5 executor_raw_output + +`executor_raw_output` 只用于审计和 debug。 + +当 `executor_structured_output` 合法时,Verifier 不应从 `executor_raw_output` 中抽取额外事实。 + +--- + +## 16. Verifier 输出结构 + +V2 Verifier 输出从 `facts_checked` 转向 `claim_checks`。 + +```json +{ + "verdict": "LOW_CONFID", + "groundedness_score": 0.62, + "claim_checks": [ + { + "claim_id": "claim-1", + "verification": "direct_observation", + "detail": "query_metrics 显示 CPU 使用率为 92%,可直接支撑 CPU 过高现象。", + "evidence_refs": [ + { + "trace_ref": "trace-1", + "tool_name": "query_metrics", + "source_invocation_ids": [394], + "note": "指标摘要包含 CPU=92%" + } + ] + } + ], + "hypothesis_checks": [ + { + "hypothesis_index": 0, + "verification": "reasonable_hypothesis", + "detail": "该假设明确标注为未证实,并列出需要补充的证据。" + } + ], + "rationale": "已有证据支持超时现象,但根因仍缺少直接证据。" +} +``` + +兼容期建议同时保留 `facts_checked`: + +- `claim_checks` 作为 V2 主字段。 +- `facts_checked` 由 `claim_checks` 映射生成,供现有 Trace Workbench、eval 和降级输出继续使用。 +- 待前端和评测全部迁移后,再移除 `facts_checked`。 + +### 16.1 字段定义 + +| 字段 | 类型 | 定义 | +|---|---|---| +| `verdict` | string | `PASS` / `LOW_CONFID` / `REJECT` | +| `groundedness_score` | number | 结构化 claim 的证据支撑评分 | +| `claim_checks` | array | 对 `claims` 的可推导性校验结果 | +| `claim_checks[].claim_id` | string | 被校验的 claim id | +| `claim_checks[].verification` | string | 可推导性判定 | +| `claim_checks[].detail` | string | 判定说明 | +| `claim_checks[].evidence_refs` | array | 使用的证据引用 | +| `hypothesis_checks` | array | 对 hypotheses 的边界校验 | +| `hypothesis_checks[].hypothesis_index` | number | hypothesis 数组下标 | +| `hypothesis_checks[].verification` | string | `reasonable_hypothesis` / `unsupported_hypothesis` / `overstated_as_fact` | +| `hypothesis_checks[].detail` | string | 判定说明 | +| `rationale` | string | 总体判定理由 | + +### 16.1.1 facts_checked 兼容映射 + +兼容期内,ChatService 需要从 `claim_checks` 生成旧字段 `facts_checked`。 + +映射规则: + +| `claim_checks[].verification` | `facts_checked[].verification` | +|---|---| +| `direct_observation` | `direct_evidence` | +| `reasonable_inference` | `indirect_support` | +| `overstated` | `indirect_support` | +| `unsupported` | `no_evidence` | +| `external_unknown` | `no_evidence` | +| `contradicted` | `contradicted` | + +`facts_checked[].fact` 建议格式: + +```text +{claim_id}: {claim_text} +``` + +`facts_checked[].is_critical` 规则: + +```text +claim_type in ["root_cause", "symptom", "impact", "risk"] -> true +其它 -> false +``` + +`facts_checked[].evidence_refs` 直接复用 `claim_checks[].evidence_refs`。 + +### 16.2 claim verification 枚举 + +| verification | 含义 | +|---|---| +| `direct_observation` | 证据直接观测到该事实 | +| `reasonable_inference` | 证据没有逐字说明,但可以合理推出 | +| `overstated` | 有部分依据,但 claim 说得太满 | +| `unsupported` | 证据不足 | +| `external_unknown` | 引入证据外的新服务、数值、错误码、根因等 | +| `contradicted` | 与证据冲突 | + +`external_unknown` 需要区分严重程度: + +| 场景 | 建议 verdict | +|---|---| +| 引入证据外的核心服务名、订单号、错误码、关键指标值、根因 | `REJECT` | +| 引入非核心背景实体或表达不清 | `LOW_CONFID` | + +--- + +## 17. Verifier 判定矩阵 + +### PASS + +必须同时满足: + +- `gatekeeper_result.status != fail` +- 所有核心 claims 均为 `direct_observation` 或 `reasonable_inference` +- 至少一个关键 claim 为 `direct_observation` +- 不存在 `overstated` +- 不存在 `unsupported` +- 不存在 `external_unknown` +- 不存在 `contradicted` + +### LOW_CONFID + +满足任一条件: + +- 存在 `unsupported` +- 存在 `external_unknown` +- 存在 `overstated` +- 所有核心 claims 都只是 `reasonable_inference` +- `executor_output_parse_status.status` 为 `missing` 或 `malformed` +- `gatekeeper_result.status = fail`,但失败不属于严重伪造 + +### REJECT + +满足任一条件: + +- 任一核心 claim 为 `contradicted` +- Gatekeeper 发现严重伪造: + - 编造 `source_invocation_ids` + - 跨 session 引用 + - `tool_name` 与 invocation 不匹配 + - 使用真实 id 但 excerpt 明显张冠李戴 +- Verifier 发现核心 `external_unknown`: + - 新增证据外服务名 + - 新增证据外订单号 + - 新增证据外错误码 + - 新增证据外关键指标值 + - 新增证据外根因 + +严重伪造由 Gatekeeper 规则元数据中的 `verdictImpact=reject_on_fail` 定义。 + +--- + +## 18. 与 V1 Verifier 的差异 + +| 项目 | V1 | V2 | +|---|---|---| +| 主校验对象 | `executor_final_answer` 和 structured output | `executor_structured_output` | +| 校验方式 | 从自然语言中抽取 facts 并逐条校验 | 对 claims 做可推导性校验 | +| 原始文本作用 | 事实抽取来源之一 | debug/fallback only | +| 输出字段 | `facts_checked` | `claim_checks` | +| Gatekeeper | 无 | 前置物理级校验 | +| malformed 输出 | 可回退自然语言校验 | 直接 LOW_CONFID | + +--- + +## 19. Composer 设计 + +Composer 是最终表达层。它不负责诊断,不调用工具,不补充新事实。 + +```text +Executor structured output + -> Gatekeeper + -> Verifier + -> Composer + -> final answer +``` + +### 19.1 职责边界 + +| 组件 | 职责 | +|---|---| +| Executor | 输出结构化诊断材料 | +| Gatekeeper | 做物理级确定性校验 | +| Verifier | 判断 claims 是否可由证据合理推出 | +| Composer | 将 Verifier 允许输出的材料组织成用户可读中文答案 | + +Composer 禁止: + +- 调用工具。 +- 重新诊断。 +- 重新判断根因。 +- 新增服务名、订单号、时间、指标值、错误码、根因。 +- 把 hypothesis 写成 confirmed claim。 +- 把相关性写成因果。 + +### 19.2 Composer 输入 + +Composer 输入应由 ChatService 根据 Executor、Gatekeeper、Verifier 的结果过滤得到。 + +```json +{ + "original_query": "用户原始问题", + "verdict": "LOW_CONFID", + "allowed_claims": [ + { + "claim_id": "claim-1", + "claim_type": "symptom", + "claim_text": "订单123支付失败期间出现 ERR_TIMEOUT,请求耗时 5.3 秒。" + } + ], + "allowed_hypotheses": [ + { + "hypothesis_text": "网关超时阈值可能偏低。", + "needed_evidence": ["网关超时阈值配置"] + } + ], + "missing_info": [ + "缺少网关侧具体配置参数,无法确认阈值是否过低。" + ], + "recommended_actions": [ + { + "action_text": "检查网关侧超时阈值配置", + "reason": "当前缺少网关配置参数,需确认阈值是否低于实际请求耗时。" + } + ], + "rationale": "已有证据支持请求超时现象,但根因仍缺少直接证据。" +} +``` + +| 字段 | 类型 | 定义 | +|---|---|---| +| `original_query` | string | 用户原始问题 | +| `verdict` | string | Verifier verdict | +| `allowed_claims` | array | Verifier 允许作为确认事实输出的 claims | +| `allowed_hypotheses` | array | Verifier 允许作为合理推测输出的 hypotheses | +| `missing_info` | array | 证据缺口 | +| `recommended_actions` | array | 允许输出的建议动作 | +| `rationale` | string | Verifier 总体判定理由 | + +Composer 不应接收 raw tool output,也不应接收未经筛选的完整 Executor 输出。 + +### 19.2.1 Composer 输入组装规则 + +ChatService 根据 Verifier 输出组装 Composer 输入: + +| Verifier 结果 | Composer 输入 | +|---|---| +| `claim_checks[].verification = direct_observation` | 放入 `allowed_claims` | +| `claim_checks[].verification = reasonable_inference` | 可放入 `allowed_claims`,但表达时不得写成唯一根因 | +| `claim_checks[].verification = overstated` | 不放入 `allowed_claims`,可降级为 `allowed_hypotheses` 或 `missing_info` | +| `claim_checks[].verification = unsupported` | 不输出为事实,放入 `missing_info` | +| `claim_checks[].verification = external_unknown` | 不输出,必要时放入 `missing_info` | +| `claim_checks[].verification = contradicted` | 不输出,触发 `REJECT` 降级表达 | + +`allowed_hypotheses` 只允许来自: + +- 原始 `hypotheses`。 +- `hypothesis_checks` 判为 `reasonable_hypothesis` 的条目。 +- 被 Verifier 判为 `overstated` 后降级的 claim。 + +Composer 执行条件: + +```text +Verifier 输出有效 verdict 后才执行 Composer。 +executor_output_parse_status = missing/malformed 时,可跳过 Composer,直接使用固定低置信模板。 +gatekeeper_result.status = fail 且 verdict = REJECT 时,Composer 只能接收 allowed_claims、missing_info、recommended_actions,不接收 allowed_hypotheses。 +``` + +REJECT 场景下: + +- `allowed_hypotheses` 必须为空。 +- `user_facing_answer` 不得出现根因结论。 +- 推荐动作只能是补证据或人工复核类动作。 + +### 19.3 Composer 输出 + +Composer 输出严格 JSON。 + +```json +{ + "answer_summary": "当前已确认订单123支付失败期间出现 ERR_TIMEOUT,请求耗时 5.3 秒;网关阈值是否过低尚未确认。", + "recommended_actions": [ + { + "action_text": "检查网关侧超时阈值配置", + "reason": "当前缺少网关配置参数,需确认阈值是否低于实际请求耗时。" + } + ], + "user_facing_answer": "您的订单123在支付失败期间出现了请求超时,记录显示 ERR_TIMEOUT,耗时约 5.3 秒。目前还不能确认网关阈值配置就是根因,因为缺少网关侧具体配置参数。建议下一步检查网关超时阈值,并与该请求耗时进行比对。" +} +``` + +| 字段 | 类型 | 定义 | +|---|---|---| +| `answer_summary` | string | 一句话摘要,只能总结 Verifier 允许输出的内容 | +| `recommended_actions` | array | 面向用户的建议动作 | +| `recommended_actions[].action_text` | string | 建议动作 | +| `recommended_actions[].reason` | string | 建议原因 | +| `user_facing_answer` | string | 最终面向用户的中文答案 | + +### 19.4 Verdict 表达规则 + +#### PASS + +- 可以表达确认结论。 +- 只能使用 `allowed_claims` 和 `recommended_actions`。 +- 只有当 `allowed_claims` 中存在 root cause 类型 claim 时,才能使用“根因已确认”类表述。 + +#### LOW_CONFID + +- 必须说明当前证据仍有缺口。 +- 必须区分“已确认信息”和“可能方向”。 +- 不得把 `allowed_hypotheses` 写成确认结论。 +- 必须包含至少一个证据缺口或下一步建议。 + +#### REJECT + +- 必须说明当前无法基于已获取证据生成可靠结论。 +- 不得输出根因结论。 +- 只输出已确认信息和下一步建议。 + +### 19.5 Composer Prompt 草案 + +```text +你是 Answer Composer。你的职责是把 Verifier 允许输出的结构化材料,组织成用户可读的中文答案。 + +边界: +- 你不负责诊断。 +- 你不调用工具。 +- 你不补充新事实。 +- 你不重新判断根因。 +- 你只能使用输入中的 allowed_claims、allowed_hypotheses、missing_info、recommended_actions、rationale。 + +输入字段: +- original_query:用户原始问题 +- verdict:PASS / LOW_CONFID / REJECT +- allowed_claims:允许作为确认事实输出的结论 +- allowed_hypotheses:允许作为合理推测输出的内容 +- missing_info:证据缺口 +- recommended_actions:建议动作 +- rationale:Verifier 判定理由 + +表达规则: +1. 如果 verdict=PASS: + - 可以表达确认结论。 + - 只能使用 allowed_claims 和 recommended_actions。 +2. 如果 verdict=LOW_CONFID: + - 必须说明当前证据仍有缺口。 + - 必须区分“已确认”和“可能方向”。 + - 不得把 allowed_hypotheses 写成确认结论。 +3. 如果 verdict=REJECT: + - 必须说明当前无法基于已获取证据生成可靠结论。 + - 不得输出根因结论。 + - 只输出已确认信息和下一步建议。 + +禁止: +- 禁止新增输入中不存在的服务名、订单号、时间、指标值、错误码、根因。 +- 禁止把相关性写成因果。 +- 禁止把假设写成事实。 +- 禁止输出 Markdown。 +- 禁止输出 JSON 之外的任何文字。 + +输出格式: +{ + "answer_summary": "...", + "recommended_actions": [ + { + "action_text": "...", + "reason": "..." + } + ], + "user_facing_answer": "..." +} +``` + +--- + +## 20. 当前代码影响面 + +本 issue 按当前三 Agent 实现增量落地,不改 Planner,不引入 Controller。 + +| 模块 | 当前职责 | 本次改动 | +|---|---|---| +| `chat-planner-prompt.md` | 选择 skill、拆解计划 | 不改 | +| `chat-executor-prompt.md` | 执行工具并输出 `executor_evidence_v1` | 改为输出 `executor_evidence_v2`,移除最终表达字段 | +| `VerifierInputHook` | 解析 Executor 输出、构造 Verifier payload | 收敛为 parse + trace summary + Gatekeeper + payload | +| `ToolTraceSummaryService` | 从 `tool_invocation` 汇总证据索引 | 原则上不改;只有 Gatekeeper 缺少比对字段时才补最小字段 | +| `chat-verifier-prompt.md` | 校验 `executor_final_answer` 与证据 | 改为校验 `executor_structured_output.claims` 的可推导性 | +| `ChatService.parseVerifierDecision(...)` | 解析 `facts_checked` | 增加 `claim_checks` 解析和 `facts_checked` 兼容生成 | +| `ChatService.persistVerifierEvaluation(...)` | 写入 verifier 审计结果 | 增加 `gatekeeper_result`、`claim_checks`、Composer 结果 | +| `ChatService` 最终答案渲染 | PASS 使用 `user_facing_answer`,LOW_CONFID/REJECT 用模板 | 改为由 Composer 或固定降级模板生成最终答案 | + +第一版落地原则: + +- Planner 不改。 +- Gatekeeper 仍放在 Verifier 的 hook,即 `VerifierInputHook`。 +- `VerifierInputHook` 原有结构校验迁移到 Gatekeeper,hook 自身只保留 JSON parse。 +- 兼容期保留 `executor_final_answer`,但只作为 raw debug 字段。 +- 兼容期保留 `facts_checked`,但由 `claim_checks` 映射生成。 + +--- + +## 21. 实施阶段 + +### 阶段一:Executor V2 输出契约 + +目标:先让 Executor 不再输出最终诊断话术,只输出结构化诊断材料。 + +改动范围: + +- 更新 `chat-executor-prompt.md`。 +- `answer_version` 从 `executor_evidence_v1` 改为 `executor_evidence_v2`。 +- 移除 `diagnosis_summary` 和 `user_facing_answer`。 +- 保留 `claims`、`hypotheses`、`recommended_actions`、`missing_info` 的整体形状。 + +验收标准: + +- Executor 示例输出可被 JSON 解析。 +- 输出中不再出现 `diagnosis_summary`、`user_facing_answer`。 +- `claims[].evidence_bindings` 仍要求非空。 +- 现有流程即使尚未接入 Composer,也不会把 Executor raw JSON 直接当最终答案泄露给用户。 + +### 阶段二:Gatekeeper 接入 VerifierInputHook + +目标:在 Verifier 之前用确定性规则拦截物理级幻觉。 + +改动范围: + +- 新增 Gatekeeper 规则接口和校验服务。 +- 新增规则索引与元数据配置。 +- 在 `VerifierInputHook` 中调用 Gatekeeper。 +- 将 `gatekeeper_result` 写入 Verifier payload。 +- 将 `gatekeeper_result` 持久化到 `self_evaluation.verifier_evaluation`。 + +验收标准: + +- JSON parse 成功但 schema 不合法时,`executor_output_parse_status.status=valid`,`gatekeeper_result.status=fail`。 +- 编造不存在的 `source_invocation_ids` 时,`failed_rules` 包含 `evidence.invocation_ref`。 +- `tool_name` 与真实 invocation 不匹配时,Gatekeeper fail。 +- 缺省 `hypotheses`、`recommended_actions`、`missing_info` 时,传给 Verifier 前会 normalize 为 `[]`。 +- 旧的 `VerifierInputHook` 结构校验不再和 Gatekeeper 重复维护。 + +### 阶段三:Verifier V2 可推导性校验 + +目标:Verifier 不再逐字扫描自然语言,而是校验 claim 是否能由证据合理推出。 + +改动范围: + +- 更新 `chat-verifier-prompt.md`。 +- Verifier 主输入改为 `executor_structured_output`。 +- 输出新增 `claim_checks`。 +- 兼容期继续输出或由代码生成 `facts_checked`。 +- `parseVerifierDecision(...)` 能处理 `claim_checks` 和旧 `facts_checked`。 + +验收标准: + +- `direct_observation`、`reasonable_inference`、`overstated`、`unsupported`、`external_unknown`、`contradicted` 均有测试覆盖。 +- `gatekeeper_result.status=fail` 时 Verifier 不允许输出 `PASS`。 +- malformed/missing Executor 输出不回退自然语言抽事实,直接进入 `LOW_CONFID`。 +- `facts_checked` 兼容字段能继续支撑现有低置信模板、retry_context 和评测用例。 + +### 阶段四:Composer 输出最终答案 + +目标:把最终用户表达从 Executor 中移出,由 Composer 基于 Verifier 允许的材料生成。 + +改动范围: + +- 新增 `chat-composer-prompt.md` 或等价 Composer 调用。 +- ChatService 在 Verifier 之后组装 Composer 输入。 +- Composer 只接收 `allowed_claims`、`allowed_hypotheses`、`missing_info`、`recommended_actions`、`rationale`。 +- PASS/LOW_CONFID/REJECT 均不再读取 Executor 的 `user_facing_answer`。 + +验收标准: + +- Composer 输出严格 JSON,包含 `answer_summary`、`recommended_actions`、`user_facing_answer`。 +- Composer 不接收 raw tool output。 +- `REJECT` 时 `allowed_hypotheses=[]`,最终答案不出现根因结论。 +- `LOW_CONFID` 时必须区分已确认信息和可能方向。 +- `PASS` 时只有存在 root cause 类型 allowed claim,才允许表达“根因已确认”。 + +### 阶段五:回归评测与审计闭环 + +目标:确认新链路真的降低证据归因幻觉,而不是只改变字段名。 + +改动范围: + +- 扩展 `VerifierInputHookTest`。 +- 扩展 `ChatServiceSequentialAgentTest`。 +- 增加诊断 eval fixtures,覆盖伪造 ID、张冠李戴、过度推断、缺证据降级。 +- 检查 `diagnosis_session.self_evaluation` 中的审计字段。 + +验收标准: + +- 伪造 invocation id 必须进入 `REJECT` 或至少不可 `PASS`。 +- excerpt 相似度不足但 invocation 合法时,默认进入 `LOW_CONFID`,不直接误杀为严重伪造。 +- Executor 把 hypothesis 写进 claim 时,Verifier 至少判 `overstated` 或 `unsupported`。 +- 最终答案中不再出现未通过 Verifier 的 claim。 +- 审计链路能从 final answer 回溯到 Composer 输入、Verifier 判定、Gatekeeper 结果、tool invocation。 + +--- + +## 22. 兼容与回滚策略 + +### 22.1 兼容字段 + +第一版保留以下兼容字段,降低一次性改动风险: + +| 字段 | 保留原因 | 后续处理 | +|---|---|---| +| `executor_final_answer` | 当前 Verifier payload、fallback、日志中仍使用该名称 | 作为 deprecated raw 字段保留一版 | +| `facts_checked` | 当前低置信模板、retry_context、评测可能依赖 | 从 `claim_checks` 映射生成 | +| `traceability_version` | 当前审计结构已有字段 | V2 可改为 `v2`,但不作为功能判断依据 | + +### 22.2 回滚开关 + +建议保留最小运行时开关: + +```text +structuredOutputV2.enabled +gatekeeper.enabled +composer.enabled +``` + +回滚策略: + +- 仅 Executor V2 出问题:关闭 `structuredOutputV2.enabled`,回到 V1 prompt。 +- Gatekeeper 误杀:关闭 `gatekeeper.enabled`,Verifier 仍可按 V2 claim 校验运行。 +- Composer 输出异常:关闭 `composer.enabled`,回到固定 LOW_CONFID/REJECT 模板;PASS 暂不直接使用 Executor 输出。 + +即使回滚 Composer,也不能恢复使用 Executor 的 `user_facing_answer`,否则会重新引入本 issue 要解决的问题。 + +--- + +## 23. 数据库与审计最小字段 + +不新增表,第一版继续写入 `diagnosis_session.self_evaluation`。 + +建议结构: + +```json +{ + "verifier_evaluation": { + "verdict": "LOW_CONFID", + "groundedness_score": 0.62, + "critical_fact_count": 1, + "claim_checks": [], + "facts_checked": [], + "rationale": "...", + "round": 1, + "traceability_version": "v2", + "executor_output_parse_status": {}, + "executor_structured_output": {}, + "tool_trace_summary": [], + "gatekeeper_result": {}, + "composer_output": {} + } +} +``` + +字段控制原则: + +- Gatekeeper 只落 `status`、`failed_rules`、`warnings`、`errors`。 +- 不在数据库中保存规则元数据完整快照。 +- 不新增 `checked_at`、`rule_set_version`、`details` 等重字段。 +- 如果后续需要复盘规则版本,再单独设计审计版本字段。 + +--- + +## 24. 实施前待确认点 + +以下问题不阻塞第一阶段,但实施前需要明确默认答案: + +| 问题 | 建议默认 | +|---|---| +| Composer 是第四个 Agent 还是普通服务调用? | 作为 `chat_composer` Agent,但由 ChatService 在 Verifier 后过滤输入再调用 | +| `schema.executor_v2` 失败是否 REJECT? | 第一版 `LOW_CONFID`,因为它可能是格式退化,不一定是伪造 | +| `evidence.excerpt_similarity` 失败是否 REJECT? | 默认 `LOW_CONFID`;只有明显张冠李戴才 REJECT | +| `claim.evidence_utilization` 是否启用? | 第一版禁用或 warn-only,避免中文分词误杀 | +| 是否改 Planner? | 不改 | +| 是否改现有重试机制? | 不改;Gatekeeper 失败第一版不触发自动回退重试 | + +--- + +## 25. Definition of Done + +本 issue 完成时,需要同时满足: + +- Executor 不再输出 `diagnosis_summary` 和 `user_facing_answer`。 +- Gatekeeper 已接入 `VerifierInputHook`,并进入 Verifier payload 与 `self_evaluation`。 +- Verifier 以 `claim_checks` 为主输出,并保留 `facts_checked` 兼容。 +- Composer 负责最终 `user_facing_answer`。 +- PASS 答案不包含未通过 Verifier 的 claim。 +- LOW_CONFID/REJECT 答案不泄露 Executor 原始结论。 +- 关键回归测试覆盖伪造 ID、工具名不匹配、excerpt 张冠李戴、hypothesis 写成事实、schema 退化。 diff --git a/openspec/changes/archive/2026-07-07-executor-v2-output-contract/.archive-ready b/openspec/changes/archive/2026-07-07-executor-v2-output-contract/.archive-ready new file mode 100644 index 0000000..effb6b4 --- /dev/null +++ b/openspec/changes/archive/2026-07-07-executor-v2-output-contract/.archive-ready @@ -0,0 +1 @@ +archive-ready diff --git a/openspec/changes/archive/2026-07-07-executor-v2-output-contract/.committed b/openspec/changes/archive/2026-07-07-executor-v2-output-contract/.committed new file mode 100644 index 0000000..d0fe822 --- /dev/null +++ b/openspec/changes/archive/2026-07-07-executor-v2-output-contract/.committed @@ -0,0 +1 @@ +committed diff --git a/openspec/changes/archive/2026-07-07-executor-v2-output-contract/.openspec.yaml b/openspec/changes/archive/2026-07-07-executor-v2-output-contract/.openspec.yaml new file mode 100644 index 0000000..aee4ef1 --- /dev/null +++ b/openspec/changes/archive/2026-07-07-executor-v2-output-contract/.openspec.yaml @@ -0,0 +1,2 @@ +schema: spec-driven +created: 2026-07-07 diff --git a/openspec/changes/archive/2026-07-07-executor-v2-output-contract/decisions.md b/openspec/changes/archive/2026-07-07-executor-v2-output-contract/decisions.md new file mode 100644 index 0000000..6df1deb --- /dev/null +++ b/openspec/changes/archive/2026-07-07-executor-v2-output-contract/decisions.md @@ -0,0 +1,112 @@ +# Decisions: executor-v2-output-contract + +## sm-flow Progress + +### Clarify + +Entry summary: implement stage one of `Executor Structured Output V2`: narrow Chat Executor output to structured diagnostic material and prevent raw JSON from leaking to users before later Gatekeeper/Verifier/Composer phases. + +Slug: `executor-v2-output-contract` + +Scale: complex overall program, but this change is the first vertical stage. It is still treated with full sm-flow gates because it changes an internal Agent output contract and must be archived before the next phase. + +### Context + +Relevant devflow history: + +- `chat-verifier-agent`: Verifier is isolated from Planner/Executor intermediate reasoning and consumes explicit verification inputs. +- `evidence-trace-hardening`: evidence-bearing tool traces are persisted and summarized through `ToolTraceSummaryService`. +- `executor-evidence-output-contract`: V1 introduced `executor_evidence_v1` with `diagnosis_summary`, structured claims, and `user_facing_answer`. + +Conflict with historical decision: + +- Previous `executor-evidence-output-contract` deliberately kept `user_facing_answer` in Executor output. +- New V2 design deliberately removes it so Composer becomes the only final-expression layer in a later phase. +- For this stage, code must bridge the gap by rendering a safe temporary Chinese answer from V2 structured fields; it must not restore Executor `user_facing_answer`. + +Current code shape: + +- `chat-executor-prompt.md` defines the V1 Executor output contract. +- `VerifierInputHook` parses Executor JSON and sets `executor_structured_output`. +- `ChatService.extractUserFacingAnswer(...)` currently reads `user_facing_answer` on PASS. +- If no replacement is added, PASS may expose raw Executor JSON after V2 removes `user_facing_answer`. + +### Grill + +Question pool: + +| Question | Mode | Resolution | +|---|---|---| +| Does stage one include Gatekeeper? | evidence-driven | No. The issue splits Gatekeeper into stage two. | +| Does stage one change Planner? | evidence-driven | No. Planner is explicitly out of scope. | +| Can `user_facing_answer` remain temporarily in Executor? | evidence-driven | No. The V2 design requires removing it in stage one. | +| How do users get readable output before Composer exists? | evidence-driven | ChatService must use a temporary structured renderer for V2 PASS output. | +| Is the internal Agent contract breaking? | evidence-driven | Yes. Removing fields from Executor JSON is internal L4, but external Chat answer behavior remains readable. | + +No user-interview questions are open for stage one because the user already approved the staged design and asked for automatic phased implementation; decision questions should pause only if implementation reveals a new product trade-off. + +### Specify + +OpenSpec artifacts: + +- `proposal.md`: scope and compatibility boundary for stage one. +- `design.md`: V2 Executor contract and temporary rendering strategy. +- `specs/chat-verifier-agent/spec.md`: delta requirements for the Executor contract. +- `tasks.md`: executable implementation and verification checklist. + +### Audit + +Architecture risk summary: + +- The first-stage change deliberately breaks the internal Executor JSON contract by removing `diagnosis_summary` and `user_facing_answer`. +- External Chat answers must remain readable Chinese, so `ChatService` needs a temporary V2 renderer before Composer exists. +- `VerifierInputHook` should remain parse-only; full schema/evidence validation is deferred to the Gatekeeper stage. +- No database schema or evidence tool signature changes are required. + +Cross-artifact alignment: + +| Source | Target | Status | +|---|---|---| +| issue background / stage one | proposal | aligned | +| proposal scope / non-goals | design | aligned | +| design contract and rendering bridge | specs | aligned | +| specs observable behavior | tasks | aligned | + +Interface impact: + +- Internal Agent output contract: L4, because `diagnosis_summary` and `user_facing_answer` are removed. +- Verifier payload: L2, because `executor_final_answer` remains raw text and `executor_structured_output` remains optional. +- External Chat/API answer: intended compatible behavior; users must still receive readable Chinese rather than raw JSON. + +### Commit + +Commit gate result: passed. + +- `proposal.md` exists and explains why this phase is needed. +- `design.md` records the V2 contract, temporary rendering strategy, non-goals, and interface impact. +- `specs/chat-verifier-agent/spec.md` expresses observable behavior for Executor V2 and user-facing rendering safety. +- `tasks.md` contains executable implementation and verification tasks. +- `cmd /c openspec validate executor-v2-output-contract` passed. +- No unresolved user-interview questions remain for this stage. + +### Apply + +Implementation summary: + +- Updated `chat-executor-prompt.md` to require `answer_version="executor_evidence_v2"`. +- Removed `diagnosis_summary` and `user_facing_answer` from the Executor final output schema and output validation rules. +- Added a temporary `ChatService` structured renderer for PASS + `executor_evidence_v2` so normal users receive readable Chinese instead of raw JSON. +- Preserved V1 `user_facing_answer` extraction for compatibility. +- Kept `VerifierInputHook` parse-only behavior compatible with V2 output. +- Adjusted `chat-verifier-prompt.md` wording so `user_facing_answer` is treated as a compatibility field, not a V2 required field. + +Verification: + +- `mvn "-Dtest=VerifierInputHookTest,ChatServiceSequentialAgentTest" test` passed. +- `cmd /c openspec validate executor-v2-output-contract` passed. + +Known limitations: + +- Gatekeeper is not implemented in this phase. +- Verifier still outputs `facts_checked`; `claim_checks` belongs to a later phase. +- The V2 renderer is temporary and should be replaced by Composer in a later phase. diff --git a/openspec/changes/archive/2026-07-07-executor-v2-output-contract/design.md b/openspec/changes/archive/2026-07-07-executor-v2-output-contract/design.md new file mode 100644 index 0000000..42c9b3c --- /dev/null +++ b/openspec/changes/archive/2026-07-07-executor-v2-output-contract/design.md @@ -0,0 +1,112 @@ +## Context + +The prior `executor_evidence_v1` contract made Executor responsible for both evidence attribution and final answer wording: + +- `diagnosis_summary` +- `user_facing_answer` + +That shape helped the first Verifier integration remain readable, but it also preserved the original problem: Executor can write unsupported or over-confident natural-language conclusions before the quality gate is complete. + +This stage implements only the first slice of the V2 migration: + +```text +Executor V2 output contract + -> existing VerifierInputHook parsing + -> existing Verifier + -> temporary ChatService structured renderer +``` + +Gatekeeper, Verifier V2 `claim_checks`, and Composer are later phases. + +## Goals / Non-Goals + +Goals: + +- Make Chat Executor emit `executor_evidence_v2`. +- Remove `diagnosis_summary` and `user_facing_answer` from Executor output. +- Keep confirmed claims, hypotheses, recommended actions, and missing information as structured fields. +- Preserve evidence-binding requirements for confirmed claims. +- Prevent PASS routing from returning raw JSON to normal Chat users. + +Non-goals: + +- No Gatekeeper implementation. +- No Verifier prompt rewrite to `claim_checks`. +- No Composer agent. +- No database schema changes. +- No Planner changes. +- No evidence tool signature changes. +- No retry behavior changes. + +## Executor V2 Contract + +Executor final output SHALL be one JSON object: + +```json +{ + "answer_version": "executor_evidence_v2", + "claims": [ + { + "claim_id": "claim-1", + "claim_type": "symptom", + "claim_text": "payment-service 出现请求超时日志。", + "support_level": "direct", + "evidence_bindings": [ + { + "source_type": "tool_trace", + "source_id": "trace-1", + "tool_name": "query_logs", + "source_invocation_ids": [394], + "evidence_excerpt": "request timeout" + } + ] + } + ], + "hypotheses": [], + "recommended_actions": [], + "missing_info": [] +} +``` + +Removed fields: + +- `diagnosis_summary` +- `user_facing_answer` + +`hypotheses`, `recommended_actions`, and `missing_info` SHOULD be present as arrays. They may be empty. + +## Temporary Rendering Strategy + +Before Composer exists, `ChatService` needs a safe PASS fallback for V2 output. + +When Verifier returns `PASS`: + +1. If Executor output has `user_facing_answer`, keep the existing V1 behavior. +2. Else, if Executor output is `executor_evidence_v2`, render a readable Chinese answer from: + - `claims[].claim_text` + - `hypotheses[].hypothesis_text` + - `missing_info[]` + - `recommended_actions[].action_text` and `reason` +3. If structured rendering fails, fall back to the existing low-confidence/degraded style rather than returning raw JSON. + +The temporary renderer is not a Composer replacement. It is only a safety bridge until the Composer phase. + +## Parser Boundary + +`VerifierInputHook` may continue parsing raw Executor output into `executor_structured_output` when it is a JSON object. In this phase, it should not enforce the full V2 schema. Schema and evidence-reference validation belong to the later Gatekeeper phase. + +## Interface Impact + +- Internal Agent output contract: L4, because two fields are removed from Executor JSON. +- Verifier payload: L2, because existing `executor_final_answer` remains raw text and `executor_structured_output` remains optional. +- External Chat/API answer: intended compatible behavior; users still receive readable Chinese, not raw JSON. + +## Risks / Mitigations + +- Risk: existing PASS path exposes raw JSON because `user_facing_answer` is gone. + - Mitigation: add temporary V2 renderer in `ChatService`. +- Risk: current Verifier prompt still mentions `user_facing_answer`. + - Mitigation: stage one keeps Verifier behavior compatible; it should verify `claims` when structured output is valid and simply find no extra `user_facing_answer`. +- Risk: tests assume V1 fields. + - Mitigation: update/add focused tests for V2 output without final-expression fields. + diff --git a/openspec/changes/archive/2026-07-07-executor-v2-output-contract/proposal.md b/openspec/changes/archive/2026-07-07-executor-v2-output-contract/proposal.md new file mode 100644 index 0000000..459d3fe --- /dev/null +++ b/openspec/changes/archive/2026-07-07-executor-v2-output-contract/proposal.md @@ -0,0 +1,33 @@ +## Why + +The current Chat Executor evidence contract still mixes diagnostic material with final user-facing prose through `diagnosis_summary` and `user_facing_answer`. This keeps the Executor in a "diagnose and narrate" mode, so unsupported details can be smuggled into the final answer before later Gatekeeper, Verifier V2, and Composer phases exist. + +This first phase narrows Executor to structured diagnostic material only and adds a temporary safe rendering path so normal Chat responses do not expose raw Executor JSON while later phases are implemented. + +## What Changes + +- **BREAKING internal Agent contract**: Chat Executor output changes from `executor_evidence_v1` to `executor_evidence_v2`. +- Remove `diagnosis_summary` and `user_facing_answer` from the Chat Executor final JSON contract. +- Keep the existing structured arrays: `claims`, `hypotheses`, `recommended_actions`, and `missing_info`. +- Preserve `claims[].evidence_bindings` and current-session evidence attribution rules. +- Adjust runtime final-answer handling so a PASS result with V2 Executor output is rendered into readable Chinese from structured fields instead of returning raw JSON. +- Keep Planner, Verifier, Gatekeeper, retry behavior, database schema, and tool signatures unchanged in this phase. + +## Capabilities + +### New Capabilities + +None. + +### Modified Capabilities + +- `chat-verifier-agent`: The Executor evidence-attribution contract is tightened so V2 structured output no longer contains final-expression fields. Verifier still receives `executor_final_answer` as raw text and `executor_structured_output` when parseable. + +## Impact + +- Affected prompt: `src/main/resources/prompts/chat-executor-prompt.md`. +- Affected runtime: `ChatService` PASS answer extraction/rendering for Executor V2. +- Affected parser boundary: `VerifierInputHook` should continue parsing JSON but must not treat schema validation as its own responsibility in this phase. +- Affected tests: ChatService sequential flow tests and VerifierInputHook parsing tests for V2 output without `user_facing_answer`. +- Interface impact: L4 for internal Agent output contract because fields are removed from Executor JSON; external HTTP/chat answer behavior must remain readable Chinese and must not expose raw JSON. + diff --git a/openspec/changes/archive/2026-07-07-executor-v2-output-contract/specs/chat-verifier-agent/spec.md b/openspec/changes/archive/2026-07-07-executor-v2-output-contract/specs/chat-verifier-agent/spec.md new file mode 100644 index 0000000..e7db139 --- /dev/null +++ b/openspec/changes/archive/2026-07-07-executor-v2-output-contract/specs/chat-verifier-agent/spec.md @@ -0,0 +1,50 @@ +## MODIFIED Requirements + +### Requirement: Executor SHALL output an evidence-attribution contract +The Chat Executor SHALL produce a machine-checkable final output that separates confirmed claims from hypotheses, recommendations, and missing information. + +#### Scenario: Executor V2 final output contains only structured diagnostic fields +- **WHEN** Executor completes a Chat diagnosis step under the V2 contract +- **THEN** its final output SHALL contain `answer_version`, `claims`, `hypotheses`, `recommended_actions`, and `missing_info` +- **AND** `answer_version` SHALL equal `executor_evidence_v2` +- **AND** the output SHOULD be parseable as one JSON object without Markdown fences +- **AND** the output SHALL NOT contain `diagnosis_summary` +- **AND** the output SHALL NOT contain `user_facing_answer` + +#### Scenario: Confirmed claims carry evidence bindings +- **WHEN** Executor emits an item under `claims` +- **THEN** the item SHALL include `claim_id`, `claim_type`, `claim_text`, `support_level`, and `evidence_bindings` +- **AND** `support_level` SHALL be one of `direct` or `indirect` +- **AND** `evidence_bindings` SHALL contain at least one evidence binding + +#### Scenario: Evidence bindings support multiple tool types +- **WHEN** Executor binds evidence to a claim +- **THEN** each binding SHALL include `source_type`, `tool_name`, `source_invocation_ids`, and `evidence_excerpt` +- **AND** the binding MAY include `source_id` +- **AND** the binding SHALL be able to reference `lookup_knowledge`, `query_logs`, `query_metrics`, or other evidence-bearing tool traces +- **AND** the binding SHALL NOT rely only on a RAG-specific `chunk_id` + +#### Scenario: Unsupported conclusions are not confirmed claims +- **WHEN** a possible root cause, detail, or remediation lacks current-session tool evidence +- **THEN** Executor SHALL place it under `hypotheses`, `recommended_actions`, or `missing_info` +- **AND** Executor SHALL NOT present it as a confirmed claim + +#### Scenario: Runbook and skill guidance do not become incident facts +- **WHEN** Executor uses runbook, skill, or historical-case guidance +- **THEN** the guidance MAY influence `recommended_actions` +- **AND** the guidance SHALL NOT be emitted as a current incident fact unless current-session tool evidence supports it + +### Requirement: User-facing Chat answers SHALL remain readable Chinese +The system SHALL preserve a readable Chinese answer for normal Chat users even when Executor emits a machine-checkable contract. + +#### Scenario: V2 machine contract is not exposed as normal user answer +- **WHEN** Executor emits `executor_evidence_v2` +- **AND** Verifier returns `PASS` +- **THEN** normal user output SHALL be rendered as readable Chinese from the structured contract or a safe fallback template +- **AND** normal user output SHALL NOT be the raw Executor JSON object + +#### Scenario: Machine contract remains available for trace inspection +- **WHEN** the Chat trace or verifier evaluation is inspected +- **THEN** the structured Executor contract MAY be shown for debugging or audit +- **AND** normal user output SHALL use the existing verifier-routed display path rather than exposing raw JSON by default + diff --git a/openspec/changes/archive/2026-07-07-executor-v2-output-contract/tasks.md b/openspec/changes/archive/2026-07-07-executor-v2-output-contract/tasks.md new file mode 100644 index 0000000..6bd2194 --- /dev/null +++ b/openspec/changes/archive/2026-07-07-executor-v2-output-contract/tasks.md @@ -0,0 +1,24 @@ +## 1. Executor V2 Prompt + +- [x] 1.1 Update `src/main/resources/prompts/chat-executor-prompt.md` so the final contract uses `answer_version="executor_evidence_v2"`. +- [x] 1.2 Remove `diagnosis_summary` and `user_facing_answer` from the required Executor output schema and validation rules. +- [x] 1.3 Keep claims, hypotheses, recommended actions, missing information, and evidence-binding rules. +- [x] 1.4 Keep Planner and tool-use behavior unchanged. + +## 2. Runtime Rendering Safety + +- [x] 2.1 Update `ChatService` PASS handling so V2 structured output is rendered into readable Chinese instead of raw JSON. +- [x] 2.2 Preserve V1 `user_facing_answer` extraction for compatibility. +- [x] 2.3 Ensure fallback behavior does not expose raw Executor JSON when structured rendering fails. + +## 3. Parser Compatibility + +- [x] 3.1 Keep `VerifierInputHook` parse-only behavior compatible with V2 output. +- [x] 3.2 Add or update tests proving V2 output without `user_facing_answer` parses into `executor_structured_output`. + +## 4. Tests And Verification + +- [x] 4.1 Add or update ChatService tests for PASS with `executor_evidence_v2`. +- [x] 4.2 Add or update tests proving final user answer does not contain raw JSON contract text. +- [x] 4.3 Run targeted tests for ChatService and VerifierInputHook. +- [x] 4.4 Validate this OpenSpec change. diff --git a/openspec/specs/chat-verifier-agent/spec.md b/openspec/specs/chat-verifier-agent/spec.md index 06643b9..bdef771 100644 --- a/openspec/specs/chat-verifier-agent/spec.md +++ b/openspec/specs/chat-verifier-agent/spec.md @@ -256,20 +256,24 @@ When Verifier returns `LOW_CONFID`, user-facing output SHALL clearly separate co ### Requirement: Executor SHALL output an evidence-attribution contract The Chat Executor SHALL produce a machine-checkable final output that separates confirmed claims from hypotheses, recommendations, and missing information. -#### Scenario: Executor final output contains required top-level fields -- **WHEN** Executor completes a Chat diagnosis step -- **THEN** its final output SHALL contain `answer_version`, `diagnosis_summary`, `claims`, `hypotheses`, `recommended_actions`, `missing_info`, and `user_facing_answer` +#### Scenario: Executor V2 final output contains only structured diagnostic fields +- **WHEN** Executor completes a Chat diagnosis step under the V2 contract +- **THEN** its final output SHALL contain `answer_version`, `claims`, `hypotheses`, `recommended_actions`, and `missing_info` +- **AND** `answer_version` SHALL equal `executor_evidence_v2` - **AND** the output SHOULD be parseable as one JSON object without Markdown fences +- **AND** the output SHALL NOT contain `diagnosis_summary` +- **AND** the output SHALL NOT contain `user_facing_answer` #### Scenario: Confirmed claims carry evidence bindings - **WHEN** Executor emits an item under `claims` - **THEN** the item SHALL include `claim_id`, `claim_type`, `claim_text`, `support_level`, and `evidence_bindings` -- **AND** `support_level` SHALL be one of `direct`, `indirect`, or `none` -- **AND** claims with `support_level=direct` or `support_level=indirect` SHALL include at least one evidence binding +- **AND** `support_level` SHALL be one of `direct` or `indirect` +- **AND** `evidence_bindings` SHALL contain at least one evidence binding #### Scenario: Evidence bindings support multiple tool types - **WHEN** Executor binds evidence to a claim -- **THEN** each binding SHALL include `source_type`, `source_id`, `tool_name`, `source_invocation_ids`, and `evidence_excerpt` +- **THEN** each binding SHALL include `source_type`, `tool_name`, `source_invocation_ids`, and `evidence_excerpt` +- **AND** the binding MAY include `source_id` - **AND** the binding SHALL be able to reference `lookup_knowledge`, `query_logs`, `query_metrics`, or other evidence-bearing tool traces - **AND** the binding SHALL NOT rely only on a RAG-specific `chunk_id` @@ -286,10 +290,11 @@ The Chat Executor SHALL produce a machine-checkable final output that separates ### Requirement: User-facing Chat answers SHALL remain readable Chinese The system SHALL preserve a readable Chinese answer for normal Chat users even when Executor emits a machine-checkable contract. -#### Scenario: User-facing answer is available -- **WHEN** Executor emits structured output -- **THEN** `user_facing_answer` SHALL be written in Chinese -- **AND** it SHALL be consistent with the confirmed claims, hypotheses, recommended actions, and missing information in the same JSON object +#### Scenario: V2 machine contract is not exposed as normal user answer +- **WHEN** Executor emits `executor_evidence_v2` +- **AND** Verifier returns `PASS` +- **THEN** normal user output SHALL be rendered as readable Chinese from the structured contract or a safe fallback template +- **AND** normal user output SHALL NOT be the raw Executor JSON object #### Scenario: Machine contract remains available for trace inspection - **WHEN** the Chat trace or verifier evaluation is inspected @@ -309,3 +314,4 @@ The system SHALL tolerate malformed or absent structured Executor output without - **WHEN** Executor output parsing fails - **THEN** the verifier evaluation or trace snapshot SHALL make the parse failure visible - **AND** the failure SHALL NOT be silently treated as a successful evidence-attribution contract + diff --git a/src/main/java/com/superbiz/agent/service/ChatService.java b/src/main/java/com/superbiz/agent/service/ChatService.java index d0e2b9f..0993ac5 100644 --- a/src/main/java/com/superbiz/agent/service/ChatService.java +++ b/src/main/java/com/superbiz/agent/service/ChatService.java @@ -444,8 +444,12 @@ public class ChatService { } if ("PASS".equals(finalDecision.verdict())) { - answer = extractUserFacingAnswer(answer) - .orElse(answer == null || answer.isBlank() ? "抱歉,多 Agent 分析未能生成有效结论。" : answer); + String executorAnswer = answer; + answer = extractUserFacingAnswer(executorAnswer) + .or(() -> renderStructuredExecutorAnswer(executorAnswer)) + .orElse(executorAnswer == null || executorAnswer.isBlank() + ? "抱歉,多 Agent 分析未能生成有效结论。" + : executorAnswer); persistVerifierEvaluation(session, finalDecision, round); break; } @@ -816,6 +820,98 @@ public class ChatService { return Optional.empty(); } + private Optional renderStructuredExecutorAnswer(String executorAnswer) { + if (executorAnswer == null || executorAnswer.isBlank()) { + return Optional.empty(); + } + try { + JsonNode root = objectMapper.readTree(sanitizeJsonPayload(executorAnswer)); + if (!"executor_evidence_v2".equals(root.path("answer_version").asText(""))) { + return Optional.empty(); + } + + StringBuilder output = new StringBuilder(); + appendTextArraySection(output, "已确认信息", root.path("claims"), "claim_text", "暂无可稳定确认的信息"); + appendHypothesesSection(output, root.path("hypotheses")); + appendStringArraySection(output, "当前缺口", root.path("missing_info"), "当前缺少足够的直接证据支撑完整结论"); + appendRecommendedActionsSection(output, root.path("recommended_actions")); + String rendered = output.toString().trim(); + return rendered.isBlank() ? Optional.empty() : Optional.of(rendered); + } catch (Exception e) { + logger.debug("Failed to render executor_evidence_v2 answer", e); + return Optional.empty(); + } + } + + private void appendTextArraySection(StringBuilder output, String title, JsonNode items, + String fieldName, String emptyText) { + output.append(title).append(":"); + if (!items.isArray() || items.isEmpty()) { + output.append("\n- ").append(emptyText); + return; + } + for (JsonNode item : items) { + String text = item.path(fieldName).asText(""); + if (!text.isBlank()) { + output.append("\n- ").append(text); + } + } + if (output.charAt(output.length() - 1) == ':') { + output.append("\n- ").append(emptyText); + } + } + + private void appendHypothesesSection(StringBuilder output, JsonNode hypotheses) { + if (!hypotheses.isArray() || hypotheses.isEmpty()) { + return; + } + output.append("\n\n可能方向:"); + for (JsonNode hypothesis : hypotheses) { + String text = hypothesis.path("hypothesis_text").asText(""); + if (text.isBlank()) { + continue; + } + String basis = hypothesis.path("basis").asText(""); + output.append("\n- ").append(text); + if (!basis.isBlank()) { + output.append("(").append(basis).append(")"); + } + } + } + + private void appendStringArraySection(StringBuilder output, String title, JsonNode items, String emptyText) { + output.append("\n\n").append(title).append(":"); + if (!items.isArray() || items.isEmpty()) { + output.append("\n- ").append(emptyText); + return; + } + for (JsonNode item : items) { + String text = item.asText(""); + if (!text.isBlank()) { + output.append("\n- ").append(text); + } + } + } + + private void appendRecommendedActionsSection(StringBuilder output, JsonNode actions) { + output.append("\n\n建议下一步:"); + if (!actions.isArray() || actions.isEmpty()) { + output.append("\n- 围绕上述证据缺口补充只读查询,再由人工复核最终结论"); + return; + } + for (JsonNode action : actions) { + String text = action.path("action_text").asText(""); + if (text.isBlank()) { + continue; + } + String reason = action.path("reason").asText(""); + output.append("\n- ").append(text); + if (!reason.isBlank()) { + output.append(":").append(reason); + } + } + } + private String buildDegradedOutput(VerifierDecision decision) { StringBuilder output = new StringBuilder(DEGRADED_PREFIX); diff --git a/src/main/resources/prompts/chat-executor-prompt.md b/src/main/resources/prompts/chat-executor-prompt.md index 703bda0..4c94442 100644 --- a/src/main/resources/prompts/chat-executor-prompt.md +++ b/src/main/resources/prompts/chat-executor-prompt.md @@ -51,7 +51,6 @@ 支持等级: - `direct`:工具返回中有直接事实。 - `indirect`:工具返回可支撑方向,但没有直接陈述完整结论。 -- `none`:不能放入 `claims`,应放入 `hypotheses`、`recommended_actions` 或 `missing_info`。 ### hypotheses `hypotheses` 用来放合理怀疑但未被工具证实的方向。 @@ -71,8 +70,7 @@ ```json { - "answer_version": "executor_evidence_v1", - "diagnosis_summary": "1-2句话总结,仅包含有证据支撑的事实和证据边界", + "answer_version": "executor_evidence_v2", "claims": [ { "claim_id": "claim-1", @@ -106,15 +104,16 @@ ], "missing_info": [ "导致无法确认完整根因的证据缺口" - ], - "user_facing_answer": "面向用户的中文回答。必须与 claims/hypotheses/recommended_actions/missing_info 一致,不得额外加入未绑定证据的确认式事实。" + ] } ``` ## 输出校验 +- `answer_version` 必须是 `executor_evidence_v2`。 +- 不得输出 `diagnosis_summary`。 +- 不得输出 `user_facing_answer`。 - `claims[*].support_level` 只能是 `direct` 或 `indirect`。 - `claims[*].evidence_bindings` 不能为空。 - `evidence_excerpt` 必须来自工具返回,不允许编造。 - 如果没有任何可确认事实,`claims` 返回空数组,并在 `missing_info` 说明缺少什么。 -- `user_facing_answer` 不得出现 `claims` 中没有、且又被写成确认结论的事实。 - 不要把其它服务、其它历史案例、其它会话的事实迁移为当前会话事实。 diff --git a/src/main/resources/prompts/chat-verifier-prompt.md b/src/main/resources/prompts/chat-verifier-prompt.md index 11891eb..633a40d 100644 --- a/src/main/resources/prompts/chat-verifier-prompt.md +++ b/src/main/resources/prompts/chat-verifier-prompt.md @@ -10,7 +10,7 @@ - `original_query`:用户原始问题 - `executor_final_answer`:本轮 Executor 最终答案 -- `executor_structured_output`:如果 Executor 输出了合法证据归因 JSON,这里会提供解析后的对象。结构包含 `claims`、`hypotheses`、`recommended_actions`、`missing_info`、`user_facing_answer` +- `executor_structured_output`:如果 Executor 输出了合法证据归因 JSON,这里会提供解析后的对象。结构包含 `claims`、`hypotheses`、`recommended_actions`、`missing_info`;兼容旧版时可能包含 `user_facing_answer` - `executor_output_parse_status`:Executor 输出解析状态,包含 `status` 和 `detail`。`status` 可能是 `valid` / `missing` / `malformed` - `tool_trace_summary`:基于真实工具调用整理出的证据索引。每一项都带有: - `trace_ref` @@ -31,7 +31,7 @@ - 必须检查 claim 的 `evidence_bindings` 是否能对应到 `tool_trace_summary` 中真实存在的 trace、tool 或 source_invocation_ids - 如果 claim 声称 direct/indirect 支撑,但 evidence binding 不存在、无法定位、或 excerpt 与工具摘要不匹配,不得判为 `direct_evidence` -然后必须扫描 `executor_structured_output.user_facing_answer`: +如果 `executor_structured_output.user_facing_answer` 存在,则必须扫描它: - 如果其中出现 confirmed-sounding facts(确认式事实、根因、指标值、错误码、服务名、修复结论) - 且这些事实没有出现在 `executor_structured_output.claims` - 必须额外加入 `facts_checked` 并按工具证据校验 diff --git a/src/test/java/com/superbiz/agent/hook/VerifierInputHookTest.java b/src/test/java/com/superbiz/agent/hook/VerifierInputHookTest.java index c4b20e3..6f495f0 100644 --- a/src/test/java/com/superbiz/agent/hook/VerifierInputHookTest.java +++ b/src/test/java/com/superbiz/agent/hook/VerifierInputHookTest.java @@ -84,6 +84,55 @@ class VerifierInputHookTest { assertEquals("valid", VerifierContextHolder.getExecutorOutputParseStatus().get("status")); } + @Test + void beforeModelAddsStructuredExecutorOutputWhenV2ContractHasNoUserFacingAnswer() throws Exception { + ToolTraceSummaryService traceSummaryService = mock(ToolTraceSummaryService.class); + when(traceSummaryService.buildVerifierTraceSummary(anyString(), anyString())).thenReturn(List.of( + Map.of("trace_ref", "trace-1", "tool_name", "query_metrics") + )); + VerifierInputHook hook = new VerifierInputHook(traceSummaryService); + VerifierContextHolder.setOriginalQuery("分析 MySQL 连接池耗尽"); + + String executorOutput = """ + { + "answer_version": "executor_evidence_v2", + "claims": [ + { + "claim_id": "claim-1", + "claim_type": "symptom", + "claim_text": "连接池 active 达到上限", + "support_level": "direct", + "evidence_bindings": [ + { + "source_type": "tool_trace", + "source_id": "trace-1", + "tool_name": "query_metrics", + "source_invocation_ids": [101], + "evidence_excerpt": "active=50 max=50" + } + ] + } + ], + "hypotheses": [], + "recommended_actions": [], + "missing_info": [] + } + """; + + AgentCommand command = hook.beforeModel( + List.of(new AssistantMessage(executorOutput)), + RunnableConfig.builder().addMetadata("sessionId", "structured-v2-session").build() + ); + + JsonNode payload = readPayload(command); + assertEquals("valid", payload.path("executor_output_parse_status").path("status").asText()); + assertEquals("executor_evidence_v2", + payload.path("executor_structured_output").path("answer_version").asText()); + assertFalse(payload.path("executor_structured_output").has("user_facing_answer")); + assertEquals("连接池 active 达到上限", + payload.path("executor_structured_output").path("claims").get(0).path("claim_text").asText()); + } + @Test void beforeModelExtractsStructuredOutputFromPrefixedJsonFence() throws Exception { ToolTraceSummaryService traceSummaryService = mock(ToolTraceSummaryService.class); diff --git a/src/test/java/com/superbiz/agent/service/ChatServiceSequentialAgentTest.java b/src/test/java/com/superbiz/agent/service/ChatServiceSequentialAgentTest.java index 2628363..d92fed8 100644 --- a/src/test/java/com/superbiz/agent/service/ChatServiceSequentialAgentTest.java +++ b/src/test/java/com/superbiz/agent/service/ChatServiceSequentialAgentTest.java @@ -273,6 +273,63 @@ class ChatServiceSequentialAgentTest { assertTrue(chatModel.verifierPromptText.contains("连接池 active 达到上限")); } + @Test + void executeChatComplexRendersExecutorEvidenceV2InsteadOfRawJsonOnPass() throws Exception { + ChatService chatService = createChatService(); + ScriptedChatModel chatModel = new ScriptedChatModel(); + chatModel.executorOutput = """ + { + "answer_version": "executor_evidence_v2", + "claims": [ + { + "claim_id": "claim-1", + "claim_type": "symptom", + "claim_text": "连接池 active 达到上限", + "support_level": "direct", + "evidence_bindings": [ + { + "source_type": "tool_trace", + "source_id": "trace-1", + "tool_name": "query_metrics", + "source_invocation_ids": [101], + "evidence_excerpt": "active=50 max=50" + } + ] + } + ], + "hypotheses": [ + { + "hypothesis_text": "连接泄漏可能参与了连接池耗尽", + "basis": "已有连接池满载证据,但缺少泄漏检测日志", + "needed_evidence": ["连接泄漏检测日志"] + } + ], + "recommended_actions": [ + { + "action_text": "补充查询连接池泄漏检测日志", + "reason": "用于确认是否存在连接未释放" + } + ], + "missing_info": ["缺少连接泄漏检测日志"] + } + """; + + ChatService.ChatResult result = chatService.executeChatComplex( + chatModel, + new ToolCallback[0], + "请分析 MySQL 连接池耗尽", + List.of(), + "sequential-v2-render-session" + ); + + assertTrue(result.answer().contains("已确认信息")); + assertTrue(result.answer().contains("连接池 active 达到上限")); + assertTrue(result.answer().contains("可能方向")); + assertTrue(result.answer().contains("建议下一步")); + assertFalse(result.answer().contains("\"answer_version\"")); + assertFalse(result.answer().contains("executor_evidence_v2")); + } + @Test void buildMethodToolsArrayIncludesLogsAndMetricsWhenAvailable() { ChatService chatService = new ChatService();