Compare commits

...
4 Commits
94 changed files with 2007 additions and 738 deletions
+16 -15
View File
@@ -2,32 +2,33 @@
## 项目
| 日期 | slug | 领域 | 关键词 | 状态 |
|---|---|---|---|---|
| 2026-07-05 | diagnosis-playbook-skills | Agent Skill/Playbook | read_skill, diagnosis playbook, progressive disclosure, payment timeout, MySQL pool, Redis timeout | openspec/changes/diagnosis-playbook-skills | implemented |
| 日期 | slug | 领域 | 关键词 | 关联 OpenSpec | 状态 |
|---|---|---|---|---|---|
| 2026-07-09 | interview-demo-quality-audit | Agent eval/demo/Prompt audit | interview demo preflight, prompt_audit, gatekeeper rules, diagnosis baseline, 12 fixtures | openspec/changes/archive/2026-07-09-interview-demo-quality-audit | archived |
| 2026-07-08 | executor-composer-final-answer | Chat quality gate/evidence attribution | chat_composer, final answer, allowed_claims, allowed_hypotheses, safe fallback, composer_output | openspec/changes/archive/2026-07-08-executor-composer-final-answer | archived |
| 2026-07-08 | diagnosis-eval-demo-gatekeeper-closure | Agent eval/demo/Gatekeeper | diagnosis eval matrix, stable demo scenarios, Gatekeeper rule set version, audit metadata | openspec/changes/archive/2026-07-08-diagnosis-eval-demo-gatekeeper-closure | archived |
| 2026-07-08 | verifier-evidence-reference-fidelity | Chat质量门禁/证据归因 | evidence_refs, raw_path, Gatekeeper severity, verifier evidence excerpt, HikariCP mock, no_evidence | openspec/changes/archive/2026-07-08-verifier-evidence-reference-fidelity | archived |
| 2026-07-07 | executor-evidence-output-contract | Chat质量门禁/证据归因 | Executor structured output, evidence bindings, Verifier structured claims, LOW_CONFID, hallucination | openspec/changes/archive/2026-07-07-executor-evidence-output-contract | archived |
| 2026-07-07 | executor-v2-output-contract | Chat质量门禁/证据归因 | executor_evidence_v2, user_facing_answer removal, diagnosis_summary removal, structured renderer | openspec/changes/archive/2026-07-07-executor-v2-output-contract | archived |
| 2026-07-07 | executor-gatekeeper-hook | Chat质量门禁/证据归因 | Gatekeeper, verifier payload, source_invocation_ids, tool_name match, self_evaluation | openspec/changes/archive/2026-07-07-executor-gatekeeper-hook | archived |
| 2026-07-07 | executor-verifier-claim-checks | Chat质量门禁/证据归因 | Verifier claim_checks, facts_checked compatibility, effective verdict guardrail, malformed output downgrade | openspec/changes/archive/2026-07-07-executor-verifier-claim-checks | archived |
| 2026-07-08 | executor-composer-final-answer | Chat quality gate/evidence attribution | chat_composer, final answer, allowed_claims, allowed_hypotheses, safe fallback, composer_output | openspec/changes/archive/2026-07-08-executor-composer-final-answer | archived |
| 2026-07-08 | diagnosis-eval-demo-gatekeeper-closure | Agent eval/demo/Gatekeeper | diagnosis eval matrix, stable demo scenarios, Gatekeeper rule set version, audit metadata | openspec/changes/archive/2026-07-08-diagnosis-eval-demo-gatekeeper-closure | archived |
| 2026-07-08 | verifier-evidence-reference-fidelity | Chat质量门禁/证据归因 | evidence_refs, raw_path, Gatekeeper severity, verifier evidence excerpt, HikariCP mock, no_evidence | openspec/changes/archive/2026-07-08-verifier-evidence-reference-fidelity | archived |
| 2026-07-06 | rag-eval-pipeline-closure | RAG/评测/回归闭环 | lookupResult fixture, LookupKnowledgeTool snapshot, evidenceBlocks, contextPack, retrievalTrace, rerankTrace, baseline diff, fallback case | devflow/projects/2026-07-06-rag-eval-pipeline-closure | archived |
| 2026-07-06 | modular-rag-pipeline | RAG/Agent工具/证据链 | modular RAG, lookup_knowledge, evidenceBlocks, contextPack, rerank, retrievalTrace, L0 hint, unfiltered retry | openspec/changes/archive/2026-07-06-modular-rag-pipeline | archived |
| 2026-07-05 | diagnosis-playbook-skills | Agent Skill/Playbook | read_skill, diagnosis playbook, progressive disclosure, payment timeout, MySQL pool, Redis timeout | openspec/changes/diagnosis-playbook-skills | implemented |
| 2026-07-05 | mvp-demo-interview-runbook | MVP Demo/Interview | Plan C, payment timeout, runbook, trace checklist, demo script | openspec/changes/archive/2026-07-05-mvp-demo-interview-runbook | archived |
| 2026-07-05 | diagnosis-eval-baseline-diff | Agent 评测/回归 Diff | baseline diff, regression detection, evidence coverage, cost signal, markdown report | openspec/changes/archive/2026-07-05-diagnosis-eval-baseline-diff | archived |
| 2026-07-04 | expand-diagnosis-eval-fixtures | Agent 评测/回归 Baseline | fixture coverage, baseline report, redis timeout, slow response, jvm memory risk | openspec/changes/archive/2026-07-05-expand-diagnosis-eval-fixtures | archived |
| 2026-07-04 | diagnosis-eval-harness | Agent 评测/回归 Harness | fixed cases, trace validation, evidence coverage, verdict distribution, markdown report | openspec/changes/archive/2026-07-04-diagnosis-eval-harness | archived |
| 2026-07-04 | evidence-trace-hardening | 证据链/降级契约/离线验证 | ToolInvocationRecorder, ToolTraceSummaryService, lookup_knowledge, query_logs, query_metrics, LOW_CONFID, REJECT | openspec/changes/archive/2026-07-04-evidence-trace-hardening | archived |
| 2026-07-03 | mvp-demo-trace-acceptance | MVP Demo/trace/acceptance | mvp-demo, trace API, diagnosis_session, agent_step, tool_invocation, feedback | openspec/changes/archive/2026-07-03-mvp-demo-trace-acceptance | archived |
| 2026-07-04 | aiops-traceable-diagnosis-entry | AIOps/trace/alert diagnosis | ai_ops, SSE, alert input, sessionId, diagnosis_session, trace API | openspec/changes/archive/2026-07-04-aiops-traceable-diagnosis-entry | archived |
| 2026-07-04 | aiops-alert-scope-control | AIOps/scope/prompt control | payload mode, auto-discovery mode, queryPrometheusAlerts, HighCPUUsage | openspec/changes/archive/2026-07-04-aiops-alert-scope-control | archived |
| 2026-05-29 | chatmodel-abstraction | 解耦/多模型路由 | ChatModel, EmbeddingModel, DeepSeek, BGE-M3, SiliconFlow, Spring AI | archived |
| 2026-06-23 | phase1-infrastructure | 基础设施/文档管理 | MySQL, Redis, Milvus, Flyway, JPA, 向量检索, 类别过滤 | archived |
| 2026-06-24 | lookup-knowledge-integration | 知识库检索 | L0精确匹配, L1语义检索, frontmatter, 混合检索 | archived |
| 2026-06-25 | doc-management-ui | 前端开发/文档管理 | 文档管理页面, CRUD, 状态监控, 纯静态页面, API集成 | archived |
| 2026-06-26 | session-storage | 会话存储/可观测 | diagnosis_session, agent_step, tool_invocation, token追踪, 多Agent路由 | openspec/changes/session-storage | archived |
| 2026-06-29 | confidence-feedback | 质量评估/反馈机制 | evidence_score, selfEvaluation, feedback, useful, not_useful, case_library, BAD_CASE, tool_invocation规则引擎, 反馈按钮, sessionId回传 | openspec/changes/confidence-feedback | archived |
| 2026-06-30 | session-dedup-knowledge-map | 去重/知识图谱 | RetrievedDocTracker, KnowledgeDomainService, knowledge_domain, covers, whenToRetrieve, Planner注入, ISS-001 | openspec/changes/archive/2026-06-30-session-dedup-knowledge-map | archived |
| 2026-07-01 | executor-action-memory-relevance | 检索质量/行动记忆 | relevanceLevel, completenessHint, Min-Max归一化, RetrievedDocTracker域级记录, Executor检索约束, ISS-002 | openspec/changes/archive/2026-07-01-executor-action-memory-relevance | archived |
| 2026-07-03 | mvp-demo-trace-acceptance | MVP Demo/trace/acceptance | mvp-demo, trace API, diagnosis_session, agent_step, tool_invocation, feedback | openspec/changes/archive/2026-07-03-mvp-demo-trace-acceptance | archived |
| 2026-07-02 | chat-verifier-agent | Chat质量门禁/可追溯验证 | Verifier, groundedness_score, facts_checked, evidence_refs, tool_trace_summary, self_evaluation | openspec/changes/archive/2026-07-03-chat-verifier-agent | archived |
| 2026-07-01 | executor-action-memory-relevance | 检索质量/行动记忆 | relevanceLevel, completenessHint, Min-Max归一化, RetrievedDocTracker域级记录, Executor检索约束, ISS-002 | openspec/changes/archive/2026-07-01-executor-action-memory-relevance | archived |
| 2026-06-30 | session-dedup-knowledge-map | 去重/知识图谱 | RetrievedDocTracker, KnowledgeDomainService, knowledge_domain, covers, whenToRetrieve, Planner注入, ISS-001 | openspec/changes/archive/2026-06-30-session-dedup-knowledge-map | archived |
| 2026-06-29 | confidence-feedback | 质量评估/反馈机制 | evidence_score, selfEvaluation, feedback, useful, not_useful, case_library, BAD_CASE, tool_invocation规则引擎, 反馈按钮, sessionId回传 | openspec/changes/confidence-feedback | archived |
| 2026-06-26 | session-storage | 会话存储/可观测 | diagnosis_session, agent_step, tool_invocation, token追踪, 多Agent路由 | openspec/changes/session-storage | archived |
| 2026-06-25 | doc-management-ui | 前端开发/文档管理 | 文档管理页面, CRUD, 状态监控, 纯静态页面, API集成 | archived |
| 2026-06-24 | lookup-knowledge-integration | 知识库检索 | L0精确匹配, L1语义检索, frontmatter, 混合检索 | archived |
| 2026-06-23 | phase1-infrastructure | 基础设施/文档管理 | MySQL, Redis, Milvus, Flyway, JPA, 向量检索, 类别过滤 | archived |
| 2026-05-29 | chatmodel-abstraction | 解耦/多模型路由 | ChatModel, EmbeddingModel, DeepSeek, BGE-M3, SiliconFlow, Spring AI | archived |
@@ -48,7 +48,7 @@
## 遗留问题
ISS-002:Executor 无约束重复调用 `lookup_knowledge`(单会话 20+ 次),knowledge map 和检索约束只注入了 Planner 未注入 Executor。详见 `mvp/issues/ISS-002-executor-unconstrained-lookup.md`。
ISS-002:Executor 无约束重复调用 `lookup_knowledge`(单会话 20+ 次),knowledge map 和检索约束只注入了 Planner 未注入 Executor。详见 `mvp/issues/archived/ISS-002-executor-unconstrained-lookup.md`。
## 已知限制
@@ -9,8 +9,8 @@
## Context
- `devflow/index.md` was checked. Relevant history includes `session-storage`, `confidence-feedback`, `executor-action-memory-relevance`, and `chat-verifier-agent`.
- `mvp/notes/agent-engineering-decisions.md` already recommends the next phase as "可复现 MVP Demo", including `mvp-demo` profile, fixed diagnosis case, one-click request, and `GET /api/diagnosis/{sessionId}/trace`.
- `mvp/issues/ISS-003-mvp-design-implementation-review.md` identifies test stability, session traceability, verifier evidence chain, upload path, and SupervisorAgent consistency as recent MVP concerns. Security cleanup is intentionally deferred by user decision.
- `mvp/archive/2026-07-09-doc-cleanup/notes/agent-engineering-decisions.md` already recommends the next phase as "可复现 MVP Demo", including `mvp-demo` profile, fixed diagnosis case, one-click request, and `GET /api/diagnosis/{sessionId}/trace`.
- `mvp/issues/active/ISS-003-mvp-design-implementation-review.md` identifies test stability, session traceability, verifier evidence chain, upload path, and SupervisorAgent consistency as recent MVP concerns. Security cleanup is intentionally deferred by user decision.
## Question Pool
@@ -2,7 +2,7 @@
## Draft Acceptance
- [x] Issue exists: `mvp/issues/executor-evidence-attribution-hallucination.md`.
- [x] Issue exists: `mvp/issues/active/executor-evidence-attribution-hallucination.md`.
- [x] OpenSpec change artifacts exist.
- [x] devflow tracking files exist.
- [x] OpenSpec validation passes.
@@ -5,7 +5,7 @@
- `devflow/projects/2026-07-07-executor-evidence-output-contract`: V1 evidence-attribution contract kept `user_facing_answer`.
- `devflow/projects/2026-07-02-chat-verifier-agent`: Verifier consumes explicit inputs and should not see intermediate reasoning.
- `devflow/projects/2026-07-04-evidence-trace-hardening`: evidence summaries and tool invocation references are the evidence foundation.
- `mvp/issues/executor-structured-output-v2.md`: staged implementation design; stage one is Executor V2 output contract.
- `mvp/issues/design-notes/executor-structured-output-v2.md`: staged implementation design; stage one is Executor V2 output contract.
## Code Evidence
@@ -41,4 +41,4 @@ Make Executor cite concrete tool evidence, make Gatekeeper validate that citatio
## OpenSpec
- Change: `openspec/changes/verifier-evidence-reference-fidelity`
- Source issue: `mvp/issues/ISS-007-verifier-evidence-summary-fidelity.md`
- Source issue: `mvp/issues/archived/ISS-007-verifier-evidence-summary-fidelity.md`
@@ -0,0 +1,46 @@
# Acceptance
## Static Verification
- `openspec validate interview-demo-quality-audit --strict`
- Result: passed.
- Coverage: OpenSpec proposal/design/spec/tasks consistency.
- PowerShell parser/runtime readiness check:
- Command: `powershell -NoProfile -ExecutionPolicy Bypass -File mvp/demo/scripts/run-interview-demo-check.ps1 -BaseUrl http://127.0.0.1:1 -OutputDir target/demo-check-syntax`
- Result: expected failure with actionable readiness message.
- Coverage: script parses under Windows PowerShell and fails before issuing diagnosis requests when service is unreachable.
## Script Verification
- `mvn -q "-Dtest=DiagnosisTraceEvaluatorTest" test`
- Result: passed.
- Coverage: 12/12 fixed eval fixtures, Prompt audit evaluator checks, Gatekeeper rule metadata checks, regenerated baseline reports.
- `mvn -q "-Dtest=ChatServiceSequentialAgentTest" test`
- Result: passed.
- Coverage: Chat verifier evaluation persists `prompt_audit`.
- `mvn -q "-Dtest=DiagnosisTraceEvaluatorTest,DiagnosisEvalBaselineDiffTest,ChatServiceSequentialAgentTest,ExecutorGatekeeperServiceTest,VerifierInputHookTest" test`
- Result: passed.
- Coverage: broader eval, baseline diff, Chat sequential flow, Gatekeeper, and Verifier input hook regression set.
- `mvn -q -DskipTests compile`
- Result: passed.
- Coverage: main source compilation.
## E2E Verification
- Started service with:
- `mvn spring-boot:run -Dspring-boot.run.profiles=mvp-demo`
- Ran:
- `powershell -NoProfile -ExecutionPolicy Bypass -File mvp/demo/scripts/run-interview-demo-check.ps1 -BaseUrl http://localhost:9900 -SessionId mvp-demo-interview-quality-audit-001`
- Result: passed.
- Summary:
- `chatSuccess=true`
- `verdict=LOW_CONFID`
- `gatekeeperStatus=fail`
- `gatekeeperRuleSetVersion=gatekeeper-rules-v1`
- `promptAuditVersion=chat-prompts-v1`
- tools included `lookup_knowledge`, `query_logs`, `query_metrics`, and `get_available_log_topics`
- Note: live E2E remains a compatibility check, not the deterministic PASS oracle. The fixed fixture baseline is the regression source of truth.
## Not Verified
- Browser UI inspection was not required for this change because the scope is backend trace/eval/demo script documentation, not frontend behavior.
@@ -0,0 +1,23 @@
# Interview Demo Quality Audit Brief
## Background
The MVP already demonstrates traceable Agent diagnosis with Planner, Executor, Gatekeeper, Verifier, Composer, evidence tools, trace persistence, and deterministic eval fixtures. The remaining interview-readiness gap is not a new Agent architecture; it is making the demo easier to run and making prompt/rule changes easier to audit.
## Goal
Stabilize the interview demo path, expand fixture-backed evaluation, and persist prompt/Gatekeeper audit metadata so the project can explain and verify Agent behavior during interviews.
## Scope
- Add prompt audit metadata to Chat verifier evaluation.
- Extend deterministic eval cases and baseline reports.
- Add an interview demo preflight/check script.
- Update MVP demo and architecture documentation.
## Non-goals
- No public API or database schema changes.
- No new SubAgent split, MCP migration, process isolation, or AIOps LLM Verifier.
- No guarantee that every live LLM run returns PASS.
@@ -0,0 +1,111 @@
# interview-demo-quality-audit Decisions
## Clarify
- Entry summary: stabilize the interview demo, expand deterministic eval coverage, and add Prompt/Gatekeeper version audit.
- Slug: `interview-demo-quality-audit`.
- Devflow scale: `standard`.
- Interface impact: L2 internal contract change because `verifier_evaluation` gains `prompt_audit`; no public HTTP API or database schema change.
## Context
- `devflow/index.md` used: related entries found for `diagnosis-eval-demo-gatekeeper-closure`, `executor-composer-final-answer`, `verifier-evidence-reference-fidelity`, `mvp-demo-interview-runbook`, and `diagnosis-eval-baseline-diff`.
- Relevant glossary:
- Evidence Tools produce incident facts and must be recorded in `tool_invocation`.
- Verifier should not use skills/runbooks as incident evidence.
- Diagnosis Playbook Skill is workflow guidance, not a fact source.
- Historical constraints that must enter OpenSpec:
- Diagnosis eval is deterministic and fixture-backed; no LLM-as-judge.
- Stable demo scenarios are documentation/payloads plus deterministic fixtures; live E2E is a compatibility check, not a guaranteed PASS oracle.
- Gatekeeper rule metadata is already metadata-only and should not become dynamic rule execution.
- Composer is the final expression layer and must not leak raw Executor JSON.
## Question Pool
| ID | Dimension | Mode | Question | Status |
|---|---|---|---|---|
| Q1 | Terminology | evidence-driven | What should the new audit metadata be called? | Resolved |
| Q2 | Boundary | evidence-driven | Does this require public API or schema changes? | Resolved |
| Q3 | Acceptance | evidence-driven | Which current assets define deterministic acceptance? | Resolved |
| Q4 | Technical | evidence-driven | Where should prompt version metadata live with minimal implementation risk? | Resolved |
| Q5 | Scope | user-interview | Should live E2E be mandatory for all scenarios? | Confirmed by objective as conditional |
## Evidence-driven Conclusions
- Q1 conclusion: use `prompt_audit` for prompt version metadata and keep existing `gatekeeper_result.rule_set_version`.
- Q2 conclusion: keep this as an internal trace/self-evaluation contract change. Do not add endpoints, tables, or new Agent roles.
- Q3 conclusion: `DiagnosisTraceEvaluatorTest`, baseline reports, fixed fixtures, and demo scripts define current acceptance style.
- Q4 conclusion: add a small Chat prompt audit catalog near `ChatService` prompt loading and persist a compact snapshot with verifier evaluation.
- Q5 conclusion: run live E2E with `mvp-demo` profile if dependencies are available; otherwise record the blocker and rely on deterministic eval/unit evidence.
## Specify / Alignment
| Check | Status | Notes |
|---|---|---|
| proposal goals -> proposal | Aligned | Proposal covers demo preflight, eval expansion, prompt audit, and docs. |
| proposal scope/constraints -> design | Aligned | Design records no public API/schema changes, prompt audit shape, eval fields, and demo script behavior. |
| design decisions -> specs/tasks | Aligned | Specs cover persisted prompt audit, evaluator checks, baseline, and demo script outputs. |
| specs observable behavior -> tasks | Aligned | Each requirement has implementation and verification tasks. |
## Audit
Input -> processing -> output chain:
```text
prompt resource metadata
-> ChatService / PromptAudit snapshot
-> verifier_evaluation.prompt_audit
-> Trace API / eval fixtures
-> DiagnosisTraceEvaluator baseline
run-interview-demo-check.ps1
-> service readiness
-> chat / trace / feedback
-> mvp/demo/output summary
```
Architecture risk assessment:
1. The audit shape is intentionally compact and internal; storing full prompt text would create noisy traces and possible sensitive-content risk.
2. Eval should assert versions by explicit metadata, not by prompt content hashes that churn during local prompt edits.
3. Live demo checks may still be LOW_CONFID because LLM output is not deterministic; deterministic fixtures remain the regression source of truth.
4. No devflow/OpenSpec conflict found.
## Commit Gate
- `openspec validate interview-demo-quality-audit --strict`: passed.
- File completeness:
- `proposal.md`: present.
- `design.md`: present.
- `specs/`: present for `chat-verifier-agent`, `diagnosis-eval-harness`, and `mvp-demo-trace-acceptance`.
- `tasks.md`: present.
- Consistency:
- Proposal goals map to design sections.
- Design decisions map to spec requirements and executable tasks.
- Task acceptance checks are verifiable.
- `.committed` marker created.
## Current Checkpoint
- Commit completed.
- Apply is authorized by the original objective: "完成后归档提交".
## Pre-apply Research
- Capability source: sm-flow built-in apply protocol. `openspec-apply-change` was not invoked directly in this session.
- Repository semantic search/LSP note: the requested `codebase-retrieval` and LSP tools were not available in the exposed toolset, so impact analysis used `rg`, direct file reads, OpenSpec/devflow artifacts, and targeted tests.
- Reference implementation and reuse:
- `ChatService.persistVerifierEvaluation(...)` is the single persistence point for Chat verifier/composer audit data; prompt audit was added there to cover normal, fallback, and degraded Composer paths.
- `DiagnosisTraceEvaluator` and `DiagnosisEvalReportWriter` are the deterministic eval extension points; no LLM judge was introduced.
- `mvp/demo/scripts/run-payment-timeout-demo.ps1` provided the request/trace/feedback flow reused by the new interview preflight script.
- Interface impact remains L2 internal trace contract: `verifier_evaluation.prompt_audit` and eval report fields are added; no public endpoint, table, or request DTO changed.
## Apply Notes
- Added compact Chat prompt audit metadata: `chat-prompts-v1`, with planner/executor/verifier/composer prompt versions and resource paths.
- Extended diagnosis eval schema, result reporting, baseline fixtures, JSON report, and Markdown report for Prompt audit and Gatekeeper rule metadata.
- Added two fixture-backed audit cases:
- `prompt-gatekeeper-audit-closure`
- `audit-metadata-low-confid`
- Added `mvp/demo/scripts/run-interview-demo-check.ps1` to run service readiness, Chat, Trace, feedback, and summary output.
- Updated MVP demo/eval/architecture docs to explain `prompt_audit.version`, `gatekeeper_result.rule_set_version`, and deterministic fixture baseline.
@@ -0,0 +1,58 @@
# Evidence
## Context Files Read
- `devflow/index.md`
- `devflow/glossary/CONTEXT.md`
- `devflow/projects/2026-07-08-diagnosis-eval-demo-gatekeeper-closure/decisions.md`
- `devflow/projects/2026-07-08-executor-composer-final-answer/decisions.md`
- `mvp/architecture/current-mvp-architecture.md`
- `mvp/architecture/agent-orchestration.md`
- `mvp/architecture/executor-evidence-pipeline-refactor.md`
- `mvp/architecture/harness-quality-gates.md`
- `mvp/demo/README.md`
- `mvp/demo/ten-minute-interview-demo.md`
- `mvp/eval/README.md`
- `mvp/eval/cases/diagnosis-cases.json`
- `src/main/java/com/superbiz/agent/service/ChatService.java`
- `src/main/java/com/superbiz/agent/service/ExecutorGatekeeperService.java`
- `src/main/java/com/superbiz/agent/eval/DiagnosisTraceEvaluator.java`
- `src/main/resources/gatekeeper/gatekeeper-rules.json`
## Tooling Note
The required `codebase-retrieval` and LSP tools were not exposed in this session. Impact analysis used `rg`, direct file reads, existing OpenSpec/devflow artifacts, and targeted tests instead.
## Implementation Evidence
- `src/main/java/com/superbiz/agent/service/ChatService.java`
- Adds `prompt_audit` under `verifier_evaluation` through the shared `persistVerifierEvaluation(...)` path.
- Uses compact metadata only: audit version, prompt names, prompt versions, and resource paths.
- `src/main/java/com/superbiz/agent/eval/DiagnosisTraceEvaluator.java`
- Adds deterministic checks for `requirePromptAudit`, `expectedPromptAuditVersion`, `expectedPromptVersions`, and `requireGatekeeperRules`.
- `src/main/java/com/superbiz/agent/eval/DiagnosisEvalReportWriter.java`
- Adds Prompt Audit and Gatekeeper rule count columns to Markdown reports.
- `mvp/eval/cases/diagnosis-cases.json`
- Expands fixed baseline to 12 fixture-backed cases.
- `mvp/eval/fixtures/prompt-gatekeeper-audit-closure-pass.json`
- Positive PASS fixture proving Prompt audit and Gatekeeper rule metadata closure.
- `mvp/eval/fixtures/audit-metadata-low-confid.json`
- LOW_CONFID fixture proving safe answer behavior while audit metadata remains present.
- `mvp/demo/scripts/run-interview-demo-check.ps1`
- Adds service readiness, Chat, Trace, feedback, and summary output for interview preflight.
## Verification Evidence
- OpenSpec:
- `openspec validate interview-demo-quality-audit --strict`: passed before archive.
- `openspec validate --specs --strict`: 10 specs passed after merging deltas into main specs.
- Unit/eval:
- `mvn -q "-Dtest=DiagnosisTraceEvaluatorTest" test`: passed.
- `mvn -q "-Dtest=ChatServiceSequentialAgentTest" test`: passed.
- `mvn -q "-Dtest=DiagnosisTraceEvaluatorTest,DiagnosisEvalBaselineDiffTest,ChatServiceSequentialAgentTest,ExecutorGatekeeperServiceTest,VerifierInputHookTest" test`: passed.
- Compile:
- `mvn -q -DskipTests compile`: passed.
- E2E:
- Started `mvn spring-boot:run -Dspring-boot.run.profiles=mvp-demo`.
- Ran `mvp/demo/scripts/run-interview-demo-check.ps1` against `http://localhost:9900`.
- Summary recorded `chatSuccess=true`, `verdict=LOW_CONFID`, `gatekeeperRuleSetVersion=gatekeeper-rules-v1`, and `promptAuditVersion=chat-prompts-v1`.
+22 -15
View File
@@ -1,8 +1,8 @@
# SuperBizAgent MVP 文档
**更新日期**:2026-07-05
**更新日期**:2026-07-09
本目录保存 MVP 阶段的架构、问题、演示、评测和数据表说明。当前架构入口已经整理到 `mvp/architecture/`,旧版架构材料已归档,避免继续把历史方案当成当前实现。
本目录保存 MVP 阶段的架构、问题、演示、评测和数据表说明。当前材料按“当前入口”和“历史归档”拆开,避免把早期设计稿当成当前实现。
## 当前入口
@@ -12,6 +12,7 @@
| [architecture/current-mvp-architecture.md](architecture/current-mvp-architecture.md) | 当前可运行系统架构 |
| [architecture/interview-one-pager.md](architecture/interview-one-pager.md) | 面试一页式架构讲解 |
| [architecture/agent-orchestration.md](architecture/agent-orchestration.md) | Agent 编排架构 |
| [architecture/executor-evidence-pipeline-refactor.md](architecture/executor-evidence-pipeline-refactor.md) | Executor 证据链路改造记录 |
| [architecture/harness-quality-gates.md](architecture/harness-quality-gates.md) | Harness 与质量门禁 |
| [architecture/rag-architecture.md](architecture/rag-architecture.md) | RAG/知识检索新架构 |
| [architecture/retrieval-observability.md](architecture/retrieval-observability.md) | 检索与可观测性架构 |
@@ -20,11 +21,12 @@
| [architecture/knowledge-base-authoring.md](architecture/knowledge-base-authoring.md) | 知识库文档编写与维护 |
| [architecture/data-model.md](architecture/data-model.md) | 数据模型总览 |
| [architecture/evolution-roadmap.md](architecture/evolution-roadmap.md) | Agent 架构演进路线 |
| [issues/rag-refactor-plan.md](issues/rag-refactor-plan.md) | RAG 重构计划和阶段拆解 |
| [issues/README.md](issues/README.md) | MVP issue 索引 |
| [issues/active/rag-refactor-plan.md](issues/active/rag-refactor-plan.md) | RAG 重构计划和阶段拆解 |
| [tables/README.md](tables/README.md) | 当前 MySQL 表说明 |
| [demo/README.md](demo/README.md) | Demo 运行和面试演示材料 |
| [demo/ten-minute-interview-demo.md](demo/ten-minute-interview-demo.md) | 10 分钟面试演示脚本 |
| [eval/README.md](eval/README.md) | 诊断评测材料 |
| [issues/README.md](issues/README.md) | MVP issue 索引 |
## 当前系统一句话
@@ -39,6 +41,7 @@ mvp/
current-mvp-architecture.md
interview-one-pager.md
agent-orchestration.md
executor-evidence-pipeline-refactor.md
harness-quality-gates.md
rag-architecture.md
retrieval-observability.md
@@ -47,12 +50,17 @@ mvp/
knowledge-base-authoring.md
data-model.md
evolution-roadmap.md
archive/2026-07-05-legacy/
archive/
issues/
README.md
rag-refactor-plan.md
ISS-*.md
rag-*.md
active/
archived/
design-notes/
rag/
tables/
README.md
*表-*.md
archive/
demo/
README.md
ten-minute-interview-demo.md
@@ -65,9 +73,7 @@ mvp/
cases/
fixtures/
reports/
notes/
plan/
tables/
archive/
```
## 当前核心设计
@@ -107,10 +113,11 @@ RAG
-> tool_invocation
```
## 旧文档说明
## 归档说明
旧版架构文档已移动到:
历史材料分两类:
- [architecture/archive/2026-07-05-legacy/](architecture/archive/2026-07-05-legacy/)
- 旧架构文档:[architecture/archive/2026-07-05-legacy/](architecture/archive/2026-07-05-legacy/)
- 本次文档清理归档:[archive/2026-07-09-doc-cleanup/](archive/2026-07-09-doc-cleanup/)
归档文档只用于追溯设计历史。当前实现和后续规划以 `architecture/current-mvp-architecture.md` 与 `architecture/rag-architecture.md` 为准。
归档文档只用于追溯设计历史。当前实现和后续规划以 `architecture/`、`issues/README.md`、`tables/README.md` 和 OpenSpec/devflow 的最新记录为准。
@@ -154,4 +154,4 @@ ALTER TABLE diagnosis_session ADD COLUMN answer LONGTEXT COMMENT 'Agent 返回
- **LLM 观点层**:在 `selfEvaluation` 的 `llm_opinion` 字段叠加 LLM 结构化观点(has_root_cause、has_solution 等),作为独立 factors,不改变现有规则逻辑
- **案例结构化字段**:useful 触发时自动提取 faultCategory / errorCode,替代暂时的 GENERAL
- **重复召回问题**:Executor Prompt 约束或工具层 session 维度去重(见 [ISS-001](../issues/ISS-001-duplicate-retrieval.md))
- **重复召回问题**:Executor Prompt 约束或工具层 session 维度去重(见 [ISS-001](../../../issues/archived/ISS-001-duplicate-retrieval.md))
+19 -2
View File
@@ -82,6 +82,23 @@ Prompt 层当前承担的门禁:
- Chat Composer 不允许补事实,尤其不能把 `$.no_evidence` 表达为“已排除/确认没有”。
- AIOps payload 模式必须聚焦输入告警。
Chat 链路还会在 `verifier_evaluation.prompt_audit` 中持久化紧凑 Prompt 审计快照:
```json
{
"version": "chat-prompts-v1",
"prompts": [
{
"name": "chat_executor",
"version": "chat-executor-v2",
"resource": "prompts/chat-executor-prompt.md"
}
]
}
```
该快照只保存版本和资源路径,不保存完整 Prompt 文本。它用于面试演示、trace 回放和离线 baseline 解释“本次诊断使用了哪套 Prompt 契约”。
## 4. Trace Hooks
`AgentLoggingHook` 是当前 Agent step 可观测性的核心。
@@ -196,7 +213,7 @@ Verifier 不再逐字核验 excerpt 真伪;这由 Gatekeeper 完成。Verifier
diagnosis_session.self_evaluation.verifier_evaluation
```
其中同时持久化 `executor_structured_output`、`gatekeeper_result`、`tool_trace_summary` 和 `composer_output`,用于 Trace 回放。
其中同时持久化 `executor_structured_output`、`gatekeeper_result`、`tool_trace_summary`、`prompt_audit` 和 `composer_output`,用于 Trace 回放。
## 7. AIOps 规则门禁
@@ -233,7 +250,7 @@ diagnosis_session.self_evaluation.aiops_rule_evaluation
- 同一工具调用次数上限。
- 工具超时的统一熔断。
- Gatekeeper 规则远程化或三层分离:索引层、元数据层、规则实现层。
- Prompt 版本记录和回滚。
- Prompt 版本回滚和更细粒度变更审计。
- Verifier 对 AIOps 报告的 LLM 级事实校验。
这些应在评测集扩大后逐步加入,避免一次性把诊断流程卡得过死。
+1 -1
View File
@@ -2,7 +2,7 @@
**更新日期**:2026-07-06
**状态**:当前主架构 + 后续演进边界
**关联计划**:`mvp/issues/rag-refactor-plan.md`
**关联计划**:[`mvp/issues/active/rag-refactor-plan.md`](../issues/active/rag-refactor-plan.md)
## 1. 架构目标
@@ -0,0 +1,17 @@
# 2026-07-09 MVP 文档清理归档
本目录保存本次清理中从当前入口移出的历史设计材料。这些文档仍有追溯价值,但不再代表当前可运行实现。
## 归档内容
| 目录 | 内容 | 归档原因 |
|---|---|---|
| `discuss/` | 早期 Executor Prompt、L0、RAG 讨论稿 | 已被当前 architecture、OpenSpec change 和 issue 取代 |
| `plan/` | `session-storage-design.md` | 会话存储已实现,当前表以 Flyway 和 `mvp/tables/` 为准 |
| `notes/` | 早期工程决策和 Demo Trace 验收笔记 | 相关内容已沉淀到 architecture、demo、eval 和 devflow |
## 使用原则
- 当前架构以 `mvp/architecture/` 为准。
- 当前表结构以 `mvp/tables/`、Flyway migration 和实体类为准。
- 当前问题入口以 `mvp/issues/README.md` 为准。
+8 -3
View File
@@ -8,7 +8,9 @@
- `interview-walkthrough.md`:面试讲解话术。
- `evidence-pipeline-scenarios.md`:PASS / LOW_CONFID / REJECT / no-evidence 场景矩阵。
- `trace-inspection-checklist.md`:Trace 字段检查清单。
- `scripts/run-interview-demo-check.ps1`:面试预检脚本,包含服务可达性、Chat、Trace、反馈和 summary 输出。
- `scripts/run-payment-timeout-demo.ps1`:本地可执行 Demo 脚本。
- `interview-q-and-a.md`:面试追问回答,覆盖 Agent 工程取舍、审计和评测。
- `requests/payment-timeout-chat.json`:固定 Chat 请求 payload。
- `requests/narrow-highcpu-chat.json`:窄范围正向观察请求。
- `requests/hikari-no-evidence-chat.json`:no-evidence 负向观察请求。
@@ -37,7 +39,7 @@ http://localhost:9900
最快方式:
```powershell
powershell -ExecutionPolicy Bypass -File mvp/demo/scripts/run-payment-timeout-demo.ps1
powershell -ExecutionPolicy Bypass -File mvp/demo/scripts/run-interview-demo-check.ps1
```
脚本会生成:
@@ -46,6 +48,7 @@ powershell -ExecutionPolicy Bypass -File mvp/demo/scripts/run-payment-timeout-de
mvp/demo/output/chat-response.json
mvp/demo/output/trace-response.json
mvp/demo/output/feedback-response.json
mvp/demo/output/interview-demo-summary.json
```
手动请求:
@@ -85,6 +88,8 @@ Invoke-RestMethod `
- `data.steps` 包含 planner / executor / verifier 等步骤
- `data.toolInvocations` 包含 `lookup_knowledge`、`query_logs`、`query_metrics` 等证据工具
- `data.session.selfEvaluation` 包含 verifier 或 rule evaluation
- Chat V2 链路中,`data.session.selfEvaluation.verifier_evaluation.prompt_audit.version` 记录 Chat Prompt 审计版本
- Chat V2 链路中,`data.session.selfEvaluation.verifier_evaluation.gatekeeper_result.rule_set_version` 记录 Gatekeeper 规则集版本
## 5. 提交反馈
@@ -176,8 +181,8 @@ AIOps 主线:
面试时不要把所有安全场景都压到 live LLM 现场表现上。建议使用:
- `scripts/run-payment-timeout-demo.ps1` 跑主路径。
- `scripts/run-interview-demo-check.ps1` 跑主路径和预检 summary。
- `evidence-pipeline-scenarios.md` 讲解 PASS / LOW_CONFID / REJECT / no-evidence 矩阵。
- `mvp/eval/reports/baseline-report.md` 证明固定 fixture 10/10 通过。
- `mvp/eval/reports/baseline-report.md` 证明固定 fixture 12/12 通过。
这样可以同时展示真实链路和确定性回归能力。
+33
View File
@@ -0,0 +1,33 @@
# 面试追问 Q&A
## 为什么不用普通 Chatbot?
这个项目的重点不是生成一段诊断文本,而是把诊断拆成可审计链路:Planner 拆解问题,Executor 调工具拿证据,Gatekeeper 用代码核验证据引用,Verifier 判断可推导性,Composer 生成最终表达。每次运行都能通过同一个 `sessionId` 回放。
## 为什么 RAG 要做成显式工具?
`lookup_knowledge` 保持显式工具调用,才能在 `tool_invocation` 里看到 Agent 查了什么、命中了什么、相关性等级是什么,以及最终答案是否真的使用了这些证据。隐式 Advisor 更方便,但不利于审计 Agent 决策。
## 怎么防止 Executor 幻觉?
Executor 不直接负责最终用户答案,而是输出 `executor_evidence_v2` 的微观事实和证据引用。Gatekeeper 会校验 `source_invocation_id`、`raw_path`、`evidence_excerpt` 是否真实存在;Verifier 再判断 claim 是否能由已验真的证据推出;Composer 只表达 Verifier 允许输出的内容。
## LOW_CONFID 是失败吗?
不是。`LOW_CONFID` 表示当前证据不足以支撑强结论,但系统仍然可以安全表达已确认事实和缺失信息。面试时可以把它作为“没有证据就不强答”的质量门禁,而不是模型能力失败。
## Prompt 改了怎么审计?
Chat verifier evaluation 里会记录 `prompt_audit.version`,并列出 planner、executor、verifier、composer 的 Prompt 版本和资源路径。它不保存完整 Prompt 文本,只保留用于回放和回归解释的紧凑元数据。
## Gatekeeper 改了怎么审计?
Gatekeeper 结果里记录 `gatekeeper_result.rule_set_version` 和已启用规则元数据摘要。规则执行仍是确定性 Java 代码,版本和规则元数据用于解释“这次引用验真用的是哪套规则”。
## 为什么现在不拆 SubAgent?
当前 MVP 的主要风险不是 Agent 数量不够,而是证据、验证和回归是否稳定。文档里的演进路线把 SubAgent 放在 P2:等故障类型、工具权限和评测集足够明确后再拆,避免只是移动复杂度。
## 为什么 baseline 比 live demo 更重要?
live demo 证明链路在当前环境能跑通,但 LLM 和外部依赖会波动。`mvp/eval` 的固定 fixture baseline 是确定性回归来源,用来判断 Prompt、工具、Gatekeeper、Verifier 或 Composer 的改动有没有让系统退化。
@@ -0,0 +1,161 @@
param(
[string]$BaseUrl = "http://localhost:9900",
[string]$SessionId = "mvp-demo-interview-payment-timeout-001",
[string]$RequestFile = "$PSScriptRoot/../requests/payment-timeout-chat.json",
[string]$OutputDir = "$PSScriptRoot/../output"
)
$ErrorActionPreference = "Stop"
function Test-ServiceReachable {
param([string]$Url)
try {
$request = [System.Net.WebRequest]::Create($Url)
$request.Method = "GET"
$request.Timeout = 5000
$response = $request.GetResponse()
$response.Close()
return $true
} catch [System.Net.WebException] {
if ($_.Exception.Response -ne $null) {
$_.Exception.Response.Close()
return $true
}
return $false
}
}
function Get-TraceData {
param($TraceResponse)
if ($TraceResponse.PSObject.Properties.Name -contains "data") {
return $TraceResponse.data
}
return $TraceResponse
}
function Get-SelfEvaluation {
param($TraceData)
if ($null -eq $TraceData -or $null -eq $TraceData.session) {
return $null
}
return $TraceData.session.selfEvaluation
}
function Get-ToolNames {
param($TraceData)
if ($null -eq $TraceData -or $null -eq $TraceData.toolInvocations) {
return @()
}
return @($TraceData.toolInvocations | ForEach-Object { $_.toolName } | Where-Object { $_ } | Sort-Object -Unique)
}
New-Item -ItemType Directory -Force -Path $OutputDir | Out-Null
Write-Host "Running interview demo preflight..."
Write-Host "BaseUrl: $BaseUrl"
Write-Host "SessionId: $SessionId"
if (-not (Test-ServiceReachable -Url $BaseUrl)) {
throw "Service is not reachable: $BaseUrl. Start the app with mvp-demo profile first: mvn spring-boot:run -Dspring-boot.run.profiles=mvp-demo"
}
$request = Get-Content -Raw -Encoding UTF8 -Path $RequestFile | ConvertFrom-Json
$request.Id = $SessionId
$body = $request | ConvertTo-Json -Depth 8
$chatRequest = @{
Method = "Post"
Uri = "$BaseUrl/api/chat"
ContentType = "application/json; charset=utf-8"
Body = $body
}
$chat = Invoke-RestMethod @chatRequest
$chatPath = Join-Path $OutputDir "chat-response.json"
$chat | ConvertTo-Json -Depth 30 | Set-Content -Encoding UTF8 -Path $chatPath
$traceRequest = @{
Method = "Get"
Uri = "$BaseUrl/api/diagnosis/$SessionId/trace"
}
$trace = Invoke-RestMethod @traceRequest
$tracePath = Join-Path $OutputDir "trace-response.json"
$trace | ConvertTo-Json -Depth 80 | Set-Content -Encoding UTF8 -Path $tracePath
$feedbackBody = @{
sessionId = $SessionId
feedback = "useful"
} | ConvertTo-Json
$feedbackRequest = @{
Method = "Post"
Uri = "$BaseUrl/api/feedback"
ContentType = "application/json; charset=utf-8"
Body = $feedbackBody
}
$feedback = Invoke-RestMethod @feedbackRequest
$feedbackPath = Join-Path $OutputDir "feedback-response.json"
$feedback | ConvertTo-Json -Depth 30 | Set-Content -Encoding UTF8 -Path $feedbackPath
$traceData = Get-TraceData -TraceResponse $trace
$selfEvaluation = Get-SelfEvaluation -TraceData $traceData
$verifierEvaluation = $null
if ($null -ne $selfEvaluation) {
$verifierEvaluation = $selfEvaluation.verifier_evaluation
}
$gatekeeperResult = $null
$promptAudit = $null
if ($null -ne $verifierEvaluation) {
$gatekeeperResult = $verifierEvaluation.gatekeeper_result
$promptAudit = $verifierEvaluation.prompt_audit
}
$verdict = $null
$gatekeeperStatus = $null
$gatekeeperRuleSetVersion = $null
$promptAuditVersion = $null
if ($null -ne $verifierEvaluation) {
$verdict = $verifierEvaluation.verdict
}
if ($null -ne $gatekeeperResult) {
$gatekeeperStatus = $gatekeeperResult.status
$gatekeeperRuleSetVersion = $gatekeeperResult.rule_set_version
}
if ($null -ne $promptAudit) {
$promptAuditVersion = $promptAudit.version
}
$toolNames = Get-ToolNames -TraceData $traceData
$summaryPath = Join-Path $OutputDir "interview-demo-summary.json"
$summary = [ordered]@{
sessionId = $SessionId
baseUrl = $BaseUrl
chatSuccess = $chat.data.success
verdict = $verdict
gatekeeperStatus = $gatekeeperStatus
gatekeeperRuleSetVersion = $gatekeeperRuleSetVersion
promptAuditVersion = $promptAuditVersion
toolNames = $toolNames
paths = [ordered]@{
chat = $chatPath
trace = $tracePath
feedback = $feedbackPath
summary = $summaryPath
}
}
$summary | ConvertTo-Json -Depth 20 | Set-Content -Encoding UTF8 -Path $summaryPath
Write-Host ""
Write-Host "Interview demo preflight completed."
Write-Host "Verdict: $($summary.verdict)"
Write-Host "Gatekeeper rules: $($summary.gatekeeperRuleSetVersion)"
Write-Host "Prompt audit: $($summary.promptAuditVersion)"
Write-Host "Summary: $summaryPath"
+9 -1
View File
@@ -37,7 +37,7 @@ http://localhost:9900
推荐使用固定脚本:
```powershell
powershell -ExecutionPolicy Bypass -File mvp/demo/scripts/run-payment-timeout-demo.ps1
powershell -ExecutionPolicy Bypass -File mvp/demo/scripts/run-interview-demo-check.ps1
```
脚本会写出:
@@ -46,6 +46,7 @@ powershell -ExecutionPolicy Bypass -File mvp/demo/scripts/run-payment-timeout-de
mvp/demo/output/chat-response.json
mvp/demo/output/trace-response.json
mvp/demo/output/feedback-response.json
mvp/demo/output/interview-demo-summary.json
```
现场话术:
@@ -98,6 +99,8 @@ data.toolInvocations[*].outputPreview
data.toolInvocations[*].retrievalLayer
data.toolInvocations[*].relevanceLevel
data.summary.hasVerifierEvaluation
data.session.selfEvaluation.verifier_evaluation.prompt_audit.version
data.session.selfEvaluation.verifier_evaluation.gatekeeper_result.rule_set_version
```
现场话术:
@@ -155,6 +158,10 @@ Verifier 不做新检索,只看工具 trace 汇总。
如果 PASS,就输出原答案。
如果 LOW_CONFID,可以补证据或加低置信提示。
如果 REJECT,就降级输出,只保留已确认信息。
Prompt 和 Gatekeeper 的版本也会进入 trace。
`prompt_audit.version` 用于说明本次 Chat 使用哪套 Prompt 契约,`gatekeeper_result.rule_set_version` 用于说明引用验真的规则版本。
固定 fixture baseline 是回归判断来源,live demo 主要证明当前环境链路可跑通。
```
## 7. 展示反馈闭环
@@ -226,6 +233,7 @@ AIOps 有两个模式。
mvp/demo/output/chat-response.json
mvp/demo/output/trace-response.json
mvp/demo/output/feedback-response.json
mvp/demo/output/interview-demo-summary.json
```
降级话术:
+3 -1
View File
@@ -1,6 +1,6 @@
# Trace 检查清单
运行 `scripts/run-payment-timeout-demo.ps1` 后,用这份清单检查 `trace-response.json`。
运行 `scripts/run-interview-demo-check.ps1` 后,用这份清单检查 `trace-response.json` 和 `interview-demo-summary.json`。
## 1. Session
@@ -11,6 +11,8 @@
| `data.session.answer` | 是否包含最终诊断答案 | 最终答案没有脱离 Trace |
| `data.session.selfEvaluation` | 是否包含 verifier 或 rule evaluation | 答案经过质量门,不只是模型原始输出 |
| `data.session.selfEvaluation.verifier_evaluation.gatekeeper_result.rule_set_version` | 如果是 Chat V2 链路,是否记录 Gatekeeper 规则版本 | 安全规则可审计、可回归 |
| `data.session.selfEvaluation.verifier_evaluation.prompt_audit.version` | 如果是 Chat V2 链路,是否记录 Prompt 审计版本 | Prompt 变更可解释、可回归 |
| `data.session.selfEvaluation.verifier_evaluation.prompt_audit.prompts[*].version` | 是否记录 planner / executor / verifier / composer 版本 | 便于定位 Prompt 变更影响 |
| `data.session.feedback` | 提交反馈后是否变为 `useful` | 用户反馈挂在同一次诊断上 |
## 2. Agent 步骤
+8 -4
View File
@@ -29,10 +29,10 @@ The baseline evaluates saved trace fixtures. It does not start the application a
The committed baseline currently contains:
```text
10 fixed cases
10 passing fixture evaluations
4 PASS verdicts
5 LOW_CONFID verdicts
12 fixed cases
12 passing fixture evaluations
5 PASS verdicts
6 LOW_CONFID verdicts
1 REJECT verdict
```
@@ -44,6 +44,8 @@ The V2 evidence-pipeline matrix covers:
- Unsupported claim filtering before the final answer.
- Composer fallback rendering without raw Executor JSON leakage.
- Gatekeeper rule set version audit for new matrix fixtures.
- Prompt audit version checks for planner, executor, verifier, and composer prompts.
- Gatekeeper rule metadata checks for enabled rule id and default severity.
## Verification
@@ -80,6 +82,8 @@ Stage 5 adds these V2 checks:
- `claim_checks` must be structurally auditable.
- Composer output must record whether normal parsing or fallback rendering was used.
- Gatekeeper rule set version can be asserted per fixture.
- Prompt audit version and per-prompt versions can be asserted per fixture.
- Gatekeeper rule metadata can be required per fixture.
- Final answers must not leak raw Executor protocol markers such as `executor_evidence_v2`, `answer_version`, `evidence_bindings`, or `claim_id`.
- Configured unsupported claim keywords must not appear as confirmed final-answer content.
+54
View File
@@ -17,6 +17,33 @@
"expectedComposerStatuses": ["valid"],
"forbiddenConfirmedClaimKeywords": ["数据库连接池"]
},
{
"id": "prompt-gatekeeper-audit-closure",
"title": "Prompt and Gatekeeper audit closure",
"question": "确认 payment-service 当前是否存在 HighCPUUsage 告警,并检查审计元数据是否完整。",
"traceFixture": "prompt-gatekeeper-audit-closure-pass.json",
"expectedRootCauseKeywords": ["payment-service", "HighCPUUsage", "92%"],
"minKeywordMatches": 2,
"requiredEvidenceTools": ["query_metrics"],
"allowedVerdicts": ["PASS"],
"forbiddenAnswerKeywords": ["根因", "修复建议", "通常情况下"],
"requireV2AuditClosure": true,
"requireClaimChecks": true,
"requireComposerOutput": true,
"expectedGatekeeperStatuses": ["pass"],
"expectedGatekeeperRuleSetVersion": "gatekeeper-rules-v1",
"expectedComposerStatuses": ["valid"],
"forbiddenConfirmedClaimKeywords": ["数据库连接池"],
"requirePromptAudit": true,
"expectedPromptAuditVersion": "chat-prompts-v1",
"expectedPromptVersions": {
"chat_planner": "chat-planner-v1",
"chat_executor": "chat-executor-v2",
"chat_verifier": "chat-verifier-v2",
"chat_composer": "chat-composer-v1"
},
"requireGatekeeperRules": true
},
{
"id": "hikari-no-evidence-negative-observation",
"title": "Hikari no-evidence negative observation",
@@ -124,6 +151,33 @@
"expectedComposerStatuses": ["valid"],
"forbiddenConfirmedClaimKeywords": ["主库故障"]
},
{
"id": "audit-metadata-low-confid",
"title": "Audit metadata low confidence",
"question": "订单超时是否可以确认由数据库主库故障导致,并检查审计元数据是否完整?",
"traceFixture": "audit-metadata-low-confid.json",
"expectedRootCauseKeywords": ["超时", "证据"],
"minKeywordMatches": 2,
"requiredEvidenceTools": ["query_logs"],
"allowedVerdicts": ["LOW_CONFID"],
"forbiddenAnswerKeywords": ["已经确认"],
"requireV2AuditClosure": true,
"requireClaimChecks": true,
"requireComposerOutput": true,
"expectedGatekeeperStatuses": ["pass"],
"expectedGatekeeperRuleSetVersion": "gatekeeper-rules-v1",
"expectedComposerStatuses": ["valid"],
"forbiddenConfirmedClaimKeywords": ["主库故障"],
"requirePromptAudit": true,
"expectedPromptAuditVersion": "chat-prompts-v1",
"expectedPromptVersions": {
"chat_planner": "chat-planner-v1",
"chat_executor": "chat-executor-v2",
"chat_verifier": "chat-verifier-v2",
"chat_composer": "chat-composer-v1"
},
"requireGatekeeperRules": true
},
{
"id": "composer-fallback-no-raw-json",
"title": "Composer fallback no raw JSON",
@@ -0,0 +1,192 @@
{
"session": {
"sessionId": "eval-audit-metadata-low-confid",
"query": "订单超时是否可以确认由数据库主库故障导致,并检查审计元数据是否完整?",
"status": "SUCCESS",
"agentFlow": "CHAT",
"totalDurationMs": 45000,
"toolCallCount": 1,
"answer": "以下结论基于当前已获取证据,仍存在部分证据缺口,请谨慎参考。\n\n已确认信息:日志显示订单接口出现超时。\n\n仍需补充信息:目前没有数据库故障日志或主库状态证据,不能把该方向写成确认根因。",
"selfEvaluation": {
"verifier_evaluation": {
"verdict": "LOW_CONFID",
"groundedness_score": 0.42,
"critical_fact_count": 2,
"prompt_audit": {
"version": "chat-prompts-v1",
"prompts": [
{
"name": "chat_planner",
"version": "chat-planner-v1",
"resource": "prompts/chat-planner-prompt.md"
},
{
"name": "chat_executor",
"version": "chat-executor-v2",
"resource": "prompts/chat-executor-prompt.md"
},
{
"name": "chat_verifier",
"version": "chat-verifier-v2",
"resource": "prompts/chat-verifier-prompt.md"
},
{
"name": "chat_composer",
"version": "chat-composer-v1",
"resource": "prompts/chat-composer-prompt.md"
}
]
},
"gatekeeper_result": {
"status": "pass",
"severity": "none",
"rule_set_version": "gatekeeper-rules-v1",
"rules": [
{
"id": "evidence.invocation",
"description": "source_invocation_id must reference an existing tool invocation",
"enabled": true,
"default_severity": "reject"
},
{
"id": "evidence.excerpt",
"description": "evidence_excerpt must be supported by recorded evidence text",
"enabled": true,
"default_severity": "reject"
}
],
"checked_bindings": [
{
"claim_id": "claim-timeout",
"tool_name": "query_logs",
"source_invocation_id": 22,
"raw_path": "$.logs[0]",
"matched_text": "order api timeout",
"status": "pass"
}
],
"failed_rules": [],
"warnings": [],
"errors": []
},
"executor_structured_output": {
"answer_version": "executor_evidence_v2",
"claims": [
{
"claim_id": "claim-timeout",
"claim_type": "symptom",
"claim_text": "订单接口出现超时",
"support_level": "direct",
"evidence_bindings": [
{
"source_type": "tool_trace",
"tool_name": "query_logs",
"source_invocation_id": 22,
"raw_path": "$.logs[0]",
"evidence_excerpt": "order api timeout"
}
]
},
{
"claim_id": "claim-db-primary",
"claim_type": "root_cause",
"claim_text": "数据库主库故障导致订单超时",
"support_level": "weak",
"evidence_bindings": [
{
"source_type": "tool_trace",
"tool_name": "query_logs",
"source_invocation_id": 22,
"raw_path": "$.logs[0]",
"evidence_excerpt": "order api timeout"
}
]
}
],
"hypotheses": [],
"recommended_actions": [
{
"action_text": "补充查询数据库主库状态和错误日志",
"reason": "当前只有订单接口超时日志"
}
],
"missing_info": ["数据库主库状态", "数据库错误日志"]
},
"claim_checks": [
{
"claim_id": "claim-timeout",
"claim_text": "订单接口出现超时",
"claim_type": "symptom",
"verification": "direct_observation",
"detail": "日志直接记录 order api timeout",
"evidence_refs": [
{
"source_invocation_id": 22,
"raw_path": "$.logs[0]"
}
]
},
{
"claim_id": "claim-db-primary",
"claim_text": "数据库主库故障导致订单超时",
"claim_type": "root_cause",
"verification": "unsupported",
"detail": "日志只能证明订单接口超时,不能证明数据库主库故障",
"evidence_refs": [
{
"source_invocation_id": 22,
"raw_path": "$.logs[0]"
}
]
}
],
"facts_checked": [],
"composer_output": {
"status": "valid",
"answer_summary": "日志显示订单接口超时,但数据库方向证据不足。",
"recommended_actions": [
{
"action_text": "补充查询数据库主库状态和错误日志",
"reason": "当前只有订单接口超时日志"
}
],
"user_facing_answer": "以下结论基于当前已获取证据,仍存在部分证据缺口,请谨慎参考。\n\n已确认信息:日志显示订单接口出现超时。\n\n仍需补充信息:目前没有数据库故障日志或主库状态证据,不能把该方向写成确认根因。"
},
"tool_trace_summary": [
{
"tool_name": "query_logs",
"success": true,
"evidence_level": "direct"
}
]
}
}
},
"steps": [],
"toolInvocations": [
{
"id": 22,
"sessionId": "eval-audit-metadata-low-confid",
"toolName": "query_logs",
"outputPreview": "order api timeout",
"retrievalDetails": {
"evidence_status": "supported",
"evidence_refs": [
{
"raw_path": "$.logs[0]",
"text": "order api timeout"
}
]
},
"success": true
}
],
"summary": {
"persistedStepCount": 3,
"returnedStepCount": 3,
"persistedToolCallCount": 1,
"returnedToolCallCount": 1,
"hasVerifierEvaluation": true,
"hasFeedback": false
}
}
@@ -0,0 +1,155 @@
{
"session": {
"sessionId": "eval-prompt-gatekeeper-audit-closure",
"query": "确认 payment-service 当前是否存在 HighCPUUsage 告警,并检查审计元数据是否完整。",
"status": "SUCCESS",
"agentFlow": "CHAT",
"totalDurationMs": 19000,
"toolCallCount": 1,
"answer": "payment-service 当前存在 HighCPUUsage 告警,CPU 使用率为 92%。",
"selfEvaluation": {
"verifier_evaluation": {
"verdict": "PASS",
"groundedness_score": 1.0,
"critical_fact_count": 1,
"prompt_audit": {
"version": "chat-prompts-v1",
"prompts": [
{
"name": "chat_planner",
"version": "chat-planner-v1",
"resource": "prompts/chat-planner-prompt.md"
},
{
"name": "chat_executor",
"version": "chat-executor-v2",
"resource": "prompts/chat-executor-prompt.md"
},
{
"name": "chat_verifier",
"version": "chat-verifier-v2",
"resource": "prompts/chat-verifier-prompt.md"
},
{
"name": "chat_composer",
"version": "chat-composer-v1",
"resource": "prompts/chat-composer-prompt.md"
}
]
},
"gatekeeper_result": {
"status": "pass",
"severity": "none",
"rule_set_version": "gatekeeper-rules-v1",
"rules": [
{
"id": "evidence.invocation",
"description": "source_invocation_id must reference an existing tool invocation",
"enabled": true,
"default_severity": "reject"
},
{
"id": "evidence.raw_path",
"description": "raw_path must exist in retrieval_details.evidence_refs",
"enabled": true,
"default_severity": "reject"
}
],
"checked_bindings": [
{
"claim_id": "claim-1",
"tool_name": "query_metrics",
"source_invocation_id": 21,
"raw_path": "$.alerts[0]",
"matched_text": "HighCPUUsage firing, service=payment-service, current=92%, duration=25m",
"status": "pass"
}
],
"failed_rules": [],
"warnings": [],
"errors": []
},
"executor_structured_output": {
"answer_version": "executor_evidence_v2",
"claims": [
{
"claim_id": "claim-1",
"claim_type": "observation",
"claim_text": "payment-service 当前存在 HighCPUUsage 告警,CPU 使用率为 92%。",
"support_level": "direct",
"evidence_bindings": [
{
"source_type": "tool_trace",
"tool_name": "query_metrics",
"source_invocation_id": 21,
"raw_path": "$.alerts[0]",
"evidence_excerpt": "HighCPUUsage firing, service=payment-service, current=92%, duration=25m"
}
]
}
],
"hypotheses": [],
"recommended_actions": [],
"missing_info": []
},
"claim_checks": [
{
"claim_id": "claim-1",
"claim_text": "payment-service 当前存在 HighCPUUsage 告警,CPU 使用率为 92%。",
"claim_type": "observation",
"verification": "direct_observation",
"detail": "已核验的指标证据直接包含服务名、告警名和 CPU 当前值。",
"evidence_refs": [
{
"source_invocation_id": 21,
"raw_path": "$.alerts[0]"
}
]
}
],
"facts_checked": [],
"composer_output": {
"status": "valid",
"answer_summary": "payment-service 当前存在 HighCPUUsage 告警。",
"recommended_actions": [],
"user_facing_answer": "payment-service 当前存在 HighCPUUsage 告警,CPU 使用率为 92%。"
},
"tool_trace_summary": [
{
"tool_name": "query_metrics",
"success": true,
"source_invocation_ids": [21],
"evidence_level": "direct"
}
]
}
}
},
"steps": [],
"toolInvocations": [
{
"id": 21,
"sessionId": "eval-prompt-gatekeeper-audit-closure",
"toolName": "query_metrics",
"outputPreview": "HighCPUUsage firing, service=payment-service, current=92%, duration=25m",
"retrievalDetails": {
"evidence_status": "supported",
"evidence_refs": [
{
"raw_path": "$.alerts[0]",
"text": "HighCPUUsage firing, service=payment-service, current=92%, duration=25m"
}
]
},
"success": true
}
],
"summary": {
"persistedStepCount": 3,
"returnedStepCount": 3,
"persistedToolCallCount": 1,
"returnedToolCallCount": 1,
"hasVerifierEvaluation": true,
"hasFeedback": false
}
}
+64 -6
View File
@@ -1,14 +1,14 @@
{
"totalCases" : 10,
"passedCases" : 10,
"totalCases" : 12,
"passedCases" : 12,
"passRate" : 1.0,
"verdictDistribution" : {
"PASS" : 4,
"LOW_CONFID" : 5,
"PASS" : 5,
"LOW_CONFID" : 6,
"REJECT" : 1
},
"averageToolCallCount" : 1.5,
"averageDurationMs" : 39800.0,
"averageToolCallCount" : 1.4166666666666667,
"averageDurationMs" : 38500.0,
"results" : [ {
"caseId" : "narrow-highcpu-observation",
"title" : "Narrow HighCPU observation",
@@ -22,10 +22,31 @@
},
"gatekeeperStatus" : "pass",
"gatekeeperRuleSetVersion" : "gatekeeper-rules-v1",
"promptAuditVersion" : null,
"composerStatus" : "valid",
"claimCheckCount" : 1,
"gatekeeperRuleCount" : 1,
"toolCallCount" : 1,
"durationMs" : 18000
}, {
"caseId" : "prompt-gatekeeper-audit-closure",
"title" : "Prompt and Gatekeeper audit closure",
"passed" : true,
"failedChecks" : [ ],
"verdict" : "PASS",
"matchedKeywordCount" : 3,
"requiredKeywordCount" : 3,
"evidenceCoverage" : {
"query_metrics" : true
},
"gatekeeperStatus" : "pass",
"gatekeeperRuleSetVersion" : "gatekeeper-rules-v1",
"promptAuditVersion" : "chat-prompts-v1",
"composerStatus" : "valid",
"claimCheckCount" : 1,
"gatekeeperRuleCount" : 2,
"toolCallCount" : 1,
"durationMs" : 19000
}, {
"caseId" : "hikari-no-evidence-negative-observation",
"title" : "Hikari no-evidence negative observation",
@@ -39,8 +60,10 @@
},
"gatekeeperStatus" : "pass",
"gatekeeperRuleSetVersion" : "gatekeeper-rules-v1",
"promptAuditVersion" : null,
"composerStatus" : "valid",
"claimCheckCount" : 1,
"gatekeeperRuleCount" : 1,
"toolCallCount" : 1,
"durationMs" : 21000
}, {
@@ -58,8 +81,10 @@
},
"gatekeeperStatus" : null,
"gatekeeperRuleSetVersion" : null,
"promptAuditVersion" : null,
"composerStatus" : null,
"claimCheckCount" : null,
"gatekeeperRuleCount" : null,
"toolCallCount" : 3,
"durationMs" : 42000
}, {
@@ -76,8 +101,10 @@
},
"gatekeeperStatus" : null,
"gatekeeperRuleSetVersion" : null,
"promptAuditVersion" : null,
"composerStatus" : null,
"claimCheckCount" : null,
"gatekeeperRuleCount" : null,
"toolCallCount" : 2,
"durationMs" : 51000
}, {
@@ -93,8 +120,10 @@
},
"gatekeeperStatus" : null,
"gatekeeperRuleSetVersion" : null,
"promptAuditVersion" : null,
"composerStatus" : null,
"claimCheckCount" : null,
"gatekeeperRuleCount" : null,
"toolCallCount" : 1,
"durationMs" : 36000
}, {
@@ -111,8 +140,10 @@
},
"gatekeeperStatus" : null,
"gatekeeperRuleSetVersion" : null,
"promptAuditVersion" : null,
"composerStatus" : null,
"claimCheckCount" : null,
"gatekeeperRuleCount" : null,
"toolCallCount" : 2,
"durationMs" : 47000
}, {
@@ -129,8 +160,10 @@
},
"gatekeeperStatus" : null,
"gatekeeperRuleSetVersion" : null,
"promptAuditVersion" : null,
"composerStatus" : null,
"claimCheckCount" : null,
"gatekeeperRuleCount" : null,
"toolCallCount" : 2,
"durationMs" : 53000
}, {
@@ -146,8 +179,10 @@
},
"gatekeeperStatus" : "fail",
"gatekeeperRuleSetVersion" : null,
"promptAuditVersion" : null,
"composerStatus" : "valid",
"claimCheckCount" : 1,
"gatekeeperRuleCount" : null,
"toolCallCount" : 1,
"durationMs" : 39000
}, {
@@ -163,10 +198,31 @@
},
"gatekeeperStatus" : "pass",
"gatekeeperRuleSetVersion" : null,
"promptAuditVersion" : null,
"composerStatus" : "valid",
"claimCheckCount" : 2,
"gatekeeperRuleCount" : null,
"toolCallCount" : 1,
"durationMs" : 44000
}, {
"caseId" : "audit-metadata-low-confid",
"title" : "Audit metadata low confidence",
"passed" : true,
"failedChecks" : [ ],
"verdict" : "LOW_CONFID",
"matchedKeywordCount" : 2,
"requiredKeywordCount" : 2,
"evidenceCoverage" : {
"query_logs" : true
},
"gatekeeperStatus" : "pass",
"gatekeeperRuleSetVersion" : "gatekeeper-rules-v1",
"promptAuditVersion" : "chat-prompts-v1",
"composerStatus" : "valid",
"claimCheckCount" : 2,
"gatekeeperRuleCount" : 2,
"toolCallCount" : 1,
"durationMs" : 45000
}, {
"caseId" : "composer-fallback-no-raw-json",
"title" : "Composer fallback no raw JSON",
@@ -180,8 +236,10 @@
},
"gatekeeperStatus" : "pass",
"gatekeeperRuleSetVersion" : null,
"promptAuditVersion" : null,
"composerStatus" : "composer_malformed",
"claimCheckCount" : 2,
"gatekeeperRuleCount" : null,
"toolCallCount" : 1,
"durationMs" : 47000
} ]
+20 -18
View File
@@ -1,28 +1,30 @@
# Diagnosis Eval Report
- Total cases: 10
- Passed cases: 10
- Total cases: 12
- Passed cases: 12
- Pass rate: 100.00%
- Average tool calls: 1.50
- Average duration ms: 39800.00
- Average tool calls: 1.42
- Average duration ms: 38500.00
## Verdict Distribution
- PASS: 4
- LOW_CONFID: 5
- PASS: 5
- LOW_CONFID: 6
- REJECT: 1
## Cases
| Case | Result | Verdict | Gatekeeper | Rule Set | Composer | Claim Checks | Keywords | Tool Calls | Duration ms | Failed Checks |
| --- | --- | --- | --- | --- | --- | ---: | --- | ---: | ---: | --- |
| narrow-highcpu-observation | PASS | PASS | pass | gatekeeper-rules-v1 | valid | 1 | 3/3 | 1 | 18000 | - |
| hikari-no-evidence-negative-observation | PASS | PASS | pass | gatekeeper-rules-v1 | valid | 1 | 3/3 | 1 | 21000 | - |
| payment-timeout | PASS | PASS | - | - | - | - | 3/3 | 3 | 42000 | - |
| mysql-pool-exhausted | PASS | LOW_CONFID | - | - | - | - | 3/3 | 2 | 51000 | - |
| redis-timeout | PASS | LOW_CONFID | - | - | - | - | 2/2 | 1 | 36000 | - |
| slow-response | PASS | PASS | - | - | - | - | 2/2 | 2 | 47000 | - |
| jvm-memory-risk | PASS | LOW_CONFID | - | - | - | - | 3/3 | 2 | 53000 | - |
| gatekeeper-fabricated-invocation | PASS | REJECT | fail | - | valid | 1 | 3/3 | 1 | 39000 | - |
| unsupported-claim-filtering | PASS | LOW_CONFID | pass | - | valid | 2 | 2/2 | 1 | 44000 | - |
| composer-fallback-no-raw-json | PASS | LOW_CONFID | pass | - | composer_malformed | 2 | 2/2 | 1 | 47000 | - |
| Case | Result | Verdict | Gatekeeper | Rule Set | Prompt Audit | Composer | Claim Checks | Rules | Keywords | Tool Calls | Duration ms | Failed Checks |
| --- | --- | --- | --- | --- | --- | --- | ---: | ---: | --- | ---: | ---: | --- |
| narrow-highcpu-observation | PASS | PASS | pass | gatekeeper-rules-v1 | - | valid | 1 | 1 | 3/3 | 1 | 18000 | - |
| prompt-gatekeeper-audit-closure | PASS | PASS | pass | gatekeeper-rules-v1 | chat-prompts-v1 | valid | 1 | 2 | 3/3 | 1 | 19000 | - |
| hikari-no-evidence-negative-observation | PASS | PASS | pass | gatekeeper-rules-v1 | - | valid | 1 | 1 | 3/3 | 1 | 21000 | - |
| payment-timeout | PASS | PASS | - | - | - | - | - | - | 3/3 | 3 | 42000 | - |
| mysql-pool-exhausted | PASS | LOW_CONFID | - | - | - | - | - | - | 3/3 | 2 | 51000 | - |
| redis-timeout | PASS | LOW_CONFID | - | - | - | - | - | - | 2/2 | 1 | 36000 | - |
| slow-response | PASS | PASS | - | - | - | - | - | - | 2/2 | 2 | 47000 | - |
| jvm-memory-risk | PASS | LOW_CONFID | - | - | - | - | - | - | 3/3 | 2 | 53000 | - |
| gatekeeper-fabricated-invocation | PASS | REJECT | fail | - | - | valid | 1 | - | 3/3 | 1 | 39000 | - |
| unsupported-claim-filtering | PASS | LOW_CONFID | pass | - | - | valid | 2 | - | 2/2 | 1 | 44000 | - |
| audit-metadata-low-confid | PASS | LOW_CONFID | pass | gatekeeper-rules-v1 | chat-prompts-v1 | valid | 2 | 2 | 2/2 | 1 | 45000 | - |
| composer-fallback-no-raw-json | PASS | LOW_CONFID | pass | - | - | composer_malformed | 2 | - | 2/2 | 1 | 47000 | - |
+23 -1
View File
@@ -34,7 +34,16 @@ baseline report:整套固定集当前认可的结果
"expectedGatekeeperStatuses": ["pass"],
"expectedGatekeeperRuleSetVersion": "gatekeeper-rules-v1",
"expectedComposerStatuses": ["valid"],
"forbiddenConfirmedClaimKeywords": ["主库故障"]
"forbiddenConfirmedClaimKeywords": ["主库故障"],
"requirePromptAudit": true,
"expectedPromptAuditVersion": "chat-prompts-v1",
"expectedPromptVersions": {
"chat_planner": "chat-planner-v1",
"chat_executor": "chat-executor-v2",
"chat_verifier": "chat-verifier-v2",
"chat_composer": "chat-composer-v1"
},
"requireGatekeeperRules": true
}
```
@@ -58,6 +67,10 @@ baseline report:整套固定集当前认可的结果
| `expectedGatekeeperRuleSetVersion` | 期望的 Gatekeeper 规则集版本 | 配置后校验 `gatekeeper_result.rule_set_version` |
| `expectedComposerStatuses` | 允许的 Composer 状态 | 实际 `composer_output.status` 不在列表中则失败 |
| `forbiddenConfirmedClaimKeywords` | 不得进入最终答案的未支持结论关键词 | 用于证明 unsupported/external_unknown claim 被过滤 |
| `requirePromptAudit` | 是否要求 Prompt 审计元数据 | 要求 `prompt_audit.version` 存在 |
| `expectedPromptAuditVersion` | 期望的 Prompt 审计目录版本 | 配置后校验 `prompt_audit.version` |
| `expectedPromptVersions` | 期望的各角色 Prompt 版本 | 校验 `prompt_audit.prompts[*].name/version` |
| `requireGatekeeperRules` | 是否要求 Gatekeeper 规则元数据 | 要求 `gatekeeper_result.rules` 非空,且每条规则有 `id`、`enabled`、`default_severity` |
## 2. Trace Fixture
@@ -72,6 +85,9 @@ fixture 是一次 Agent 运行后的 trace 快照。评测器只读取当前规
| `session.selfEvaluation.verifier_evaluation.verdict` | Verifier 判定 | 必须存在并符合 case 的 `allowedVerdicts` |
| `session.selfEvaluation.verifier_evaluation.gatekeeper_result.status` | Gatekeeper 结果 | V2 case 必须存在;`fail` 不允许搭配 `PASS` |
| `session.selfEvaluation.verifier_evaluation.gatekeeper_result.rule_set_version` | Gatekeeper 规则集版本 | 新矩阵 case 可显式断言该版本 |
| `session.selfEvaluation.verifier_evaluation.gatekeeper_result.rules` | Gatekeeper 规则元数据摘要 | 审计 case 可要求规则列表非空且字段完整 |
| `session.selfEvaluation.verifier_evaluation.prompt_audit.version` | Chat Prompt 审计目录版本 | 审计 case 可显式断言该版本 |
| `session.selfEvaluation.verifier_evaluation.prompt_audit.prompts` | 各 Chat Prompt 名称、版本和资源路径 | 审计 case 可断言 planner、executor、verifier、composer 版本 |
| `session.selfEvaluation.verifier_evaluation.claim_checks` | Verifier V2 claim 级校验 | V2 case 必须存在;每项需要 `claim_id`、`verification`、`detail` |
| `session.selfEvaluation.verifier_evaluation.composer_output.status` | Composer 渲染状态 | V2 case 必须存在;记录 `valid`、`composer_malformed` 等 |
| `session.selfEvaluation.verifier_evaluation.executor_structured_output.claims[*].evidence_bindings` | Executor claim 证据绑定 | 如果结构化输出存在,每条 claim 需要证据绑定 |
@@ -95,8 +111,10 @@ Java 类型:`DiagnosisEvalResult`
| `evidenceCoverage` | 每个必需工具是否出现 |
| `gatekeeperStatus` | 读到的 `gatekeeper_result.status` |
| `gatekeeperRuleSetVersion` | 读到的 `gatekeeper_result.rule_set_version` |
| `promptAuditVersion` | 读到的 `prompt_audit.version` |
| `composerStatus` | 读到的 `composer_output.status` |
| `claimCheckCount` | `claim_checks` 数量 |
| `gatekeeperRuleCount` | `gatekeeper_result.rules` 数量 |
| `toolCallCount` | trace 中工具调用总数 |
| `durationMs` | trace 总耗时 |
@@ -130,6 +148,10 @@ Executor structured output
- V2 case 必须有 `gatekeeper_result`、`claim_checks`、`composer_output`。
- `gatekeeper_result.status = fail` 时,Verifier verdict 不能是 `PASS`。
- 配置 `expectedGatekeeperRuleSetVersion` 的 case 必须匹配 `gatekeeper_result.rule_set_version`。
- 配置 `requirePromptAudit` 的 case 必须包含 `prompt_audit.version`。
- 配置 `expectedPromptAuditVersion` 的 case 必须匹配 `prompt_audit.version`。
- 配置 `expectedPromptVersions` 的 case 必须能在 `prompt_audit.prompts` 中找到对应角色和版本。
- 配置 `requireGatekeeperRules` 的 case 必须包含非空 `gatekeeper_result.rules`,且每条规则有 `id`、`enabled`、`default_severity`。
- `claim_checks[*].verification` 只能是 `direct_observation`、`reasonable_inference`、`overstated`、`unsupported`、`external_unknown`、`contradicted`。
- Composer 输出必须记录 `status`。
- 最终答案不能泄漏 `executor_evidence_v2`、`answer_version`、`evidence_bindings`、`claim_id`。
+55 -38
View File
@@ -1,47 +1,64 @@
# 已知问题记录
# MVP Issues 索引
| # | 标题 | 严重程度 | 状态 | 文件 |
|---|---|---|---|---|
| ISS-001 | Executor 重复召回同一文档 | 中 | 已修复 | [ISS-001-duplicate-retrieval.md](ISS-001-duplicate-retrieval.md) |
| ISS-002 | Executor 无约束重复调用 lookup_knowledge | 中 | 已修复 | [ISS-002-executor-unconstrained-lookup.md](ISS-002-executor-unconstrained-lookup.md) |
| ISS-003 | MVP 设计与实现 Review 收敛 | 高 | 待规划 | [ISS-003-mvp-design-implementation-review.md](ISS-003-mvp-design-implementation-review.md) |
| ISS-004 | Executor 域级检索水位控制(Phase 2) | 低 | 待规划 | [ISS-004-executor-domain-hard-limit.md](ISS-004-executor-domain-hard-limit.md) |
| ISS-005 | 证据链补齐与降级契约收敛 | 高 | 已归档 | [ISS-005-evidence-trace-hardening.md](ISS-005-evidence-trace-hardening.md) |
| ISS-006 | 固定诊断评测集与回归 Harness | 高 | 已归档 | [ISS-006-diagnosis-eval-harness.md](ISS-006-diagnosis-eval-harness.md) |
| ISS-007 | Verifier 证据摘要保真与工具命中质量问题 | 高 | 已实施 | [ISS-007-verifier-evidence-summary-fidelity.md](ISS-007-verifier-evidence-summary-fidelity.md) |
| ISS-008 | Executor 窄范围查询越界 | 中 | 已修复 | [ISS-008-executor-narrow-scope-overreach.md](ISS-008-executor-narrow-scope-overreach.md) |
| ISS-009 | negative_observation 精确引用 no-evidence 结果 | 中 | 已修复 | [ISS-009-negative-observation-no-evidence-reference.md](ISS-009-negative-observation-no-evidence-reference.md) |
| executor-evidence-attribution-hallucination | Executor 证据归因幻觉 | 高 | 待规划 | [executor-evidence-attribution-hallucination.md](executor-evidence-attribution-hallucination.md) |
| executor-self-evidence-loop-design-note | Executor 自证循环与证据摘要链路设计记录 | 高 | 已形成方向 | [executor-self-evidence-loop-design-note.md](executor-self-evidence-loop-design-note.md) |
| expand-diagnosis-eval-fixtures | 补齐固定诊断评测 fixture 与 baseline | 中 | 已归档 | [expand-diagnosis-eval-fixtures.md](expand-diagnosis-eval-fixtures.md) |
| diagnosis-eval-baseline-diff | 诊断评测 baseline diff 与回归判断 | 中 | 已归档 | [diagnosis-eval-baseline-diff.md](diagnosis-eval-baseline-diff.md) |
| mvp-demo-interview-runbook | Plan C 面试可复现 Demo 包 | 中 | 已归档 | [mvp-demo-interview-runbook.md](mvp-demo-interview-runbook.md) |
**更新日期**:2026-07-09
**状态**:按活跃问题、设计笔记、RAG 问题集和已归档问题整理
## RAG 重构计划
## 目录约定
| 目录 | 用途 |
|---|---|
| [active/](active/) | 仍需要规划或实现的问题 |
| [design-notes/](design-notes/) | 已形成方向、用于指导后续实现的设计记录 |
| [rag/](rag/) | RAG 子问题集合;多数已合并到 RAG 重构计划 |
| [archived/](archived/) | 已修复、已实施或已归档的问题 |
## 活跃问题
| 名称 | 标题 | 严重程度 | 状态 | 文件 |
|---|---|---|---|---|
| rag-refactor-plan | RAG 检索重构计划 | 高 | 待规划 | [rag-refactor-plan.md](rag-refactor-plan.md) |
| ISS-003 | MVP 设计与实现 Review 收敛 | 高 | 待规划 | [active/ISS-003-mvp-design-implementation-review.md](active/ISS-003-mvp-design-implementation-review.md) |
| ISS-004 | Executor 域级检索水位控制 | 低 | 待规划 | [active/ISS-004-executor-domain-hard-limit.md](active/ISS-004-executor-domain-hard-limit.md) |
| executor-evidence-attribution-hallucination | Executor 证据归因幻觉 | 高 | 待规划 | [active/executor-evidence-attribution-hallucination.md](active/executor-evidence-attribution-hallucination.md) |
| rag-refactor-plan | RAG 检索重构计划 | 高 | 待规划 | [active/rag-refactor-plan.md](active/rag-refactor-plan.md) |
## RAG 检索问题
## 设计笔记
| 名称 | 标题 | 严重程度 | 状态 | 文件 |
|---|---|---|---|---|
| chunk-context-reconstruction | RAG 切片上下文重建缺失 | 高 | 已合并到重构计划 | [rag-chunk-context-reconstruction.md](rag-chunk-context-reconstruction.md) |
| breadcrumb-embedding-gap | RAG breadcrumb 未参与向量语义 | 高 | 已合并到重构计划 | [rag-breadcrumb-embedding-gap.md](rag-breadcrumb-embedding-gap.md) |
| l0-l1-fusion-ranking | RAG L0 和 L1 未真正融合排序 | 中 | 已合并到重构计划 | [rag-l0-l1-fusion-ranking.md](rag-l0-l1-fusion-ranking.md) |
| l0-keyword-matching-quality | RAG L0 关键词匹配质量不足 | 中 | 已合并到重构计划 | [rag-l0-keyword-matching-quality.md](rag-l0-keyword-matching-quality.md) |
| l1-score-calibration | RAG L1 分数阈值未校准 | 中 | 已合并到重构计划 | [rag-l1-score-calibration.md](rag-l1-score-calibration.md) |
| context-packing-and-reranking | RAG 缺少上下文打包和 Rerank | 中 | 已合并到重构计划 | [rag-context-packing-and-reranking.md](rag-context-packing-and-reranking.md) |
| upload-chunk-parameter-drift | RAG 上传切片参数未真正生效 | 低 | 已合并到重构计划 | [rag-upload-chunk-parameter-drift.md](rag-upload-chunk-parameter-drift.md) |
| query-rewrite-gap | RAG 查询改写能力薄弱 | 中 | 已合并到重构计划 | [rag-query-rewrite-gap.md](rag-query-rewrite-gap.md) |
| 名称 | 标题 | 状态 | 文件 |
|---|---|---|---|
| executor-self-evidence-loop-design-note | Executor 自证循环与证据摘要链路设计记录 | 已形成方向 | [design-notes/executor-self-evidence-loop-design-note.md](design-notes/executor-self-evidence-loop-design-note.md) |
| executor-structured-output-v2 | Executor 结构化输出 V2 阶段设计 | 部分已实施,保留为后续改造参考 | [design-notes/executor-structured-output-v2.md](design-notes/executor-structured-output-v2.md) |
## RAG 框架化改造
## RAG 问题集
| 名称 | 标题 | 严重程度 | 状态 | 文件 |
|---|---|---|---|---|
| spring-ai-vectorstore-migration | RAG 迁移到 Spring AI VectorStore 检索抽象 | 高 | 已合并到重构计划 | [rag-spring-ai-vectorstore-migration.md](rag-spring-ai-vectorstore-migration.md) |
| spring-ai-query-transformer | RAG 接入 Spring AI Query Transformer | 中 | 已合并到重构计划 | [rag-spring-ai-query-transformer.md](rag-spring-ai-query-transformer.md) |
| spring-ai-document-postprocessor | RAG 使用 DocumentPostProcessor 做后处理 | 中 | 已合并到重构计划 | [rag-spring-ai-document-postprocessor.md](rag-spring-ai-document-postprocessor.md) |
| l0-domain-entity-hint | RAG 将 L0 降级为领域和实体 Hint | 中 | 已合并到重构计划 | [rag-l0-domain-entity-hint.md](rag-l0-domain-entity-hint.md) |
| spring-ai-advisor-boundary | RAG 明确 Spring AI Advisor 与 Agent Tool 的边界 | 中 | 已合并到重构计划 | [rag-spring-ai-advisor-boundary.md](rag-spring-ai-advisor-boundary.md) |
这些问题已经收敛到 [active/rag-refactor-plan.md](active/rag-refactor-plan.md),单个文件保留用于追溯原始问题和设计背景。
| 名称 | 标题 | 状态 | 文件 |
|---|---|---|---|
| breadcrumb-embedding-gap | RAG breadcrumb 未参与向量语义 | 已合并到重构计划 | [rag/rag-breadcrumb-embedding-gap.md](rag/rag-breadcrumb-embedding-gap.md) |
| chunk-context-reconstruction | RAG 切片上下文重建缺失 | 已合并到重构计划 | [rag/rag-chunk-context-reconstruction.md](rag/rag-chunk-context-reconstruction.md) |
| context-packing-and-reranking | RAG 缺少上下文打包和 Rerank | 已合并到重构计划 | [rag/rag-context-packing-and-reranking.md](rag/rag-context-packing-and-reranking.md) |
| l0-domain-entity-hint | RAG 将 L0 降级为领域和实体 Hint | 已合并到重构计划 | [rag/rag-l0-domain-entity-hint.md](rag/rag-l0-domain-entity-hint.md) |
| l0-keyword-matching-quality | RAG L0 关键词匹配质量不足 | 已合并到重构计划 | [rag/rag-l0-keyword-matching-quality.md](rag/rag-l0-keyword-matching-quality.md) |
| l0-l1-fusion-ranking | RAG L0 和 L1 未真正融合排序 | 已合并到重构计划 | [rag/rag-l0-l1-fusion-ranking.md](rag/rag-l0-l1-fusion-ranking.md) |
| l1-score-calibration | RAG L1 分数阈值未校准 | 已合并到重构计划 | [rag/rag-l1-score-calibration.md](rag/rag-l1-score-calibration.md) |
| query-rewrite-gap | RAG 查询改写能力薄弱 | 已合并到重构计划 | [rag/rag-query-rewrite-gap.md](rag/rag-query-rewrite-gap.md) |
| spring-ai-advisor-boundary | Spring AI Advisor 与 Agent Tool 边界 | 已合并到重构计划 | [rag/rag-spring-ai-advisor-boundary.md](rag/rag-spring-ai-advisor-boundary.md) |
| spring-ai-document-postprocessor | 使用 DocumentPostProcessor 做后处理 | 已合并到重构计划 | [rag/rag-spring-ai-document-postprocessor.md](rag/rag-spring-ai-document-postprocessor.md) |
| spring-ai-query-transformer | 接入 Spring AI Query Transformer | 已合并到重构计划 | [rag/rag-spring-ai-query-transformer.md](rag/rag-spring-ai-query-transformer.md) |
| spring-ai-vectorstore-migration | 迁移到 Spring AI VectorStore 检索抽象 | 已合并到重构计划 | [rag/rag-spring-ai-vectorstore-migration.md](rag/rag-spring-ai-vectorstore-migration.md) |
| upload-chunk-parameter-drift | 上传切片参数未真正生效 | 已合并到重构计划 | [rag/rag-upload-chunk-parameter-drift.md](rag/rag-upload-chunk-parameter-drift.md) |
## 已归档问题
| 名称 | 标题 | 状态 | 文件 |
|---|---|---|---|
| ISS-001 | Executor 重复召回同一文档 | 已修复 | [archived/ISS-001-duplicate-retrieval.md](archived/ISS-001-duplicate-retrieval.md) |
| ISS-002 | Executor 无约束重复调用 lookup_knowledge | 已修复 | [archived/ISS-002-executor-unconstrained-lookup.md](archived/ISS-002-executor-unconstrained-lookup.md) |
| ISS-005 | 证据链补齐与降级契约收敛 | 已归档 | [archived/ISS-005-evidence-trace-hardening.md](archived/ISS-005-evidence-trace-hardening.md) |
| ISS-006 | 固定诊断评测集与回归 Harness | 已归档 | [archived/ISS-006-diagnosis-eval-harness.md](archived/ISS-006-diagnosis-eval-harness.md) |
| ISS-007 | Verifier 证据摘要保真与工具命中质量问题 | 已实施 | [archived/ISS-007-verifier-evidence-summary-fidelity.md](archived/ISS-007-verifier-evidence-summary-fidelity.md) |
| ISS-008 | Executor 窄范围查询越界 | 已修复 | [archived/ISS-008-executor-narrow-scope-overreach.md](archived/ISS-008-executor-narrow-scope-overreach.md) |
| ISS-009 | negative_observation 精确引用 no-evidence 结果 | 已修复 | [archived/ISS-009-negative-observation-no-evidence-reference.md](archived/ISS-009-negative-observation-no-evidence-reference.md) |
| diagnosis-eval-baseline-diff | 诊断评测 baseline diff 与回归判断 | 已归档 | [archived/diagnosis-eval-baseline-diff.md](archived/diagnosis-eval-baseline-diff.md) |
| expand-diagnosis-eval-fixtures | 补齐固定诊断评测 fixture 与 baseline | 已归档 | [archived/expand-diagnosis-eval-fixtures.md](archived/expand-diagnosis-eval-fixtures.md) |
| mvp-demo-interview-runbook | Plan C 面试可复现 Demo 包 | 已归档 | [archived/mvp-demo-interview-runbook.md](archived/mvp-demo-interview-runbook.md) |
@@ -347,19 +347,19 @@ RAG、Agent、AIOps、数据库记录互相关联,必须分阶段推进,每
本计划合并以下问题和改造方向:
- [rag-chunk-context-reconstruction.md](rag-chunk-context-reconstruction.md)
- [rag-breadcrumb-embedding-gap.md](rag-breadcrumb-embedding-gap.md)
- [rag-l0-l1-fusion-ranking.md](rag-l0-l1-fusion-ranking.md)
- [rag-l0-keyword-matching-quality.md](rag-l0-keyword-matching-quality.md)
- [rag-l1-score-calibration.md](rag-l1-score-calibration.md)
- [rag-context-packing-and-reranking.md](rag-context-packing-and-reranking.md)
- [rag-upload-chunk-parameter-drift.md](rag-upload-chunk-parameter-drift.md)
- [rag-query-rewrite-gap.md](rag-query-rewrite-gap.md)
- [rag-spring-ai-vectorstore-migration.md](rag-spring-ai-vectorstore-migration.md)
- [rag-spring-ai-query-transformer.md](rag-spring-ai-query-transformer.md)
- [rag-spring-ai-document-postprocessor.md](rag-spring-ai-document-postprocessor.md)
- [rag-l0-domain-entity-hint.md](rag-l0-domain-entity-hint.md)
- [rag-spring-ai-advisor-boundary.md](rag-spring-ai-advisor-boundary.md)
- [rag-chunk-context-reconstruction.md](../rag/rag-chunk-context-reconstruction.md)
- [rag-breadcrumb-embedding-gap.md](../rag/rag-breadcrumb-embedding-gap.md)
- [rag-l0-l1-fusion-ranking.md](../rag/rag-l0-l1-fusion-ranking.md)
- [rag-l0-keyword-matching-quality.md](../rag/rag-l0-keyword-matching-quality.md)
- [rag-l1-score-calibration.md](../rag/rag-l1-score-calibration.md)
- [rag-context-packing-and-reranking.md](../rag/rag-context-packing-and-reranking.md)
- [rag-upload-chunk-parameter-drift.md](../rag/rag-upload-chunk-parameter-drift.md)
- [rag-query-rewrite-gap.md](../rag/rag-query-rewrite-gap.md)
- [rag-spring-ai-vectorstore-migration.md](../rag/rag-spring-ai-vectorstore-migration.md)
- [rag-spring-ai-query-transformer.md](../rag/rag-spring-ai-query-transformer.md)
- [rag-spring-ai-document-postprocessor.md](../rag/rag-spring-ai-document-postprocessor.md)
- [rag-l0-domain-entity-hint.md](../rag/rag-l0-domain-entity-hint.md)
- [rag-spring-ai-advisor-boundary.md](../rag/rag-spring-ai-advisor-boundary.md)
---
@@ -4,7 +4,7 @@
**严重程度**:中(影响 token 消耗和上下文质量,不影响功能正确性)
**发现时间**:2026-06-30
**修复版本**:session-dedup-knowledge-map
**历史架构文档**:[会话级去重与知识域地图](../architecture/archive/2026-07-05-legacy/session-dedup-knowledge-map.md)
**历史架构文档**:[会话级去重与知识域地图](../../architecture/archive/2026-07-05-legacy/session-dedup-knowledge-map.md)
---
+41
View File
@@ -0,0 +1,41 @@
# Agent 步骤表:agent_step
**状态**:当前表
**来源**:`V005__create_session_storage.sql`、`V006__fix_agent_step_json_to_text.sql`、`AgentStep`
## 定位
`agent_step` 记录一次诊断过程中每个 Agent 步骤的模型输入、输出、耗时和 Token 消耗。页面展示执行链路时应优先按 `step_index` 排序。
## 字段
| 字段 | 类型 | 必填 | 说明 |
|---|---|---|---|
| `id` | BIGINT | 是 | 自增主键 |
| `session_id` | VARCHAR(64) | 是 | 关联 `diagnosis_session.session_id` |
| `step_index` | INT | 是 | 步骤序号,从 0 开始 |
| `agent_name` | VARCHAR(32) | 是 | Agent 名称,例如 planner、executor、verifier、composer |
| `model_input` | TEXT | 否 | 模型输入摘要;`V006` 已从 JSON 改为 TEXT |
| `model_output` | TEXT | 否 | 模型输出摘要;`V006` 已从 JSON 改为 TEXT |
| `thought` | TEXT | 否 | Agent 思考过程或调试摘要 |
| `has_tool_call` | BOOLEAN | 否 | 本步骤是否触发工具调用 |
| `duration_ms` | INT | 否 | 本步骤耗时 |
| `token_count` | INT | 否 | 本步骤 Token 消耗 |
| `created_at` | DATETIME | 是 | 创建时间 |
## 索引
| 索引 | 字段 | 用途 |
|---|---|---|
| `idx_session_step` | `session_id, step_index` | Trace 页面按会话和步骤顺序查询 |
| `idx_agent_name` | `agent_name` | 按 Agent 类型筛选 |
## 关系
- `agent_step.session_id` 逻辑关联 `diagnosis_session.session_id`。
- `tool_invocation.step_id` 可关联 `agent_step.id`,但当前允许为空且不强制外键。
## 注意点
- 前端展示步骤时应按 `step_index` 排序,而不是按 `created_at` 或数据库返回顺序。
- Verifier 应在 Executor 循环完成后出现;如果 `step_index` 中 Verifier 提前,通常意味着编排或记录顺序有问题。
+40
View File
@@ -0,0 +1,40 @@
# MVP 数据表索引
**更新日期**:2026-07-09
**状态**:当前表文档入口
本目录保存当前 MVP 使用的数据表说明。详细结构以 Flyway migration 和实体类为准;本目录用于面试讲解、排查索引和快速理解数据流。
## 当前表
| 表 | 用途 | 文档 |
|---|---|---|
| `diagnosis_session` | 会话级主记录,保存 query、状态、最终答案和自评估 | [诊断会话表-diagnosis_session.md](诊断会话表-diagnosis_session.md) |
| `agent_step` | Agent 步骤记录,按 `step_index` 回放执行链路 | [Agent步骤表-agent_step.md](Agent步骤表-agent_step.md) |
| `tool_invocation` | 工具调用记录,支撑 Trace、Verifier 和评测 | [工具调用表-tool_invocation.md](工具调用表-tool_invocation.md) |
| `api_document` | 知识库文档元数据,和向量库 chunk 通过 `doc_id` 关联 | [文档元数据表-api_document.md](文档元数据表-api_document.md) |
| `knowledge_domain` | 知识域元数据,支撑 RAG domain hint 和检索策略 | [知识域表-knowledge_domain.md](知识域表-knowledge_domain.md) |
| `case_library` | 用户反馈沉淀出的高质量诊断案例 | [案例库表-case_library.md](案例库表-case_library.md) |
## 已归档表
| 表 | 归档原因 | 文档 |
|---|---|---|
| `diagnosis_record` | 已由 `V007` 删除,被 `diagnosis_session + agent_step + tool_invocation` 替代 | [archive/2026-07-09-doc-cleanup/旧诊断记录表-diagnosis_record.md](archive/2026-07-09-doc-cleanup/旧诊断记录表-diagnosis_record.md) |
## 核心关系
```text
diagnosis_session.session_id
-> agent_step.session_id
-> tool_invocation.session_id
-> case_library.diagnosis_id
api_document.doc_id
-> vector chunk metadata.docId / doc_id
knowledge_domain.domain_id
-> api_document metadata.category / vector chunk metadata.category
```
当前实现主要使用逻辑关联,不依赖数据库外键。
-332
View File
@@ -1,332 +0,0 @@
# api_document - 文档元数据表
## 表定位
**文档管理表**:管理接口文档的元信息,不负责文档检索(检索由 Milvus 负责)
## 设计理念
### 文档管理,不是文档检索
**核心定位**:
- MySQL 负责文档元数据管理(状态、版本、去重)
- Milvus 负责文档内容存储和检索
- 通过 doc_id 关联两者
**MVP版本原则**:
- ✅ 最简字段,满足基本管理需求
- ✅ 文件去重(基于 file_hash)
- ✅ 状态追踪(索引进度)
- ✅ 硬删除(同步删除 Milvus 数据)
- ❌ 暂不支持:软删除、启用开关、版本管理(Phase 2)
---
## 表结构(MVP版)
```sql
CREATE TABLE api_document (
-- 主键
id BIGINT PRIMARY KEY AUTO_INCREMENT,
doc_id VARCHAR(64) UNIQUE NOT NULL COMMENT '文档唯一ID(UUID),关联Milvus',
-- 文档分类
fault_category VARCHAR(32) DEFAULT 'EXTERNAL_API' COMMENT '文档类别',
fault_source VARCHAR(128) COMMENT '文档归属(省份/服务名)',
api_name VARCHAR(128) COMMENT '接口名称',
version VARCHAR(32) DEFAULT 'v1.0' COMMENT '文档版本',
-- 文件信息
file_name VARCHAR(256) NOT NULL COMMENT '原始文件名',
file_path VARCHAR(512) COMMENT '文件存储路径',
file_hash VARCHAR(64) COMMENT '文件MD5 hash(用于去重)',
file_size BIGINT COMMENT '文件大小(字节)',
-- 索引状态
status VARCHAR(16) DEFAULT 'PENDING' COMMENT '索引状态(PENDING/PROCESSING/INDEXED/FAILED)',
chunk_count INT DEFAULT 0 COMMENT '分块数量',
error_message TEXT COMMENT '失败原因',
-- 时间字段
indexed_at DATETIME COMMENT '索引完成时间',
created_at DATETIME DEFAULT CURRENT_TIMESTAMP,
updated_at DATETIME DEFAULT CURRENT_TIMESTAMP ON UPDATE CURRENT_TIMESTAMP,
-- 索引
UNIQUE INDEX uk_file_hash (file_hash),
INDEX idx_doc_id (doc_id),
INDEX idx_fault_source (fault_source),
INDEX idx_status (status),
INDEX idx_created_at (created_at)
) ENGINE=InnoDB DEFAULT CHARSET=utf8mb4 COMMENT='文档元数据表(MVP版)';
```
---
## 字段说明
| 字段 | 类型 | 必填 | 说明 |
|------|------|------|------|
| doc_id | VARCHAR(64) | 是 | **核心**:文档唯一ID,关联 Milvus |
| fault_category | VARCHAR(32) | 否 | 文档类别 |
| fault_source | VARCHAR(128) | 否 | 文档归属(省份/服务名)|
| api_name | VARCHAR(128) | 否 | 接口名称 |
| version | VARCHAR(32) | 否 | 文档版本 |
| file_name | VARCHAR(256) | 是 | 原始文件名 |
| file_path | VARCHAR(512) | 否 | 文件存储路径 |
| file_hash | VARCHAR(64) | 否 | **去重关键**:文件MD5 |
| file_size | BIGINT | 否 | 文件大小 |
| status | VARCHAR(16) | 是 | **状态追踪**:PENDING/PROCESSING/INDEXED/FAILED |
| chunk_count | INT | 否 | 分块数量 |
| error_message | TEXT | 否 | 失败原因 |
| indexed_at | DATETIME | 否 | 索引完成时间 |
---
## 核心设计决策
### 1. doc_id:MySQL 与 Milvus 的桥梁
```
作用:
- MySQL:通过 doc_id 管理文档元数据
- Milvus:每个 chunk 的 metadata 中携带 doc_id
关联关系:
api_document (MySQL)
doc_id: doc-001
↓ 1:N
Milvus chunks
chunk_1: {doc_id: 'doc-001', text: '...', vector: [...]}
chunk_2: {doc_id: 'doc-001', text: '...', vector: [...]}
管理操作:
- 删除文档:
DELETE FROM milvus_collection WHERE metadata["doc_id"] == 'doc-001';
DELETE FROM api_document WHERE doc_id = 'doc-001';
```
### 2. file_hash:文件去重
```
去重流程:
1. 用户上传文件
↓
2. 计算文件 MD5
file_hash = md5(file_content)
↓
3. 检查是否已存在
SELECT * FROM api_document WHERE file_hash = 'abc123...';
↓
4a. 如果存在 → 提示"文档已存在"
4b. 如果不存在 → 继续导入
唯一约束:UNIQUE INDEX uk_file_hash (file_hash)
```
### 3. status:状态追踪
```
状态流转:
PENDING (待处理)
↓
PROCESSING (处理中)
↓ 成功
INDEXED (已索引)
↓ 失败
FAILED (失败)
用途:
- 批量导入时监控进度
- 失败重试
- 统计索引成功率
```
### 4. 硬删除策略(MVP)
```
删除文档时:
1. 删除 Milvus 中的所有分块
2. 删除 MySQL 元数据
3. 可选:删除原始文件
特点:
- 简单直接
- 数据彻底删除
- 不可恢复(需谨慎)
Phase 2 可增强:
- 软删除(archived_at)
- 启用开关(enabled)
```
---
## 数据流
### 场景1:导入新文档
```
1. 用户上传文件
↓
2. 计算 hash
↓
3. 检查去重(MySQL)
↓
4. 插入元数据(status=PROCESSING)
↓
5. 后台处理:解析 → 分块 → 向量化 → 存入 Milvus
↓
6. 更新状态(status=INDEXED, chunk_count=15)
```
### 场景2:删除文档
```
1. 用户删除文档
↓
2. 删除 Milvus 数据(WHERE metadata["doc_id"] == 'xxx')
↓
3. 删除 MySQL 元数据
↓
4. 可选:删除原始文件
```
### 场景3:重新索引
```
1. 删除旧数据(Milvus + MySQL)
↓
2. 重新导入(同场景1)
```
---
## 典型查询
```sql
-- 查看文档列表
SELECT doc_id, file_name, version, status, chunk_count, indexed_at
FROM api_document
WHERE fault_source = '广东'
AND status = 'INDEXED'
ORDER BY indexed_at DESC;
-- 查询失败的文档
SELECT doc_id, file_name, error_message
FROM api_document
WHERE status = 'FAILED';
-- 统计各状态文档数量
SELECT status, COUNT(*) as count
FROM api_document
GROUP BY status;
```
---
## 与 Milvus 的协作
### Milvus Collection Schema
```python
{
"collection_name": "api_doc_collection",
"fields": [
{"name": "id", "type": "VARCHAR", "is_primary": true},
{"name": "content", "type": "VARCHAR"},
{"name": "vector", "type": "FLOAT_VECTOR", "dim": 1536},
{"name": "metadata", "type": "JSON"}
]
}
# metadata 结构
{
"doc_id": "doc-001", # 关联 MySQL
"_source": "/path/to/file",
"_file_name": "xxx.docx",
"chunkIndex": 0,
"totalChunks": 15
}
```
### Java 代码示例
```java
// 插入时携带 doc_id
Map<String, Object> metadata = new HashMap<>();
metadata.put("doc_id", docId); // 关联 MySQL
metadata.put("_source", filePath);
metadata.put("chunkIndex", chunkIndex);
// 删除文档的所有分块
String expr = String.format("metadata[\"doc_id\"] == \"%s\"", docId);
milvusClient.delete(DeleteParam.newBuilder()
.withCollectionName(COLLECTION_NAME)
.withExpr(expr)
.build());
```
---
## 数据示例
```sql
-- 外部接口文档
INSERT INTO api_document VALUES
(1, 'doc-001', 'EXTERNAL_API', '广东', '社保查询', 'v2.1',
'广东社保查询v2.1.docx', '/docs/guangdong/social-v2.1.docx',
'abc123...', 1048576,
'INDEXED', 15, NULL, '2024-06-15 10:30:00', NOW(), NOW());
-- 内部服务文档
INSERT INTO api_document VALUES
(2, 'doc-002', 'INTERNAL_ERROR', 'order-service', '订单服务API', 'v1.0',
'订单服务API文档.pdf', '/docs/internal/order-service-api.pdf',
'def456...', 2097152,
'INDEXED', 20, NULL, '2024-06-14 15:20:00', NOW(), NOW());
-- 处理失败的文档
INSERT INTO api_document VALUES
(3, 'doc-003', 'EXTERNAL_API', '江苏', '公积金查询', 'v1.5',
'江苏公积金查询.html', '/docs/jiangsu/fund-v1.5.html',
'ghi789...', 512000,
'FAILED', 0, '不支持HTML格式', NULL, NOW(), NOW());
```
---
## 数据量预估
```
预估:100-200 条
- 外部接口文档:50-100 条
- 内部服务文档:20-50 条
- 其他文档:30-50 条
存储:
- 单条记录:约 1KB
- 200 条:约 200KB
结论:数据量很小
```
---
## MVP 版本的简化
```
Phase 1(当前):
✅ 基础字段和表结构
✅ 文件去重(file_hash)
✅ 状态追踪(status)
✅ 硬删除
✅ 通过 doc_id 关联 Milvus
Phase 2(未来增强):
❌ enabled(启用开关)
❌ archived_at(软删除)
❌ batch_id(批次管理)
❌ status 细化
❌ tags(标签分类)
```
@@ -1,4 +1,6 @@
# diagnosis_record - 诊断记录表
# diagnosis_record - 旧诊断记录表
> 归档说明:`diagnosis_record` 已在 `V007__drop_diagnosis_record.sql` 中删除,当前主模型是 `diagnosis_session + agent_step + tool_invocation`。本文只用于追溯早期设计。
## 表定位
-265
View File
@@ -1,265 +0,0 @@
# case_library - 案例库表
## 表定位
**知识沉淀表**:存储高质量诊断案例,支持相似案例推荐
## 设计理念
### 知识沉淀,系统越用越智能
**核心价值**:
- 质量过滤:只存储高质量案例(成功诊断 + 用户反馈有用)
- 知识沉淀:历史诊断经验可复用
- 提升准确率:相似问题提供历史参考
- 加速诊断:快速推荐相似案例
**MVP版本设计原则**:
- ✅ 能用:满足基本案例推荐功能
- ✅ 简单:字段不多,逻辑清晰
- ✅ 可扩展:后续可增加字段
---
## 表结构(MVP版)
```sql
CREATE TABLE case_library (
-- 主键
id BIGINT PRIMARY KEY AUTO_INCREMENT,
case_id VARCHAR(64) UNIQUE NOT NULL COMMENT '案例唯一ID(UUID)',
-- 来源关联
diagnosis_id VARCHAR(64) COMMENT '关联诊断记录(可选,人工录入时为空)',
source_type VARCHAR(16) DEFAULT 'AUTO' COMMENT '来源类型(AUTO:自动生成/MANUAL:人工录入)',
-- 案例分类
fault_category VARCHAR(32) COMMENT '故障类别(EXTERNAL_API/INTERNAL_ERROR/DATABASE...)',
fault_source VARCHAR(128) COMMENT '故障源(省份/服务名/类名...)',
fault_target VARCHAR(256) COMMENT '故障目标(接口URL/方法名/SQL...)',
error_code VARCHAR(64) COMMENT '错误码',
-- 案例内容
title VARCHAR(256) NOT NULL COMMENT '案例标题(简短描述)',
root_cause TEXT NOT NULL COMMENT '根因分析',
solution TEXT NOT NULL COMMENT '解决方案',
-- 简单统计
reference_count INT DEFAULT 0 COMMENT '引用次数(被推荐的次数)',
-- 元数据
created_by VARCHAR(64) COMMENT '创建人',
created_at DATETIME DEFAULT CURRENT_TIMESTAMP,
updated_at DATETIME DEFAULT CURRENT_TIMESTAMP ON UPDATE CURRENT_TIMESTAMP,
-- 索引
INDEX idx_fault_category (fault_category),
INDEX idx_error_code (error_code),
INDEX idx_fault_source (fault_source),
INDEX idx_fault_target (fault_target(100)),
INDEX idx_diagnosis_id (diagnosis_id),
INDEX idx_reference_count (reference_count),
INDEX idx_created_at (created_at)
) ENGINE=InnoDB DEFAULT CHARSET=utf8mb4 COMMENT='案例库表(MVP版)';
```
---
## 字段说明
| 字段 | 类型 | 必填 | 说明 |
|------|------|------|------|
| case_id | VARCHAR(64) | 是 | 案例唯一标识(UUID)|
| diagnosis_id | VARCHAR(64) | 否 | 关联诊断记录(人工录入时为空)|
| source_type | VARCHAR(16) | 是 | 来源:AUTO(自动)/MANUAL(人工)|
| fault_category | VARCHAR(32) | 否 | 故障类别 |
| fault_source | VARCHAR(128) | 否 | 故障源 |
| fault_target | VARCHAR(256) | 否 | 故障目标(与 diagnosis_record 一致)|
| error_code | VARCHAR(64) | 否 | 错误码 |
| title | VARCHAR(256) | 是 | 案例标题 |
| root_cause | TEXT | 是 | 根因分析(核心内容)|
| solution | TEXT | 是 | 解决方案(核心内容)|
| reference_count | INT | 是 | 引用次数(用于排序)|
---
## 核心设计决策
### 1. 案例来源
```
来源1:自动生成(source_type=AUTO)
├─ 触发条件:诊断成功 + 用户反馈"有用"
├─ 关联诊断:diagnosis_id 不为空
└─ 质量保证:用户验证过
来源2:人工录入(source_type=MANUAL)
├─ 运维团队总结的经典案例
├─ diagnosis_id 为空
└─ 质量最高
注意:诊断失败或用户反馈"无用"的不自动生成案例
```
### 2. 简化的评分机制(MVP)
```
MVP版本:只按 reference_count 排序
- 引用次数多的排前面
- 简单有效
Phase 2 可增强:
- 增加 useful_count(用户反馈有用次数)
- 增加 score(综合评分)
- 增加 is_featured(人工标记的经典案例)
```
### 3. 与 diagnosis_record 的关系
```
关系:一对一(可选)
- 一次诊断 → 可以生成一个案例
- 通过 diagnosis_id 关联
- diagnosis_id 可为空(人工录入案例)
流程:
diagnosis_record(成功)
↓
用户反馈"有用"
↓
自动生成 case_library
↓
后续可人工修正、合并相似案例
```
---
## 数据示例
### 示例1:外部接口故障案例
```sql
INSERT INTO case_library VALUES
(1, 'case-001', 'diag-001', 'AUTO', 'EXTERNAL_API', '广东', '/api/v1/guangdong/social-security', '40003',
'广东社保查询idCard字段缺失',
'请求报文中未传入idCard字段,导致参数校验失败',
'前端表单增加idCard必填校验;后端增加参数校验提示',
15, 'system', NOW(), NOW());
```
### 示例2:内部错误案例
```sql
INSERT INTO case_library VALUES
(2, 'case-002', 'diag-045', 'AUTO', 'INTERNAL_ERROR', 'order-service', 'OrderController.createOrder()', 'NullPointerException',
'订单服务创建订单空指针异常',
'OrderController.createOrder()方法中user对象为null,未做空判断',
'在第45行添加空判断:if (user == null) throw new BizException("用户信息不存在")',
8, 'system', NOW(), NOW());
```
### 示例3:人工录入案例
```sql
INSERT INTO case_library VALUES
(3, 'case-003', NULL, 'MANUAL', 'DATABASE', 'mysql-master-01', 'UPDATE orders SET status=? WHERE order_id=?', '1213',
'订单库存更新死锁通用处理',
'两个事务互相等待对方释放锁',
'调整事务加锁顺序:统一先锁订单,再锁库存;或使用乐观锁',
3, 'admin', NOW(), NOW());
```
---
## 典型查询
### 精确匹配查询
```sql
-- 按错误码查询
SELECT * FROM case_library
WHERE error_code = '40003'
ORDER BY reference_count DESC
LIMIT 5;
-- 按故障类别 + 错误码 + 故障目标查询
SELECT * FROM case_library
WHERE fault_category = 'INTERNAL_ERROR'
AND error_code = 'NullPointerException'
AND fault_target = 'OrderController.createOrder()'
ORDER BY reference_count DESC
LIMIT 5;
```
### 统计分析
```sql
-- 统计案例分布
SELECT
fault_category,
COUNT(*) as count,
AVG(reference_count) as avg_reference
FROM case_library
GROUP BY fault_category
ORDER BY count DESC;
-- Top 引用案例
SELECT title, reference_count, created_at
FROM case_library
ORDER BY reference_count DESC
LIMIT 10;
```
---
## 与 Milvus 的配合
### 混合检索策略
```
1. 精确匹配(MySQL)
- 按 error_code 查询
- 按 fault_category + fault_source 查询
- 优点:快速、准确
2. 语义检索(Milvus)
- 将案例内容向量化
- 按语义相似度查询
- 优点:能找到相似但不同错误码的案例
3. 混合策略(推荐)
Step 1: 先精确匹配(MySQL)
Step 2: 如果结果 < 3 个,补充语义检索(Milvus)
Step 3: 合并去重,按 reference_count 排序
Step 4: 返回 Top 5
```
---
## 数据量预估
```
预估:500-1000 条
- 初期:每月新增 10-20 条
- 稳定期:每月新增 5-10 条
- 总量:1-2 年达到稳定
存储:
- 单条记录:约 2KB
- 1000 条:约 2MB
结论:数据量很小
```
---
## MVP 版本的简化
```
Phase 1(当前):
✅ 基础字段和表结构
✅ 自动生成案例
✅ 人工录入案例
✅ 按 reference_count 简单排序
Phase 2(未来增强):
❌ useful_count + score(复杂评分)
❌ 版本管理
❌ 标签分类(tags)
❌ 案例合并功能
```
@@ -0,0 +1,66 @@
# 工具调用表:tool_invocation
**状态**:当前表
**来源**:`V005__create_session_storage.sql`、`V010__add_relevance_level_to_tool_invocation.sql`、`ToolInvocation`
## 定位
`tool_invocation` 记录 Agent 显式调用工具的事实,包括工具名、入参、输出摘要、检索层级、证据引用和失败信息。它是 Trace、Verifier、评测和人工排查的共同数据源。
## 字段
| 字段 | 类型 | 必填 | 说明 |
|---|---|---|---|
| `id` | BIGINT | 是 | 自增主键 |
| `session_id` | VARCHAR(64) | 是 | 关联 `diagnosis_session.session_id` |
| `step_id` | BIGINT | 否 | 可关联 `agent_step.id` |
| `tool_name` | VARCHAR(64) | 是 | 工具名称,例如 `lookup_knowledge`、日志查询、指标查询 |
| `input_params` | JSON | 是 | 工具入参 |
| `output_preview` | TEXT | 否 | 工具输出摘要或前缀 |
| `output_length` | INT | 否 | 工具输出字符数 |
| `retrieval_layer` | VARCHAR(8) | 否 | 检索层级,例如 `L0`、`L1`、`L0+L1` |
| `l0_match_count` | INT | 否 | L0 命中数量 |
| `l1_match_count` | INT | 否 | L1 命中数量 |
| `is_truncated` | BOOLEAN | 否 | 输出是否被截断 |
| `relevance_level` | VARCHAR(20) | 否 | 归一化质量等级:`PRECISE`、`HIGHLY_RELEVANT`、`REFERENCE`、`DEDUPED` |
| `dedup_reason` | VARCHAR(32) | 否 | 去重原因,例如 `doc_retrieved`、`domain_retrieved` |
| `retrieval_details` | JSON | 否 | 检索明细、证据引用、Gatekeeper 可用导航信息 |
| `duration_ms` | INT | 否 | 工具耗时 |
| `success` | BOOLEAN | 否 | 工具是否成功 |
| `error_message` | TEXT | 否 | 失败原因 |
| `created_at` | DATETIME | 是 | 创建时间 |
## 索引
| 索引 | 字段 | 用途 |
|---|---|---|
| `idx_session_id` | `session_id` | 按会话查询工具调用 |
| `idx_tool_name` | `tool_name` | 按工具类型排查 |
| `idx_retrieval_layer` | `retrieval_layer` | 观察 RAG L0/L1 行为 |
## 关系
- `tool_invocation.session_id` 逻辑关联 `diagnosis_session.session_id`。
- `tool_invocation.step_id` 可关联 `agent_step.id`,但当前不强制。
## 关键 JSON
`retrieval_details` 是扩展字段。当前重要结构包括:
```json
{
"evidence_status": "supported",
"evidence_refs": [
{
"raw_path": "$.logs[0]",
"text": "工具返回中可核对的最小证据文本"
}
]
}
```
## 注意点
- Verifier 不应只信任 RAG 证据;所有工具只要能提供 `evidence_refs`,都应该进入可校验证据链。
- `output_preview` 只适合展示和排查,不应被当成完整原始输出。
- `$.no_evidence` 只代表“本次工具未命中证据”,不能推导为“故障不存在”。
@@ -0,0 +1,51 @@
# 文档元数据表:api_document
**状态**:当前表
**来源**:`V003__create_api_document.sql`、`V004__add_metadata_to_api_document.sql`、`ApiDocument`
## 定位
`api_document` 是知识库文档的 MySQL 元数据表。它不保存向量正文,正文切片和向量检索由 Milvus/Zilliz collection 承担;两侧通过 `doc_id` 和 chunk metadata 逻辑关联。
## 字段
| 字段 | 类型 | 必填 | 说明 |
|---|---|---|---|
| `id` | BIGINT | 是 | 自增主键 |
| `doc_id` | VARCHAR(64) | 是 | 文档唯一 ID,关联向量库 chunk metadata |
| `fault_category` | VARCHAR(32) | 否 | 文档类别,默认 `EXTERNAL_API`;实体侧使用 `FaultCategory` |
| `fault_source` | VARCHAR(128) | 否 | 文档归属,例如服务名、省份或系统来源 |
| `api_name` | VARCHAR(128) | 否 | 接口或文档主题名称 |
| `version` | VARCHAR(32) | 否 | 文档版本,默认 `v1.0` |
| `file_name` | VARCHAR(256) | 是 | 原始文件名 |
| `file_path` | VARCHAR(512) | 否 | 文件存储路径 |
| `file_hash` | VARCHAR(64) | 否 | 文件 MD5,用于去重 |
| `file_size` | BIGINT | 否 | 文件大小,单位字节 |
| `status` | VARCHAR(16) | 否 | 索引状态:`PENDING`、`PROCESSING`、`INDEXED`、`FAILED` |
| `chunk_count` | INT | 否 | 向量库切片数量 |
| `error_message` | TEXT | 否 | 索引失败原因 |
| `metadata` | TEXT | 否 | frontmatter 元数据 JSON 字符串 |
| `indexed_at` | DATETIME | 否 | 索引完成时间 |
| `created_at` | DATETIME | 是 | 创建时间 |
| `updated_at` | DATETIME | 是 | 更新时间 |
## 索引
| 索引 | 字段 | 用途 |
|---|---|---|
| `uk_file_hash` | `file_hash` | 文件去重 |
| `idx_doc_id` | `doc_id` | 按文档 ID 查询 |
| `idx_fault_source` | `fault_source` | 按来源筛选 |
| `idx_status` | `status` | 查看索引状态 |
| `idx_created_at` | `created_at` | 按上传时间排序 |
## 关系
- `api_document.doc_id` 与向量库 chunk metadata 中的 `docId` / `doc_id` 逻辑关联。
- `knowledge_domain.domain_id` 与文档 metadata 中的 `category` 形成领域聚合关系;当前没有数据库外键。
## 注意点
- 删除文档时需要同时处理 MySQL 元数据和向量库 chunk。
- `metadata` 是 JSON 字符串,不是 MySQL JSON 列。
- 表字段以 Flyway 为准;实体默认值和枚举可能与迁移脚本的 SQL 默认值存在历史差异,排查时优先看实际迁移和数据库结构。
+50
View File
@@ -0,0 +1,50 @@
# 案例库表:case_library
**状态**:当前表
**来源**:`V002__create_case_library.sql`、`CaseLibrary`
## 定位
`case_library` 保存高质量诊断案例,用于后续相似案例推荐和知识沉淀。当前自动沉淀路径来自 `useful` 用户反馈:系统把 `diagnosis_session` 中的 query 和 answer 映射为案例内容。
## 字段
| 字段 | 类型 | 必填 | 说明 |
|---|---|---|---|
| `id` | BIGINT | 是 | 自增主键 |
| `case_id` | VARCHAR(64) | 是 | 案例唯一 ID |
| `diagnosis_id` | VARCHAR(64) | 否 | 关联诊断会话;当前自动生成时存 `diagnosis_session.session_id` |
| `source_type` | VARCHAR(16) | 否 | 来源类型:`AUTO` 或 `MANUAL` |
| `fault_category` | VARCHAR(32) | 否 | 故障类别,实体侧使用 `FaultCategory` |
| `fault_source` | VARCHAR(128) | 否 | 故障源,例如服务、系统或省份 |
| `fault_target` | VARCHAR(256) | 否 | 故障目标,例如接口、方法、SQL 或组件 |
| `error_code` | VARCHAR(64) | 否 | 错误码或异常类型 |
| `title` | VARCHAR(256) | 是 | 案例标题 |
| `root_cause` | TEXT | 是 | 根因分析 |
| `solution` | TEXT | 是 | 解决方案 |
| `reference_count` | INT | 否 | 被推荐次数 |
| `created_by` | VARCHAR(64) | 否 | 创建人 |
| `created_at` | DATETIME | 是 | 创建时间 |
| `updated_at` | DATETIME | 是 | 更新时间 |
## 索引
| 索引 | 字段 | 用途 |
|---|---|---|
| `idx_fault_category` | `fault_category` | 按故障类别筛选 |
| `idx_error_code` | `error_code` | 按错误码精确匹配 |
| `idx_fault_source` | `fault_source` | 按故障源筛选 |
| `idx_fault_target` | `fault_target(100)` | 按故障目标筛选 |
| `idx_diagnosis_id` | `diagnosis_id` | 追溯来源会话 |
| `idx_reference_count` | `reference_count` | 推荐排序 |
| `idx_created_at` | `created_at` | 时间排序 |
## 关系
- `case_library.diagnosis_id` 当前逻辑关联 `diagnosis_session.session_id`,不是旧的 `diagnosis_record`。
- 人工录入案例可以不填写 `diagnosis_id`。
## 注意点
- 旧文档里提到的 `diagnosis_record` 已被 `V007` 删除,不再是当前主模型。
- 当前自动沉淀仍比较粗:`root_cause` 和 `solution` 都可能来自完整 answer。后续可从结构化结论中拆分根因、证据和修复建议。
@@ -0,0 +1,36 @@
# 知识域表:knowledge_domain
**状态**:当前表
**来源**:`V009__add_knowledge_domain.sql`、`KnowledgeDomain`
## 定位
`knowledge_domain` 保存知识库领域级元数据,用来帮助 Planner/Executor 判断什么时候检索某一类知识,并为 RAG 的 domain hint、去重和可观测性提供基础信息。
## 字段
| 字段 | 类型 | 必填 | 说明 |
|---|---|---|---|
| `id` | BIGINT | 是 | 自增主键 |
| `domain_id` | VARCHAR(64) | 是 | 领域 ID,通常对应文档 category,例如 `payment`、`infrastructure` |
| `description` | VARCHAR(256) | 否 | 领域描述 |
| `when_to_retrieve` | TEXT | 否 | 何时检索该领域的提示说明 |
| `document_count` | INT | 是 | 当前领域文档数量 |
| `created_at` | DATETIME | 是 | 创建时间 |
| `updated_at` | DATETIME | 是 | 更新时间 |
## 索引
| 索引 | 字段 | 用途 |
|---|---|---|
| unique | `domain_id` | 保证领域 ID 唯一 |
## 关系
- `knowledge_domain.domain_id` 与 `api_document.metadata` 或向量库 chunk metadata 中的 `category` 逻辑关联。
- 当前没有数据库外键,领域文档数量由服务逻辑维护。
## 注意点
- `when_to_retrieve` 是检索策略提示,不是事实证据。
- Executor / Verifier 不能把领域描述当作诊断结论依据;事实仍应来自工具返回的证据块或证据引用。
@@ -0,0 +1,47 @@
# 诊断会话表:diagnosis_session
**状态**:当前主表
**来源**:`V005__create_session_storage.sql`、`V008__add_answer_to_diagnosis_session.sql`、`DiagnosisSession`
## 定位
`diagnosis_session` 是一次 Chat 或 AIOps 诊断的会话级主记录,负责保存用户问题、执行状态、最终答案、总体统计和自评估结果。
## 字段
| 字段 | 类型 | 必填 | 说明 |
|---|---|---|---|
| `id` | BIGINT | 是 | 自增主键 |
| `session_id` | VARCHAR(64) | 是 | 会话唯一 ID,Trace API 和反馈接口使用它 |
| `query` | TEXT | 是 | 用户原始问题或 AIOps 输入摘要 |
| `status` | VARCHAR(16) | 否 | `PENDING`、`RUNNING`、`SUCCESS`、`FAILED` |
| `agent_flow` | VARCHAR(32) | 否 | `CHAT` 或 `AI_OPS` |
| `total_duration_ms` | INT | 否 | 总耗时,单位毫秒 |
| `total_token_count` | INT | 否 | 总 Token 消耗 |
| `step_count` | INT | 否 | Agent 步骤数 |
| `tool_call_count` | INT | 否 | 工具调用次数 |
| `answer` | LONGTEXT | 否 | 返回给用户的最终答案 |
| `self_evaluation` | JSON | 否 | rule、verifier、aiops 等自评估结果容器 |
| `feedback` | VARCHAR(16) | 否 | 用户反馈:`useful`、`not_useful` 或空 |
| `created_at` | DATETIME | 是 | 创建时间 |
| `updated_at` | DATETIME | 是 | 更新时间 |
## 索引
| 索引 | 字段 | 用途 |
|---|---|---|
| `session_id` unique | `session_id` | 会话唯一约束 |
| `idx_created_at` | `created_at` | 按时间查询 |
| `idx_status` | `status` | 按状态筛选 |
| `idx_agent_flow` | `agent_flow` | 区分 Chat / AIOps |
## 关系
- `agent_step.session_id` 逻辑关联 `diagnosis_session.session_id`。
- `tool_invocation.session_id` 逻辑关联 `diagnosis_session.session_id`。
- `case_library.diagnosis_id` 在自动生成案例时保存 `diagnosis_session.session_id`。
## 注意点
- 当前没有数据库外键,Trace 聚合依赖 `session_id`。
- `self_evaluation` 是扩展容器,里面可能包含 `rule_evaluation`、`verifier_evaluation`、`aiops_rule_evaluation`。
@@ -30,7 +30,7 @@
| Source | Alignment |
|---|---|
| Issue objective | Stage four in `mvp/issues/executor-structured-output-v2.md` requires Composer output final answer from Verifier-allowed material. Covered by proposal, design, specs, and tasks. |
| Issue objective | Stage four in `mvp/issues/design-notes/executor-structured-output-v2.md` requires Composer output final answer from Verifier-allowed material. Covered by proposal, design, specs, and tasks. |
| Proposal -> design | Proposal says Composer owns final expression; design defines input filtering, output parsing, fallback, and audit. |
| Design -> specs | Design decisions are reflected in `chat-composer-agent` requirements and modified `chat-verifier-agent` routing requirements. |
| Specs -> tasks | Each required behavior has implementation and test tasks, including malformed fallback and no raw output leakage. |
@@ -2,7 +2,7 @@
## Discover Context
- Source issue: `mvp/issues/ISS-007-verifier-evidence-summary-fidelity.md`.
- Source issue: `mvp/issues/archived/ISS-007-verifier-evidence-summary-fidelity.md`.
- Related existing specs: `chat-verifier-agent`, `evidence-trace-hardening`.
- Related devflow records: `executor-gatekeeper-hook`, `executor-verifier-claim-checks`, `executor-composer-final-answer`, `evidence-trace-hardening`.
- Current repo instruction requested semantic code search and LSP confirmation before code changes; those tools are not exposed in this environment, so implementation will use `rg`, direct code reading, and focused tests as fallback evidence.
@@ -0,0 +1 @@
committed
@@ -0,0 +1,117 @@
# Design: Interview Demo Quality Audit
## Overview
The change adds auditability and demo readiness without changing Agent routing.
```text
ChatService prompt resources
-> PromptAuditService
-> verifier_evaluation.prompt_audit
-> Trace API
-> DiagnosisTraceEvaluator
-> baseline report
mvp/demo/scripts/run-interview-demo-check.ps1
-> health/readiness check
-> payment timeout chat
-> trace fetch
-> feedback
-> demo output bundle
```
## Prompt Audit
Add a compact `prompt_audit` object under:
```text
diagnosis_session.self_evaluation.verifier_evaluation.prompt_audit
```
Shape:
```json
{
"version": "chat-prompts-v1",
"prompts": [
{
"name": "chat_planner",
"version": "chat-planner-v1",
"resource": "prompts/chat-planner-prompt.md"
}
]
}
```
Design choices:
- Use explicit local metadata, not full prompt hashes, because the interview goal is explainable version audit rather than cryptographic integrity.
- Keep this metadata in code or a small resource catalog near prompt loading.
- Persist prompt audit with every Chat verifier evaluation, including fallback/degraded paths.
- Do not include full prompt text in trace.
## Eval Expansion
Extend `DiagnosisEvalCase` with optional fields:
- `requirePromptAudit`
- `expectedPromptAuditVersion`
- `expectedPromptVersions`
- `requireGatekeeperRules`
Evaluator behavior:
- If `requirePromptAudit=true`, `verifier_evaluation.prompt_audit.version` must exist.
- If `expectedPromptAuditVersion` is set, it must match.
- If `expectedPromptVersions` is set, each listed prompt name/version pair must exist.
- If `requireGatekeeperRules=true`, `gatekeeper_result.rules` must be a non-empty list and each item must include `id`, `enabled`, and `default_severity`.
Add at least two fixture-backed cases:
- A positive audit closure case that requires prompt audit + Gatekeeper rules.
- A metadata-gap negative case represented as `LOW_CONFID`/safe final answer, used to prove the evaluator catches missing audit metadata when configured.
The baseline must remain fully passing after fixtures are updated.
## Demo Stabilization
Add a PowerShell script:
```text
mvp/demo/scripts/run-interview-demo-check.ps1
```
Responsibilities:
- Accept base URL and session id parameters.
- Check that the service is reachable.
- Run the existing payment-timeout chat request.
- Fetch trace for the same session id.
- Submit useful feedback.
- Write outputs under `mvp/demo/output/`.
- Emit a concise summary with session id, verdict, Gatekeeper rule version, prompt audit version, and output paths.
The script should fail fast with actionable messages when the service is unavailable.
## Documentation
Add/update:
- `mvp/demo/README.md`: mention the preflight script.
- `mvp/demo/ten-minute-interview-demo.md`: use the preflight script as the recommended path.
- `mvp/demo/interview-q-and-a.md`: concise interview answers for Agent engineering tradeoffs.
- `mvp/architecture/harness-quality-gates.md`: record prompt audit as part of the quality gate.
## Verification
Required:
- Targeted unit/eval tests for prompt audit persistence and evaluator checks.
- Regenerated baseline JSON/Markdown reports.
- OpenSpec validation.
E2E:
- If local dependencies are available, run Spring Boot with `mvp-demo` profile and execute the new preflight script.
- If unavailable, record the reason and rely on deterministic unit/eval evidence.
@@ -0,0 +1,58 @@
# Change: Interview Demo Quality Audit
## Problem
SuperBizAgent MVP is now strong enough to demonstrate traceable Agent engineering, but the interview path still has three gaps:
- The live demo has a run script, but no preflight command that checks service readiness and produces a concise interview evidence bundle.
- The diagnosis eval baseline covers the V2 evidence pipeline, but it does not yet assert prompt-version audit data and has limited coverage for audit metadata gaps.
- Gatekeeper already exposes `rule_set_version`, but prompt versions are not persisted with the verifier evaluation, making prompt changes harder to explain, compare, and roll back in an interview.
This change stabilizes the MVP as an interview artifact rather than adding a new diagnosis architecture.
## Proposed Solution
Implement a small internal quality/audit increment:
1. Add prompt version audit metadata to Chat verifier evaluation.
2. Extend deterministic diagnosis eval cases/fixtures to assert prompt audit metadata and audit-metadata failures.
3. Add an interview demo preflight script and documentation that can be run before or during a demo to verify service readiness, execute the payment timeout path, fetch trace, and record key audit fields.
4. Add/update MVP documentation for interview Q&A and the new audit/preflight workflow.
## Scope
In scope:
- Internal `diagnosis_session.self_evaluation.verifier_evaluation` audit JSON.
- Diagnosis eval case schema, evaluator checks, fixtures, and baseline reports.
- MVP demo scripts/docs.
- Architecture/demo documentation for prompt and Gatekeeper version audit.
Out of scope:
- Public HTTP API changes.
- Database schema changes.
- New Agent roles, MCP tool server migration, process isolation, or AIOps LLM Verifier.
- Replacing existing `Planner -> Executor -> Gatekeeper -> Verifier -> Composer` orchestration.
- Guaranteeing live LLM `PASS` for every demo run. Live demo compatibility and deterministic fixture regression are separate acceptance paths.
## Context Constraints From devflow
- Evidence Tools produce incident facts and must be recorded in `tool_invocation`.
- Chat quality gates are layered: Gatekeeper verifies evidence references, Verifier judges derivability, Composer controls expression.
- `diagnosis eval` is deterministic and fixture-backed; no LLM-as-judge.
- Demo assets should be runnable, but interview safety should not depend solely on live LLM behavior.
- Gatekeeper rule metadata is metadata-only; dynamic rule execution is out of scope.
## Interface Impact
Level: L2 internal contract change.
Reason: `verifier_evaluation` gains a compact `prompt_audit` object. Existing public API shape remains the same, and the value is exposed only through already-existing trace/self-evaluation JSON.
## Risks
- Baseline report churn is expected when adding cases; JSON and Markdown reports must be regenerated together.
- Prompt audit must be deterministic and stable enough for eval fixtures; avoid hashing full prompt text with environment-specific content.
- Demo preflight must not hardcode secrets and must tolerate local service unavailability with clear failure messages.
@@ -0,0 +1,16 @@
## MODIFIED Requirements
### Requirement: Verifier SHALL be observable
The Verifier's verdict SHALL be persisted for observability.
#### Scenario: prompt audit written to verifier evaluation
- **WHEN** the Chat verifier evaluation is persisted
- **THEN** the system SHALL include a `prompt_audit` object under `diagnosis_session.self_evaluation.verifier_evaluation`
- **AND** `prompt_audit.version` SHALL identify the Chat prompt audit catalog version
- **AND** `prompt_audit.prompts` SHALL include the planner, executor, verifier, and composer prompt names and versions
- **AND** full prompt text SHALL NOT be persisted in `prompt_audit`
#### Scenario: prompt audit available on fallback paths
- **WHEN** Chat verifier parsing fails, Composer parsing fails, or Chat produces a degraded answer
- **THEN** the persisted verifier evaluation SHALL still include `prompt_audit`
@@ -0,0 +1,21 @@
## MODIFIED Requirements
### Requirement: Diagnosis eval SHALL validate trace fixtures deterministically
The diagnosis eval harness SHALL evaluate saved trace fixtures without invoking an LLM judge.
#### Scenario: prompt audit assertions are enforced
- **WHEN** an eval case sets `requirePromptAudit=true`
- **THEN** the evaluator SHALL require `verifier_evaluation.prompt_audit.version`
- **AND** when `expectedPromptAuditVersion` is configured, it SHALL match exactly
- **AND** when `expectedPromptVersions` is configured, each configured prompt name SHALL appear with the expected version
#### Scenario: Gatekeeper rule metadata assertions are enforced
- **WHEN** an eval case sets `requireGatekeeperRules=true`
- **THEN** the evaluator SHALL require `verifier_evaluation.gatekeeper_result.rules` to be non-empty
- **AND** each rule item SHALL include `id`, `enabled`, and `default_severity`
#### Scenario: expanded baseline remains passing
- **WHEN** the committed fixture set is evaluated
- **THEN** every case SHALL pass
- **AND** baseline JSON and Markdown reports SHALL reflect the expanded case count and verdict distribution
@@ -0,0 +1,22 @@
## MODIFIED Requirements
### Requirement: MVP demo SHALL be reproducible for interviews
The MVP demo SHALL provide a repeatable way to show a diagnosis answer, trace, verifier evaluation, and feedback.
#### Scenario: interview demo check script records an evidence bundle
- **WHEN** the user runs the interview demo check script against a running `mvp-demo` service
- **THEN** the script SHALL submit a fixed Chat diagnosis request
- **AND** it SHALL fetch the trace for the same session id
- **AND** it SHALL submit useful feedback for that session
- **AND** it SHALL write chat, trace, feedback, and summary outputs under `mvp/demo/output/`
#### Scenario: interview demo check fails with actionable readiness output
- **WHEN** the target service is not reachable
- **THEN** the script SHALL fail before issuing diagnosis requests
- **AND** the failure message SHALL name the base URL and the expected startup profile
#### Scenario: interview documentation explains audit fields
- **WHEN** an interviewer asks how prompt or Gatekeeper changes are audited
- **THEN** the demo documentation SHALL point to `prompt_audit.version` and `gatekeeper_result.rule_set_version`
- **AND** it SHALL explain that deterministic eval fixtures are the regression source of truth
@@ -0,0 +1,32 @@
## 1. OpenSpec And devflow
- [x] 1.1 Create OpenSpec proposal/design/spec/tasks for `interview-demo-quality-audit`.
- [x] 1.2 Record context, question pool, interface impact, audit, and verification plan in devflow decisions.
- [x] 1.3 Pass OpenSpec validation and create `.committed`.
## 2. Prompt/Gatekeeper Version Audit
- [x] 2.1 Add compact Chat prompt audit metadata for planner, executor, verifier, and composer prompts.
- [x] 2.2 Persist `prompt_audit` under `verifier_evaluation` for Chat verifier/composer outcomes.
- [x] 2.3 Add focused tests proving prompt audit appears in persisted verifier evaluation.
- [x] 2.4 Extend eval checks for prompt audit and Gatekeeper rule metadata.
## 3. Eval Expansion
- [x] 3.1 Extend diagnosis eval case/result schema for prompt audit fields.
- [x] 3.2 Add fixture-backed cases for audit closure coverage.
- [x] 3.3 Regenerate baseline JSON and Markdown reports.
- [x] 3.4 Update eval docs/schema.
## 4. Interview Demo Stabilization
- [x] 4.1 Add `run-interview-demo-check.ps1` with service preflight, chat, trace, feedback, and summary output.
- [x] 4.2 Update demo README and 10-minute script to use the preflight path.
- [x] 4.3 Add interview Q&A documentation focused on Agent engineering tradeoffs.
## 5. Verification And Archive
- [x] 5.1 Run targeted tests for ChatService/prompt audit and diagnosis eval.
- [x] 5.2 Run relevant broader regression tests.
- [x] 5.3 Run E2E demo check with `mvp-demo` profile if dependencies are available; otherwise record the blocker.
- [x] 5.4 Archive the OpenSpec change, update devflow artifacts, and commit implementation + archive.
@@ -145,6 +145,17 @@ The Verifier's verdict and downstream final-answer composition SHALL be persiste
- **THEN** `diagnosis_session.self_evaluation.verifier_evaluation` SHALL include `gatekeeper_result`
- **AND** existing verifier fields such as `verdict`, `facts_checked`, `executor_output_parse_status`, and `tool_trace_summary` SHALL be preserved
#### Scenario: prompt audit written to verifier evaluation
- **WHEN** the Chat verifier evaluation is persisted
- **THEN** the system SHALL include a `prompt_audit` object under `diagnosis_session.self_evaluation.verifier_evaluation`
- **AND** `prompt_audit.version` SHALL identify the Chat prompt audit catalog version
- **AND** `prompt_audit.prompts` SHALL include the planner, executor, verifier, and composer prompt names and versions
- **AND** full prompt text SHALL NOT be persisted in `prompt_audit`
#### Scenario: prompt audit available on fallback paths
- **WHEN** Chat verifier parsing fails, Composer parsing fails, or Chat produces a degraded answer
- **THEN** the persisted verifier evaluation SHALL still include `prompt_audit`
### Requirement: self_evaluation SHALL be a container object
The `diagnosis_session.self_evaluation` field SHALL store multiple evaluation channels in one JSON object.
@@ -195,3 +195,22 @@ The evaluation harness SHALL be able to assert the Gatekeeper rule set version r
- **WHEN** an evaluation case declares `expectedGatekeeperRuleSetVersion`
- **AND** the fixture has a different or missing rule set version
- **THEN** the case SHALL fail with a clear failed check
### Requirement: Diagnosis eval SHALL validate trace fixtures deterministically
The diagnosis eval harness SHALL evaluate saved trace fixtures without invoking an LLM judge.
#### Scenario: prompt audit assertions are enforced
- **WHEN** an eval case sets `requirePromptAudit=true`
- **THEN** the evaluator SHALL require `verifier_evaluation.prompt_audit.version`
- **AND** when `expectedPromptAuditVersion` is configured, it SHALL match exactly
- **AND** when `expectedPromptVersions` is configured, each configured prompt name SHALL appear with the expected version
#### Scenario: Gatekeeper rule metadata assertions are enforced
- **WHEN** an eval case sets `requireGatekeeperRules=true`
- **THEN** the evaluator SHALL require `verifier_evaluation.gatekeeper_result.rules` to be non-empty
- **AND** each rule item SHALL include `id`, `enabled`, and `default_severity`
#### Scenario: expanded baseline remains passing
- **WHEN** the committed fixture set is evaluated
- **THEN** every case SHALL pass
- **AND** baseline JSON and Markdown reports SHALL reflect the expanded case count and verdict distribution
@@ -56,6 +56,26 @@ The MVP demo SHALL provide scripts and request payloads for running the payment-
- **WHEN** the demo script finishes successfully
- **THEN** it SHALL write chat, trace, and feedback responses under a demo output directory
### Requirement: MVP demo SHALL be reproducible for interviews
The MVP demo SHALL provide a repeatable way to show a diagnosis answer, trace, verifier evaluation, and feedback.
#### Scenario: interview demo check script records an evidence bundle
- **WHEN** the user runs the interview demo check script against a running `mvp-demo` service
- **THEN** the script SHALL submit a fixed Chat diagnosis request
- **AND** it SHALL fetch the trace for the same session id
- **AND** it SHALL submit useful feedback for that session
- **AND** it SHALL write chat, trace, feedback, and summary outputs under `mvp/demo/output/`
#### Scenario: interview demo check fails with actionable readiness output
- **WHEN** the target service is not reachable
- **THEN** the script SHALL fail before issuing diagnosis requests
- **AND** the failure message SHALL name the base URL and the expected startup profile
#### Scenario: interview documentation explains audit fields
- **WHEN** an interviewer asks how prompt or Gatekeeper changes are audited
- **THEN** the demo documentation SHALL point to `prompt_audit.version` and `gatekeeper_result.rule_set_version`
- **AND** it SHALL explain that deterministic eval fixtures are the regression source of truth
### Requirement: MVP demo SHALL provide a trace inspection checklist
The MVP demo SHALL document which trace fields to inspect for evidence, verifier behavior, and session-level auditability.
@@ -6,6 +6,7 @@ import lombok.Data;
import lombok.NoArgsConstructor;
import java.util.List;
import java.util.Map;
@Data
@Builder
@@ -29,4 +30,8 @@ public class DiagnosisEvalCase {
private String expectedGatekeeperRuleSetVersion;
private List<String> expectedComposerStatuses;
private List<String> forbiddenConfirmedClaimKeywords;
private Boolean requirePromptAudit;
private String expectedPromptAuditVersion;
private Map<String, String> expectedPromptVersions;
private Boolean requireGatekeeperRules;
}
@@ -46,8 +46,8 @@ public class DiagnosisEvalReportWriter {
}
builder.append("## Cases\n\n");
builder.append("| Case | Result | Verdict | Gatekeeper | Rule Set | Composer | Claim Checks | Keywords | Tool Calls | Duration ms | Failed Checks |\n");
builder.append("| --- | --- | --- | --- | --- | --- | ---: | --- | ---: | ---: | --- |\n");
builder.append("| Case | Result | Verdict | Gatekeeper | Rule Set | Prompt Audit | Composer | Claim Checks | Rules | Keywords | Tool Calls | Duration ms | Failed Checks |\n");
builder.append("| --- | --- | --- | --- | --- | --- | --- | ---: | ---: | --- | ---: | ---: | --- |\n");
for (DiagnosisEvalResult result : report.getResults()) {
builder.append("| ")
.append(result.getCaseId())
@@ -60,10 +60,14 @@ public class DiagnosisEvalReportWriter {
.append(" | ")
.append(valueOrDash(result.getGatekeeperRuleSetVersion()))
.append(" | ")
.append(valueOrDash(result.getPromptAuditVersion()))
.append(" | ")
.append(valueOrDash(result.getComposerStatus()))
.append(" | ")
.append(result.getClaimCheckCount() == null ? "-" : result.getClaimCheckCount())
.append(" | ")
.append(result.getGatekeeperRuleCount() == null ? "-" : result.getGatekeeperRuleCount())
.append(" | ")
.append(result.getMatchedKeywordCount()).append("/").append(result.getRequiredKeywordCount())
.append(" | ")
.append(result.getToolCallCount() == null ? "-" : result.getToolCallCount())
@@ -24,8 +24,10 @@ public class DiagnosisEvalResult {
private Map<String, Boolean> evidenceCoverage;
private String gatekeeperStatus;
private String gatekeeperRuleSetVersion;
private String promptAuditVersion;
private String composerStatus;
private Integer claimCheckCount;
private Integer gatekeeperRuleCount;
private Integer toolCallCount;
private Integer durationMs;
}
@@ -67,8 +67,10 @@ public class DiagnosisTraceEvaluator {
.evidenceCoverage(emptyCoverage(evalCase.getRequiredEvidenceTools()))
.gatekeeperStatus(null)
.gatekeeperRuleSetVersion(null)
.promptAuditVersion(null)
.composerStatus(null)
.claimCheckCount(null)
.gatekeeperRuleCount(null)
.toolCallCount(null)
.durationMs(null)
.build());
@@ -123,10 +125,13 @@ public class DiagnosisTraceEvaluator {
String gatekeeperStatus = extractNestedString(trace, "verifier_evaluation", "gatekeeper_result", "status");
String gatekeeperRuleSetVersion = extractNestedString(trace, "verifier_evaluation",
"gatekeeper_result", "rule_set_version");
String promptAuditVersion = extractNestedString(trace, "verifier_evaluation",
"prompt_audit", "version");
String composerStatus = extractNestedString(trace, "verifier_evaluation", "composer_output", "status");
Integer claimCheckCount = countList(trace, "verifier_evaluation", "claim_checks");
Integer gatekeeperRuleCount = countNestedList(trace, "verifier_evaluation", "gatekeeper_result", "rules");
failedChecks.addAll(validateV2AuditClosure(evalCase, trace, normalizedAnswer, verdict,
gatekeeperStatus, gatekeeperRuleSetVersion, composerStatus));
gatekeeperStatus, gatekeeperRuleSetVersion, promptAuditVersion, composerStatus));
Integer toolCallCount = trace.getToolInvocations() == null ? 0 : trace.getToolInvocations().size();
Integer durationMs = trace.getSession() == null ? null : trace.getSession().getTotalDurationMs();
@@ -142,8 +147,10 @@ public class DiagnosisTraceEvaluator {
.evidenceCoverage(evidenceCoverage)
.gatekeeperStatus(gatekeeperStatus)
.gatekeeperRuleSetVersion(gatekeeperRuleSetVersion)
.promptAuditVersion(promptAuditVersion)
.composerStatus(composerStatus)
.claimCheckCount(claimCheckCount)
.gatekeeperRuleCount(gatekeeperRuleCount)
.toolCallCount(toolCallCount)
.durationMs(durationMs)
.build();
@@ -237,6 +244,7 @@ public class DiagnosisTraceEvaluator {
String verdict,
String gatekeeperStatus,
String gatekeeperRuleSetVersion,
String promptAuditVersion,
String composerStatus) {
List<String> failedChecks = new ArrayList<>();
boolean requireV2AuditClosure = Boolean.TRUE.equals(evalCase.getRequireV2AuditClosure());
@@ -259,6 +267,10 @@ public class DiagnosisTraceEvaluator {
failedChecks.add("gatekeeper rule set version not expected: "
+ valueOrMissing(gatekeeperRuleSetVersion));
}
if (Boolean.TRUE.equals(evalCase.getRequireGatekeeperRules())) {
failedChecks.addAll(validateGatekeeperRules(trace));
}
failedChecks.addAll(validatePromptAudit(evalCase, trace, promptAuditVersion));
failedChecks.addAll(validateClaimChecks(trace, requireClaimChecks));
@@ -290,6 +302,78 @@ public class DiagnosisTraceEvaluator {
return failedChecks;
}
private List<String> validatePromptAudit(DiagnosisEvalCase evalCase,
DiagnosisTraceResponse trace,
String promptAuditVersion) {
List<String> failedChecks = new ArrayList<>();
Object promptAudit = nestedValue(trace, "verifier_evaluation", "prompt_audit");
if (Boolean.TRUE.equals(evalCase.getRequirePromptAudit()) && !(promptAudit instanceof Map<?, ?>)) {
failedChecks.add("missing prompt_audit");
return failedChecks;
}
if (!isBlank(evalCase.getExpectedPromptAuditVersion())
&& !evalCase.getExpectedPromptAuditVersion().equals(promptAuditVersion)) {
failedChecks.add("prompt audit version not expected: " + valueOrMissing(promptAuditVersion));
}
if (evalCase.getExpectedPromptVersions() == null || evalCase.getExpectedPromptVersions().isEmpty()) {
return failedChecks;
}
if (!(promptAudit instanceof Map<?, ?> audit)) {
failedChecks.add("missing prompt_audit");
return failedChecks;
}
Object promptsValue = audit.get("prompts");
if (!(promptsValue instanceof List<?> prompts)) {
failedChecks.add("prompt_audit missing prompts");
return failedChecks;
}
Map<String, String> actualVersions = new LinkedHashMap<>();
for (Object promptValue : prompts) {
if (promptValue instanceof Map<?, ?> prompt) {
String name = stringValue(prompt.get("name"));
String version = stringValue(prompt.get("version"));
if (!isBlank(name)) {
actualVersions.put(name, version);
}
}
}
for (Map.Entry<String, String> expected : evalCase.getExpectedPromptVersions().entrySet()) {
String actual = actualVersions.get(expected.getKey());
if (!expected.getValue().equals(actual)) {
failedChecks.add("prompt version not expected: "
+ expected.getKey() + "=" + valueOrMissing(actual));
}
}
return failedChecks;
}
private List<String> validateGatekeeperRules(DiagnosisTraceResponse trace) {
Object rulesValue = nestedNestedValue(trace, "verifier_evaluation", "gatekeeper_result", "rules");
if (!(rulesValue instanceof List<?> rules) || rules.isEmpty()) {
return List.of("gatekeeper_result missing rules");
}
List<String> failedChecks = new ArrayList<>();
for (Object item : rules) {
if (!(item instanceof Map<?, ?> rule)) {
failedChecks.add("gatekeeper rule metadata is not an object");
continue;
}
String id = stringValue(rule.get("id"));
Object enabled = rule.get("enabled");
String severity = stringValue(rule.get("default_severity"));
if (isBlank(id)) {
failedChecks.add("gatekeeper rule metadata missing id");
}
if (!(enabled instanceof Boolean)) {
failedChecks.add("gatekeeper rule metadata missing enabled: " + valueOrMissing(id));
}
if (isBlank(severity)) {
failedChecks.add("gatekeeper rule metadata missing default_severity: " + valueOrMissing(id));
}
}
return failedChecks;
}
private List<String> validateClaimChecks(DiagnosisTraceResponse trace, boolean required) {
Object claimChecks = nestedValue(trace, "verifier_evaluation", "claim_checks");
if (!(claimChecks instanceof List<?> claimCheckList)) {
@@ -347,6 +431,19 @@ public class DiagnosisTraceEvaluator {
return value instanceof List<?> list ? list.size() : null;
}
private Integer countNestedList(DiagnosisTraceResponse trace, String firstKey, String secondKey, String thirdKey) {
Object value = nestedNestedValue(trace, firstKey, secondKey, thirdKey);
return value instanceof List<?> list ? list.size() : null;
}
private Object nestedNestedValue(DiagnosisTraceResponse trace, String firstKey, String secondKey, String thirdKey) {
Object value = nestedValue(trace, firstKey, secondKey);
if (!(value instanceof Map<?, ?> map)) {
return null;
}
return map.get(thirdKey);
}
private int countMatches(String normalizedAnswer, List<String> keywords) {
int count = 0;
for (String keyword : safeList(keywords)) {
@@ -60,6 +60,7 @@ public class ChatService {
private static final Logger logger = LoggerFactory.getLogger(ChatService.class);
private static final String LOW_CONFID_DISCLAIMER = "以下结论基于当前已获取证据,仍存在部分证据缺口,请谨慎参考。";
private static final String DEGRADED_PREFIX = "当前无法基于已获取证据生成可靠结论,建议人工介入。";
private static final String CHAT_PROMPT_AUDIT_VERSION = "chat-prompts-v1";
/** 封装 answer + 后端生成的 sessionId,用于 feedback 关联 */
public record ChatResult(String answer, String sessionId) {}
@@ -884,6 +885,7 @@ public class ChatService {
verifierEvaluation.put("rationale", decision.rationale());
verifierEvaluation.put("round", round);
verifierEvaluation.put("traceability_version", "v1");
verifierEvaluation.put("prompt_audit", promptAuditSnapshot());
verifierEvaluation.put("executor_output_parse_status",
Optional.ofNullable(VerifierContextHolder.getExecutorOutputParseStatus())
.orElse(Map.of("status", "missing", "detail", "executor parse status unavailable")));
@@ -902,6 +904,26 @@ public class ChatService {
diagnosisSessionRepository.save(session);
}
private Map<String, Object> promptAuditSnapshot() {
Map<String, Object> audit = new LinkedHashMap<>();
audit.put("version", CHAT_PROMPT_AUDIT_VERSION);
audit.put("prompts", List.of(
promptAuditItem("chat_planner", "chat-planner-v1", "prompts/chat-planner-prompt.md"),
promptAuditItem("chat_executor", "chat-executor-v2", "prompts/chat-executor-prompt.md"),
promptAuditItem("chat_verifier", "chat-verifier-v2", "prompts/chat-verifier-prompt.md"),
promptAuditItem("chat_composer", "chat-composer-v1", "prompts/chat-composer-prompt.md")
));
return audit;
}
private Map<String, Object> promptAuditItem(String name, String version, String resource) {
Map<String, Object> item = new LinkedHashMap<>();
item.put("name", name);
item.put("version", version);
item.put("resource", resource);
return item;
}
private Map<String, Object> defaultGatekeeperPass() {
GatekeeperRuleCatalog catalog = GatekeeperRuleCatalog.fallback();
return Map.of(
@@ -100,13 +100,13 @@ class DiagnosisEvalBaselineDiffTest {
}
private void degradeRedisCase(DiagnosisEvalReport report) {
report.setPassedCases(9);
report.setPassRate(0.9);
report.setPassedCases(11);
report.setPassRate(11.0 / 12.0);
report.setAverageToolCallCount(3.0);
report.setAverageDurationMs(39800.0);
report.setAverageDurationMs(38500.0);
report.setVerdictDistribution(new LinkedHashMap<>());
report.getVerdictDistribution().put("PASS", 4L);
report.getVerdictDistribution().put("LOW_CONFID", 4L);
report.getVerdictDistribution().put("PASS", 5L);
report.getVerdictDistribution().put("LOW_CONFID", 5L);
report.getVerdictDistribution().put("REJECT", 2L);
DiagnosisEvalResult redis = result(report, "redis-timeout");
@@ -24,11 +24,11 @@ class DiagnosisTraceEvaluatorTest {
DiagnosisEvalReport report = evaluator.evaluate(cases, Path.of("mvp/eval/fixtures"));
assertEquals(10, report.getTotalCases());
assertEquals(10, report.getPassedCases());
assertEquals(12, report.getTotalCases());
assertEquals(12, report.getPassedCases());
assertEquals(1.0, report.getPassRate(), 0.001);
assertEquals(4L, report.getVerdictDistribution().get("PASS"));
assertEquals(5L, report.getVerdictDistribution().get("LOW_CONFID"));
assertEquals(5L, report.getVerdictDistribution().get("PASS"));
assertEquals(6L, report.getVerdictDistribution().get("LOW_CONFID"));
assertEquals(1L, report.getVerdictDistribution().get("REJECT"));
DiagnosisEvalResult narrowHighCpu = result(report, "narrow-highcpu-observation");
@@ -36,6 +36,12 @@ class DiagnosisTraceEvaluatorTest {
assertEquals("gatekeeper-rules-v1", narrowHighCpu.getGatekeeperRuleSetVersion());
assertEquals("pass", narrowHighCpu.getGatekeeperStatus());
DiagnosisEvalResult promptGatekeeperAudit = result(report, "prompt-gatekeeper-audit-closure");
assertTrue(promptGatekeeperAudit.isPassed());
assertEquals("chat-prompts-v1", promptGatekeeperAudit.getPromptAuditVersion());
assertEquals("gatekeeper-rules-v1", promptGatekeeperAudit.getGatekeeperRuleSetVersion());
assertEquals(2, promptGatekeeperAudit.getGatekeeperRuleCount());
DiagnosisEvalResult hikariNoEvidence = result(report, "hikari-no-evidence-negative-observation");
assertTrue(hikariNoEvidence.isPassed());
assertEquals("gatekeeper-rules-v1", hikariNoEvidence.getGatekeeperRuleSetVersion());
@@ -60,6 +66,12 @@ class DiagnosisTraceEvaluatorTest {
DiagnosisEvalResult composerFallback = result(report, "composer-fallback-no-raw-json");
assertTrue(composerFallback.isPassed());
assertEquals("composer_malformed", composerFallback.getComposerStatus());
DiagnosisEvalResult auditMetadataLowConfid = result(report, "audit-metadata-low-confid");
assertTrue(auditMetadataLowConfid.isPassed());
assertEquals("LOW_CONFID", auditMetadataLowConfid.getVerdict());
assertEquals("chat-prompts-v1", auditMetadataLowConfid.getPromptAuditVersion());
assertEquals(2, auditMetadataLowConfid.getGatekeeperRuleCount());
}
@Test
@@ -195,6 +207,76 @@ class DiagnosisTraceEvaluatorTest {
"gatekeeper rule set version not expected: old-rules"));
}
@Test
void evaluateFailsWhenPromptAuditMissingOrMismatches() {
DiagnosisEvalCase evalCase = DiagnosisEvalCase.builder()
.id("prompt-audit")
.title("Prompt audit")
.expectedRootCauseKeywords(List.of())
.requiredEvidenceTools(List.of())
.allowedVerdicts(List.of("PASS"))
.requirePromptAudit(true)
.expectedPromptAuditVersion("chat-prompts-v1")
.expectedPromptVersions(java.util.Map.of("chat_executor", "chat-executor-v2"))
.build();
DiagnosisTraceResponse trace = DiagnosisTraceResponse.builder()
.session(DiagnosisTraceResponse.SessionTrace.builder()
.answer("安全回答")
.selfEvaluation(java.util.Map.of(
"verifier_evaluation", java.util.Map.of(
"verdict", "PASS",
"prompt_audit", java.util.Map.of(
"version", "old-prompts",
"prompts", java.util.List.of(java.util.Map.of(
"name", "chat_executor",
"version", "chat-executor-v1"
))
)
)))
.build())
.toolInvocations(List.of())
.build();
DiagnosisEvalResult result = evaluator.evaluate(evalCase, trace);
assertFalse(result.isPassed());
assertTrue(result.getFailedChecks().contains(
"prompt audit version not expected: old-prompts"));
assertTrue(result.getFailedChecks().contains(
"prompt version not expected: chat_executor=chat-executor-v1"));
}
@Test
void evaluateFailsWhenGatekeeperRulesAreMissing() {
DiagnosisEvalCase evalCase = DiagnosisEvalCase.builder()
.id("gatekeeper-rules")
.title("Gatekeeper rules")
.expectedRootCauseKeywords(List.of())
.requiredEvidenceTools(List.of())
.allowedVerdicts(List.of("PASS"))
.requireGatekeeperRules(true)
.build();
DiagnosisTraceResponse trace = DiagnosisTraceResponse.builder()
.session(DiagnosisTraceResponse.SessionTrace.builder()
.answer("安全回答")
.selfEvaluation(java.util.Map.of(
"verifier_evaluation", java.util.Map.of(
"verdict", "PASS",
"gatekeeper_result", java.util.Map.of(
"status", "pass",
"rule_set_version", "gatekeeper-rules-v1"
)
)))
.build())
.toolInvocations(List.of())
.build();
DiagnosisEvalResult result = evaluator.evaluate(evalCase, trace);
assertFalse(result.isPassed());
assertTrue(result.getFailedChecks().contains("gatekeeper_result missing rules"));
}
@Test
void evaluateFailsWhenUnsupportedClaimLeaksIntoFinalAnswer() {
@@ -394,10 +394,20 @@ class ChatServiceSequentialAgentTest {
verify(mergeService).mergeVerifierEvaluation(isNull(), captor.capture());
Map<String, Object> verifierEvaluation = captor.getValue();
assertTrue(verifierEvaluation.containsKey("gatekeeper_result"));
assertTrue(verifierEvaluation.containsKey("prompt_audit"));
@SuppressWarnings("unchecked")
Map<String, Object> gatekeeperResult = (Map<String, Object>) verifierEvaluation.get("gatekeeper_result");
assertEquals("pass", gatekeeperResult.get("status"));
assertEquals("none", gatekeeperResult.get("severity"));
@SuppressWarnings("unchecked")
Map<String, Object> promptAudit = (Map<String, Object>) verifierEvaluation.get("prompt_audit");
assertEquals("chat-prompts-v1", promptAudit.get("version"));
@SuppressWarnings("unchecked")
List<Map<String, Object>> prompts = (List<Map<String, Object>>) promptAudit.get("prompts");
assertEquals(4, prompts.size());
assertTrue(prompts.stream().anyMatch(prompt ->
"chat_executor".equals(prompt.get("name"))
&& "chat-executor-v2".equals(prompt.get("version"))));
}
@Test